{"id":"W4414968375","doi":"10.48550/arxiv.2509.00072","title":"Test of Time: Rethinking Temporal Signal of Benchmark Contamination","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Semantic Web and Ontologies","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Bundesministerium für Bildung und Forschung; Government of Canada; Canadian Institute for Advanced Research","keywords":"Benchmark (surveying); Cutoff; Pipeline (software); Construct (python library); Measure (data warehouse); Scalability; Scale (ratio); Frontier","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01261508,0.001053303,0.000833091,0.001265067,0.0007693681,0.002485494,0.001520587,0.001591741,0.004049824],"category_scores_gemma":[0.1769036,0.0004270208,0.0005861016,0.00139308,0.002122499,0.003571715,0.003648924,0.003618686,0.001883572],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001457962,"about_ca_system_score_gemma":0.001080591,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002087064,"about_ca_topic_score_gemma":0.001853849,"domain_scores_codex":[0.989123,0.00463649,0.0006041289,0.001757007,0.003383436,0.0004959604],"domain_scores_gemma":[0.877147,0.08316228,0.005895764,0.02286925,0.009000294,0.001925495],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.006550577,0.00174268,0.1293515,0.001476578,0.0005118034,0.000832753,0.009591247,0.05930169,0.1895026,0.02513923,0.02521055,0.5507888],"study_design_scores_gemma":[0.0002999407,0.004430443,0.1670013,0.000281989,0.0002097976,0.0008943056,0.002393089,0.4838695,0.2560072,0.05609132,0.02817222,0.0003489708],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8281914,0.0007476967,0.1463095,0.001615054,0.0002958333,0.0002488879,0.002216421,0.008643084,0.0117322],"genre_scores_gemma":[0.9716368,0.00005324668,0.02329292,0.0003492376,0.00005290239,0.0001938673,0.001677166,0.00123245,0.001511332],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9873849,"threshold_uncertainty_score":0.06671572,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03438101978987016,"score_gpt":0.2691918863281281,"score_spread":0.234810866538258,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}