{"id":"W4412945576","doi":"10.18653/v1/2025.acl-long.407","title":"Data Laundering: Artificially Boosting Benchmark Results through Knowledge Distillation","year":2025,"lang":"en","type":"article","venue":"","topic":"Imbalanced Data Classification Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Boosting (machine learning); Benchmark (surveying); Computer science; Data mining; Machine learning; Artificial intelligence; Geology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.01728496,0.002280649,0.001246059,0.002298818,0.0009881373,0.003603127,0.003411749,0.001970692,0.002436655],"category_scores_gemma":[0.09827007,0.0005642919,0.0007641398,0.001595412,0.002607212,0.005657284,0.004779741,0.004629134,0.001681378],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001595942,"about_ca_system_score_gemma":0.002058589,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002852334,"about_ca_topic_score_gemma":0.00464248,"domain_scores_codex":[0.9869198,0.006551857,0.0006648643,0.001848371,0.003382589,0.0006325378],"domain_scores_gemma":[0.9551902,0.0246923,0.002695872,0.0112097,0.005084746,0.001127211],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00234635,0.0009423415,0.03368522,0.0009270835,0.0005664608,0.0003648386,0.001324332,0.2782035,0.02860151,0.03108157,0.02459513,0.5973616],"study_design_scores_gemma":[0.0001730017,0.0007896808,0.004143191,0.0001667798,0.0001029684,0.0001596043,0.0002406104,0.8968244,0.04179949,0.04579666,0.009703796,0.00009988123],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.41405,0.002514637,0.5355411,0.004674803,0.001038407,0.0005227997,0.001820514,0.02208129,0.01775652],"genre_scores_gemma":[0.8436599,0.000154348,0.1493704,0.001104486,0.0001262068,0.0002800776,0.0018935,0.001055095,0.002355859],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9980293,"threshold_uncertainty_score":0.09141272,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1078842696340384,"score_gpt":0.372544980241367,"score_spread":0.2646607106073286,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}