{"id":"W2015688842","doi":"10.1080/09528130903010295","title":"Warning: statistical benchmarking is addictive. Kicking the habit in machine learning","year":2009,"lang":"en","type":"article","venue":"Journal of Experimental & Theoretical Artificial Intelligence","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa; National Research Council Canada","funders":"","keywords":"Benchmarking; Computer science; Addiction; Set (abstract data type); Habit; Machine learning; Test (biology); Artificial intelligence; Measure (data warehouse); Psychology; Data mining; Social psychology; Psychiatry; Management","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.06696237,0.001695062,0.002446139,0.003052777,0.003327424,0.007971239,0.004899037,0.02612318,0.007574204],"category_scores_gemma":[0.3572686,0.001240535,0.001941999,0.003667991,0.01419215,0.01030326,0.004308315,0.05518869,0.01285384],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002481974,"about_ca_system_score_gemma":0.006002941,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002862857,"about_ca_topic_score_gemma":0.003633377,"domain_scores_codex":[0.9294394,0.03009094,0.006752445,0.003888368,0.02873403,0.001094685],"domain_scores_gemma":[0.6040722,0.2607274,0.01688552,0.02238892,0.08750878,0.008417167],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00007235425,0.00004498194,0.0006080634,0.0002154519,0.00003790494,0.0000795978,0.0002281417,0.0001352351,0.0002705717,0.01286001,0.9664,0.01904768],"study_design_scores_gemma":[0.0002129043,0.0002910793,0.002789437,0.001566512,0.00006700354,0.0009387475,0.0005480668,0.002975229,0.001249909,0.1034965,0.8855982,0.0002664354],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"commentary","genre_scores_codex":[0.0004438136,0.002498128,0.008406499,0.938508,0.04661938,0.00004849581,0.0001730174,0.0008142595,0.002488388],"genre_scores_gemma":[0.005351583,0.001155965,0.01026657,0.9473751,0.02770402,0.0001446082,0.0001124278,0.0004524738,0.007437353],"genre_candidate":"commentary","genre_consensus":"commentary","teacher_disagreement_score":0.9330376,"threshold_uncertainty_score":0.3541351,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02424670070729992,"score_gpt":0.3166015831447657,"score_spread":0.2923548824374658,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}