{"id":"W4403532992","doi":"10.1016/j.knosys.2024.112477","title":"Backtest overfitting in the machine learning era: A comparison of out-of-sample testing methods in a synthetic controlled environment","year":2024,"lang":"en","type":"article","venue":"Knowledge-Based Systems","topic":"Statistical Methods and Inference","field":"Mathematics","cited_by":8,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto; York University","funders":"","keywords":"Overfitting; Sample (material); Artificial intelligence; Computer science; Machine learning; Artificial neural network","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02696516,0.001309951,0.001602962,0.001000338,0.0007078265,0.001382537,0.002596628,0.002040024,0.001316025],"category_scores_gemma":[0.1107104,0.0004263946,0.001078303,0.0007656215,0.002117952,0.002420636,0.001903758,0.002032551,0.0002970944],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001051892,"about_ca_system_score_gemma":0.001197435,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004028544,"about_ca_topic_score_gemma":0.003896612,"domain_scores_codex":[0.9869211,0.01011842,0.0005266126,0.0009009773,0.001332947,0.0001999524],"domain_scores_gemma":[0.7871135,0.1917818,0.003476187,0.0103852,0.006133284,0.001110026],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01397209,0.003531418,0.02507092,0.001203908,0.001790821,0.0003006996,0.001738757,0.6075523,0.006155848,0.01545748,0.003758269,0.3194675],"study_design_scores_gemma":[0.0003983912,0.002444455,0.006957043,0.0000855972,0.0001931757,0.0001371078,0.0002130672,0.9715545,0.006612071,0.01015991,0.001165616,0.0000791047],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.511776,0.001748872,0.4817494,0.0004293217,0.000243503,0.0002648293,0.000288098,0.001182297,0.002317734],"genre_scores_gemma":[0.8740056,0.0003323772,0.1229012,0.0001931248,0.00005464465,0.000239558,0.0005897788,0.0005707112,0.001112965],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9730349,"threshold_uncertainty_score":0.1426071,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1717677044474834,"score_gpt":0.437279618389988,"score_spread":0.2655119139425046,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}