{"id":"W4393337821","doi":"10.2139/ssrn.4778909","title":"Backtest Overfitting in the Machine Learning Era a Comparison of Out-of-Sample Testing Methods in a Synthetic Controlled Environment","year":2024,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Fault Detection and Control Systems","field":"Engineering","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto; York University","funders":"","keywords":"Overfitting; Sample (material); Artificial intelligence; Machine learning; Computer science; Econometrics; Mathematics; Artificial neural network; Chemistry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.009464702,0.001335076,0.00159148,0.001064743,0.0004309624,0.001024906,0.001492168,0.001507917,0.0008682171],"category_scores_gemma":[0.04134405,0.0003240476,0.000768156,0.0007147303,0.001129683,0.001695934,0.001026804,0.001241421,0.0002442384],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006284749,"about_ca_system_score_gemma":0.0005957366,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003191268,"about_ca_topic_score_gemma":0.00317687,"domain_scores_codex":[0.9957137,0.002801541,0.0002417376,0.0004452989,0.0006490243,0.0001486745],"domain_scores_gemma":[0.9326548,0.05753746,0.001394995,0.00409181,0.003620652,0.0007003647],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004091489,0.001064708,0.01592572,0.0005945819,0.0007456447,0.0002448563,0.0004255861,0.7227765,0.005755424,0.002935321,0.0025261,0.242914],"study_design_scores_gemma":[0.0001000559,0.0008966346,0.005811561,0.00003345574,0.00008370053,0.0001142234,0.00007312001,0.9859269,0.004293321,0.002208697,0.0004287576,0.0000296911],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7887748,0.002715176,0.2047095,0.0003695409,0.0001987045,0.00006486829,0.0002231,0.001233843,0.001710529],"genre_scores_gemma":[0.9649879,0.0002305966,0.03307861,0.00009515557,0.00004114227,0.00003646035,0.0004087288,0.0002877974,0.0008334933],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9905353,"threshold_uncertainty_score":0.05005467,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02198588307187619,"score_gpt":0.3022069790022538,"score_spread":0.2802210959303776,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}