{"id":"W4390936974","doi":"10.2139/ssrn.4686376","title":"Backtest Overfitting in the Machine Learning Era: A Comparison of Out-of-Sample Testing Methods in a Synthetic Controlled Environment","year":2024,"lang":"en","type":"article","venue":"SSRN Electronic Journal","topic":"Market Dynamics and Volatility","field":"Economics, Econometrics and Finance","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto; York University","funders":"","keywords":"Overfitting; Sharpe ratio; Machine learning; Computer science; Artificial intelligence; Volatility (finance); Econometrics; Random walk; Robustness (evolution); Sample size determination; Statistic; Artificial neural network; Statistics; Economics; Mathematics; Finance","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02346775,0.001316766,0.001484012,0.0009993807,0.0006088121,0.001356233,0.002416373,0.0020119,0.001929111],"category_scores_gemma":[0.09287658,0.0003972567,0.001026408,0.0007483356,0.001577862,0.002457783,0.001725041,0.001931534,0.0003898404],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006748414,"about_ca_system_score_gemma":0.0009411819,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002622828,"about_ca_topic_score_gemma":0.00255468,"domain_scores_codex":[0.9900238,0.008004448,0.0004370181,0.0006973193,0.0006639033,0.000173577],"domain_scores_gemma":[0.8171,0.1655235,0.002840519,0.008629991,0.004949864,0.0009560658],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01137699,0.003168206,0.02948617,0.0009484955,0.00152618,0.0003293956,0.001107643,0.6220639,0.003900643,0.01130518,0.004561566,0.3102256],"study_design_scores_gemma":[0.0003645737,0.001455136,0.004795772,0.00005815028,0.0001203606,0.00008831094,0.0001257429,0.9839029,0.003157044,0.005206612,0.0006851421,0.00004024742],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.596746,0.001495042,0.3964066,0.0005138676,0.0002867392,0.0002435661,0.0003643828,0.001515686,0.002428036],"genre_scores_gemma":[0.9116086,0.0002629483,0.08527131,0.0001871561,0.00007340182,0.0001976591,0.0008393219,0.0005412585,0.001018314],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02346775,"threshold_uncertainty_score":0.1241108,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03729797008610252,"score_gpt":0.3000459057425797,"score_spread":0.2627479356564772,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}