{"id":"W1813435591","doi":"10.3982/ecta7163","title":"The Complexity of Forecast Testing","year":2008,"lang":"en","type":"article","venue":"Econometrica","topic":"Computability, Logic, AI Algorithms","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kellogg's (Canada)","funders":"","keywords":"Test (biology); Sequence (biology); Computer science; Forecast skill; Econometrics; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007481771,0.0009204451,0.001782782,0.00171243,0.002569548,0.006735384,0.00345731,0.002967723,0.01182538],"category_scores_gemma":[0.08195262,0.00102302,0.002467666,0.001980416,0.006023587,0.01603572,0.004682651,0.004635453,0.001144811],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00518608,"about_ca_system_score_gemma":0.004612892,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009103703,"about_ca_topic_score_gemma":0.004913072,"domain_scores_codex":[0.9885821,0.004834471,0.0007592501,0.0020069,0.002503434,0.00131389],"domain_scores_gemma":[0.8647214,0.1191083,0.00388471,0.00745565,0.003324324,0.001505831],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001085212,0.0002774162,0.0113336,0.0006495655,0.0003098348,0.0007644637,0.001297776,0.1964427,0.001564016,0.6947117,0.01381565,0.07774816],"study_design_scores_gemma":[0.00009292697,0.0000361213,0.001135753,0.00003751697,0.00004709336,0.0001428879,0.0001775371,0.1993343,0.0007005094,0.7955489,0.002714106,0.000032368],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3588771,0.002114074,0.5511268,0.03229235,0.0003459889,0.0005729114,0.004688984,0.00194022,0.0480416],"genre_scores_gemma":[0.8933648,0.0009637026,0.09409442,0.001134891,0.0003826607,0.0004327441,0.002798361,0.0003010227,0.006527383],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01182538,"threshold_uncertainty_score":0.03956783,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1934291853155568,"score_gpt":0.2551249838271196,"score_spread":0.06169579851156284,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}