{"id":"W4412975185","doi":"10.31234/osf.io/a7kdx_v6","title":"The Forecasting Proficiency Test: A General Use Assessment of Forecasting Ability","year":2025,"lang":"en","type":"article","venue":"","topic":"Forecasting Techniques and Applications","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Variance (accounting); Probabilistic logic; Computer science; Test (biology); Variety (cybernetics); Probabilistic forecasting; Machine learning; Econometrics; Artificial intelligence; Bayesian probability; Economics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.004953165,0.0001371655,0.0002290129,0.0001400449,0.0006930174,0.0003640062,0.001000291,0.00005979775,0.00004604794],"category_scores_gemma":[0.0127693,0.00007534223,0.0001492907,0.001576272,0.0002796258,0.0002128933,0.0004352252,0.0001704889,0.000002788643],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006227122,"about_ca_system_score_gemma":0.0002279146,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001375533,"about_ca_topic_score_gemma":0.0000995493,"domain_scores_codex":[0.997339,0.0001037546,0.001043872,0.0004666061,0.0007244607,0.0003222872],"domain_scores_gemma":[0.9895958,0.008178814,0.0004074445,0.001022429,0.0007412335,0.00005423964],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000007850862,0.0002178736,0.431245,0.00001610326,0.00001255692,0.000001007897,0.00008940179,0.0005199999,0.002957536,0.3230555,0.01087242,0.2310047],"study_design_scores_gemma":[0.0001251637,0.00009137514,0.06710459,0.00004149764,0.00001167044,0.000006034002,0.0002097139,0.7966484,0.002361602,0.124086,0.009187596,0.0001263773],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6670575,0.00002327701,0.2685979,0.001455729,0.0001479371,0.0008361345,0.00001966122,0.0001442764,0.06171755],"genre_scores_gemma":[0.9003326,0.000002102004,0.09412196,0.00008200354,0.00002344159,0.0001081717,0.000001303445,0.000006089522,0.005322278],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7961283,"threshold_uncertainty_score":0.9955466,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2557744791340968,"score_gpt":0.4548820474820007,"score_spread":0.1991075683479039,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}