{"id":"W4410720035","doi":"10.31234/osf.io/a7kdx_v1","title":"The Forecasting Proficiency Test: A General Use Assessment of Forecasting Ability","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Forecasting Techniques and Applications","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"Open Philanthropy Project","keywords":"Test (biology); Computer science; Econometrics; Economics; Geology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.009437447,0.0004087314,0.0005914827,0.0002621675,0.0005740614,0.001536588,0.002330298,0.0002769869,0.00008778739],"category_scores_gemma":[0.01277241,0.0002254641,0.0005239203,0.001159082,0.0004590937,0.0001264761,0.005503271,0.001178626,0.00001578179],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001751502,"about_ca_system_score_gemma":0.0007160179,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003702918,"about_ca_topic_score_gemma":0.0001587911,"domain_scores_codex":[0.993849,0.0001948226,0.002189429,0.001356008,0.001864254,0.0005464433],"domain_scores_gemma":[0.9860049,0.009006842,0.001194327,0.002437592,0.001222996,0.0001332865],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002180306,0.0007016475,0.1118458,0.0005614678,0.0001357525,0.00002410708,0.0009209422,0.01059928,0.002363767,0.3407589,0.04725937,0.4848072],"study_design_scores_gemma":[0.00003613686,0.00006709818,0.003269435,0.0001534219,0.00002949442,0.00001559688,0.00009756764,0.6537482,0.0003879307,0.3395282,0.002456032,0.000210899],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.843812,0.0002120027,0.09856307,0.002392236,0.001085152,0.003097701,0.0003728082,0.0006249952,0.04983997],"genre_scores_gemma":[0.8625247,0.00001013479,0.132554,0.00003518941,0.0001729522,0.0004496023,0.00001191015,0.00003717513,0.00420427],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6431488,"threshold_uncertainty_score":0.9994999,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3578384490180099,"score_gpt":0.4638868639794628,"score_spread":0.1060484149614528,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}