{"id":"W3124201545","doi":"","title":"Evaluating Strategic Forecasters","year":2017,"lang":"en","type":"preprint","venue":"RePEc: Research Papers in Economics","topic":"Auction Theory and Applications","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Social Sciences and Humanities Research Council of Canada; Rice University; Indian Statistical Institute; National Science Foundation","keywords":"Generality; Robustness (evolution); Principal (computer security); Computer science; Quality (philosophy); Mechanism design; Simple (philosophy); Perception; Prediction market; Event (particle physics); Principal–agent problem; Econometrics; Microeconomics; Operations research; Economics; Computer security; Mathematics; Psychology; Finance","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.01775415,0.0002499546,0.0005146782,0.0006478641,0.0007353632,0.001200083,0.002975246,0.0003858534,0.0006963753],"category_scores_gemma":[0.004240682,0.0002296362,0.0002384085,0.000157355,0.0006566014,0.0002169811,0.001872806,0.001670941,0.0002881241],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004024488,"about_ca_system_score_gemma":0.0008831582,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000299726,"about_ca_topic_score_gemma":0.0001481097,"domain_scores_codex":[0.9951864,0.0008066086,0.001061324,0.001392273,0.0009246711,0.0006287115],"domain_scores_gemma":[0.9931219,0.002465786,0.0007035743,0.003113581,0.0003777882,0.0002174279],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00006979829,0.00007893987,0.001150994,0.00002021654,0.0000355076,0.000008364811,0.0004111843,0.03545476,0.000118325,0.01495671,0.0001279488,0.9475673],"study_design_scores_gemma":[0.0003939323,0.0000802959,0.001726654,0.0001009112,0.000006698038,0.0000111702,0.003518512,0.08146178,0.0001162072,0.8973373,0.01485664,0.0003899399],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5909942,0.00004819466,0.00004153215,0.0008453514,0.0006014116,0.0007089038,0.00005775053,0.00002929532,0.4066733],"genre_scores_gemma":[0.9810587,0.0005217069,0.001121741,0.00004419025,0.0003364222,0.0003467371,0.00002273475,0.00003362762,0.01651416],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9471773,"threshold_uncertainty_score":0.9998367,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.476135220608249,"score_gpt":0.5345493742352441,"score_spread":0.05841415362699504,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}