{"id":"W4388484811","doi":"10.1016/j.tics.2023.10.005","title":"When expert predictions fail","year":2023,"lang":"en","type":"review","venue":"Trends in Cognitive Sciences","topic":"Philosophy and History of Science","field":"Arts and Humanities","cited_by":15,"is_retracted":false,"has_abstract":false,"ca_institutions":"Defence Research and Development Canada; The Scarborough Hospital; University of Toronto; University of Waterloo","funders":"Social Sciences and Humanities Research Council of Canada; John Templeton Foundation; National Science Foundation","keywords":"Humility; Psychology; Context (archaeology); Epistemology; Cognitive science; Data science; Social psychology; Computer science; Political science","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["sts","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0006214853,0.0003029642,0.0007022465,0.001489054,0.001078431,0.0002944211,0.0007167971,0.000087207,0.003946513],"category_scores_gemma":[0.00009487134,0.0002284817,0.0002703379,0.0006234358,0.004118674,0.0005637355,0.0001168376,0.0003169692,0.0006772718],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008282346,"about_ca_system_score_gemma":0.0002142992,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002178898,"about_ca_topic_score_gemma":0.001889742,"domain_scores_codex":[0.9978385,0.0001117929,0.0004399988,0.0007157839,0.0004873951,0.0004065075],"domain_scores_gemma":[0.9991272,0.0003847559,0.0002021636,0.0001386587,0.00006324753,0.00008393907],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[9.469593e-7,0.00003646403,3.566454e-7,0.0002245843,0.00001893191,0.00001344315,0.01761899,1.134423e-7,6.097323e-9,0.02802362,0.008246392,0.9458162],"study_design_scores_gemma":[0.00006184022,0.0001071697,0.000001175964,0.00531558,0.00006510237,0.000003606911,0.00255993,0.000009329606,2.47594e-8,0.005083242,0.9865066,0.0002863651],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[6.413894e-7,0.6479583,0.000002041987,0.0001196852,0.001150038,0.0001036968,0.0002113694,0.0001124895,0.3503417],"genre_scores_gemma":[0.0001363028,0.9654494,0.00001886578,0.0001002387,0.0009579363,0.000190089,0.00004652853,0.00001926512,0.03308138],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.9782602,"threshold_uncertainty_score":0.9985915,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6076804368050733,"score_gpt":0.4531422221793591,"score_spread":0.1545382146257142,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}