{"id":"W4414527039","doi":"10.1007/978-3-031-96684-2_12","title":"Can ChatGPT Make Explanatory Inferences? Benchmarks for Abductive Reasoning","year":2025,"lang":"en","type":"book-chapter","venue":"Synthese Library/Synthese library","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Abductive reasoning; Generative grammar; Set (abstract data type); Inference; Verbal reasoning; Generative model; Non-monotonic logic; Visual reasoning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008044717,0.001213799,0.0009963356,0.002115446,0.001814186,0.006880528,0.004464364,0.003082479,0.04602358],"category_scores_gemma":[0.09972426,0.0008125744,0.001398075,0.003084572,0.003006737,0.01975877,0.004314309,0.004946805,0.008094726],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001634548,"about_ca_system_score_gemma":0.001938612,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003497476,"about_ca_topic_score_gemma":0.003640895,"domain_scores_codex":[0.9938118,0.00300659,0.0003224876,0.0007593406,0.001736459,0.0003633053],"domain_scores_gemma":[0.8934144,0.09048107,0.001065998,0.00981574,0.004213133,0.001009651],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009031141,0.0003363315,0.002721793,0.0009235263,0.0001018698,0.0004080978,0.001711202,0.03369568,0.001334178,0.5802274,0.05272393,0.3249129],"study_design_scores_gemma":[0.00008108804,0.0000475702,0.0005049511,0.0002974692,0.00005474513,0.0001390938,0.000832765,0.2012916,0.003590568,0.7686421,0.0244873,0.0000308791],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06366473,0.004250899,0.6857688,0.01397576,0.0009906974,0.000416286,0.004330358,0.009726008,0.2168765],"genre_scores_gemma":[0.5523938,0.002068941,0.4104116,0.0009593828,0.0005504985,0.0003948233,0.008618579,0.00261825,0.02198412],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.04602358,"threshold_uncertainty_score":0.1539642,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01453471825925627,"score_gpt":0.214162609239223,"score_spread":0.1996278909799667,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}