{"id":"W6948037944","doi":"10.48448/wqbv-ak02","title":"Categorical Syllogisms Revisited: A Review of the Logical Reasoning Abilities of LLMs for Analyzing Categorical Syllogisms","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Syllogism; Categorical variable; Inference; Interpretability; Interpretation (philosophy); Interpreter","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts"],"consensus_categories":[],"category_scores_codex":[0.00634585,0.0008829591,0.002246116,0.001115311,0.0002359504,0.0001309789,0.003242547,0.0005853277,0.0006998154],"category_scores_gemma":[0.006992338,0.0005380611,0.0008865512,0.005833428,0.006102615,0.000200515,0.001151735,0.001052917,0.0002864974],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007338827,"about_ca_system_score_gemma":0.001495317,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005778823,"about_ca_topic_score_gemma":0.00006805475,"domain_scores_codex":[0.9921647,0.0004132631,0.001960127,0.001891255,0.00233392,0.001236675],"domain_scores_gemma":[0.9942672,0.0006749316,0.001726161,0.002156143,0.0008574753,0.0003181015],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00006891029,0.0006258938,0.000630415,0.04128435,0.0004042698,0.00003968759,0.0004090329,0.0001197504,0.004692649,0.4288676,0.5179263,0.00493114],"study_design_scores_gemma":[0.002567995,0.00273294,0.0005585585,0.1351726,0.006839055,0.0007653272,0.001905714,0.02044166,0.002553785,0.1309953,0.6890867,0.006380258],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"review","genre_gemma":"empirical","genre_scores_codex":[0.0003959837,0.5829298,0.04730992,0.005332781,0.005480878,0.0150574,0.002966651,0.002488833,0.3380378],"genre_scores_gemma":[0.4757692,0.0502486,0.09777285,0.003254827,0.004637526,0.001583443,0.0009359679,0.006568171,0.3592294],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.5326812,"threshold_uncertainty_score":0.9997071,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03599498155418739,"score_gpt":0.3369193540830015,"score_spread":0.3009243725288141,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}