{"id":"W4385572407","doi":"10.18653/v1/2023.findings-acl.148","title":"Hence, Socrates is mortal: A Benchmark for Natural Language Syllogistic Reasoning","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Syllogism; Computer science; Natural language processing; Artificial intelligence; Paraphrase; Natural language understanding; Natural language; Construct (python library); Deductive reasoning; Benchmark (surveying); Programming language; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003345989,0.003628771,0.001103354,0.007827924,0.002184547,0.00401151,0.005794766,0.003442183,0.01532907],"category_scores_gemma":[0.02173139,0.0009080998,0.003398237,0.006879834,0.001321233,0.007008891,0.00346155,0.004307416,0.01656236],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003249429,"about_ca_system_score_gemma":0.003636757,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02786563,"about_ca_topic_score_gemma":0.04784911,"domain_scores_codex":[0.9945649,0.001405829,0.0006848799,0.001448969,0.001416759,0.0004784991],"domain_scores_gemma":[0.9901949,0.003521736,0.0004700848,0.003011224,0.002100964,0.0007011598],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007430663,0.001052886,0.009752278,0.003642407,0.0003355424,0.000369652,0.0004706196,0.01489113,0.00288229,0.01236406,0.8520886,0.1014074],"study_design_scores_gemma":[0.0008158088,0.0005474666,0.0168926,0.001021187,0.0002454302,0.001074285,0.001075588,0.159666,0.01315717,0.02706667,0.7782291,0.0002086438],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.07268326,0.005490597,0.02497085,0.00235599,0.001401892,0.001178243,0.7742726,0.05930507,0.05834163],"genre_scores_gemma":[0.03312975,0.0005174252,0.03308482,0.0005087139,0.00008740115,0.0005360665,0.9274263,0.001326738,0.003382718],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02786563,"threshold_uncertainty_score":0.05540687,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02139751244422667,"score_gpt":0.2942475370270555,"score_spread":0.2728500245828288,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}