{"id":"W4409796959","doi":"10.1109/citrex64975.2025.10974935","title":"Can LLMs Identify Event Causality More Accurately through Debate? A Systematic Assessment of LLMs’ Factuality and Reasoning","year":2025,"lang":"en","type":"article","venue":"","topic":"Law, Economics, and Judicial Systems","field":"Economics, Econometrics and Finance","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Concordia University","funders":"","keywords":"Causality (physics); Event (particle physics); Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01615534,0.001406289,0.001085298,0.006332131,0.001286991,0.005869419,0.001993373,0.003439418,0.008888641],"category_scores_gemma":[0.1806911,0.0005794062,0.001578831,0.002287231,0.002025507,0.01954358,0.00623345,0.002544637,0.002134358],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001528616,"about_ca_system_score_gemma":0.002333882,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001750258,"about_ca_topic_score_gemma":0.001818605,"domain_scores_codex":[0.9832403,0.006805154,0.002441094,0.003160468,0.003670319,0.0006826803],"domain_scores_gemma":[0.8905708,0.07602454,0.01697164,0.007574194,0.007118561,0.001740292],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002380329,0.0006453592,0.2050996,0.002653843,0.0005301383,0.001875562,0.02300906,0.01568709,0.02108529,0.0952318,0.01564214,0.6161598],"study_design_scores_gemma":[0.0004112526,0.0007444523,0.09729907,0.001669462,0.0008023776,0.002358065,0.01283008,0.3134044,0.03132845,0.453007,0.08553995,0.0006054461],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5275374,0.002842247,0.4190626,0.006961009,0.0003862554,0.001397215,0.0043946,0.006303783,0.03111493],"genre_scores_gemma":[0.8534988,0.0004382684,0.1391082,0.0007821578,0.0001292655,0.0004599205,0.003276124,0.0002891937,0.002018045],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01615534,"threshold_uncertainty_score":0.08543867,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06302887704037352,"score_gpt":0.3320769680520762,"score_spread":0.2690480910117027,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}