{"id":"W4416035479","doi":"10.18653/v1/2025.emnlp-main.1563","title":"Are Language Models Consequentialist or Deontological Moral Reasoners?","year":2025,"lang":"","type":"article","venue":"","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Bundesministerium für Bildung und Forschung; Natural Sciences and Engineering Research Council of Canada; Government of Canada","keywords":"Natural language; Natural (archaeology); Empirical research; Consequentialism; On Language","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008684142,0.0006218763,0.001083457,0.001497876,0.001290851,0.006169828,0.003230975,0.003452948,0.0129861],"category_scores_gemma":[0.03275435,0.0008419721,0.001518531,0.0008945517,0.009753097,0.01889047,0.00342954,0.004687826,0.002841795],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001748114,"about_ca_system_score_gemma":0.001608417,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001886241,"about_ca_topic_score_gemma":0.001852017,"domain_scores_codex":[0.9955921,0.002003936,0.0001948456,0.000835372,0.0009525829,0.000421195],"domain_scores_gemma":[0.9794887,0.01150575,0.002075238,0.004109108,0.001982483,0.0008386247],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005242039,0.00003675563,0.0006073967,0.00007995031,0.00004254386,0.00004570257,0.0009301436,0.0009900643,0.0001309271,0.98105,0.005668275,0.01036583],"study_design_scores_gemma":[0.00001885088,0.000003319122,0.0001434807,0.00002117904,0.00001097903,0.00002485567,0.0001450985,0.002690726,0.00006678566,0.9933524,0.003515325,0.000006989662],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.08711084,0.006383792,0.4415546,0.2016485,0.001748795,0.0001345489,0.001566327,0.001589935,0.2582626],"genre_scores_gemma":[0.926235,0.001707553,0.0474022,0.007579235,0.00103422,0.0002046807,0.0008055402,0.0003711605,0.01466034],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0129861,"threshold_uncertainty_score":0.04592669,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1066088690734464,"score_gpt":0.3441283254344819,"score_spread":0.2375194563610355,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}