{"id":"W4389519056","doi":"10.18653/v1/2023.emnlp-main.435","title":"A Mechanistic Interpretation of Arithmetic Reasoning in Language Models using Causal Mediation Analysis","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Azrieli Foundation; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; Open Philanthropy Project; National Science Foundation","keywords":"Computer science; Mediation; Security token; Causal reasoning; Set (abstract data type); Interpretation (philosophy); Theoretical computer science; Natural language processing; Process (computing); Language model; Causal model; Artificial intelligence; Arithmetic; Cognition; Programming language; Mathematics; Psychology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002025628,0.0006397271,0.000535,0.0008447616,0.0005240753,0.002281077,0.001952321,0.001050647,0.008795614],"category_scores_gemma":[0.01085735,0.0006635307,0.001397729,0.0004684576,0.002081882,0.004845805,0.002185926,0.00222862,0.0005516761],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001072166,"about_ca_system_score_gemma":0.0009928179,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002609194,"about_ca_topic_score_gemma":0.00180276,"domain_scores_codex":[0.9990589,0.000470588,0.00004453068,0.0001869714,0.0001413248,0.00009763233],"domain_scores_gemma":[0.9964634,0.002344323,0.0003432663,0.0004712,0.000232536,0.000145231],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003811761,0.0002045721,0.004385484,0.0002885985,0.0002755795,0.0007067925,0.00232312,0.1869544,0.03133839,0.7162112,0.001978823,0.05495184],"study_design_scores_gemma":[0.00003218377,0.00006736746,0.000815667,0.00001789631,0.00005396098,0.0001066712,0.0001339669,0.5633747,0.003778987,0.4307334,0.0008526081,0.0000325728],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07136711,0.000134598,0.9171103,0.00172046,0.00005151473,0.000103774,0.0001932649,0.0009502796,0.008368628],"genre_scores_gemma":[0.9096063,0.0001281804,0.08795433,0.0002084418,0.00003638645,0.0001396028,0.0001570903,0.0001037971,0.001665711],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008795614,"threshold_uncertainty_score":0.02942425,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02509132893953309,"score_gpt":0.2908553248754486,"score_spread":0.2657639959359155,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}