{"id":"W4402671301","doi":"10.18653/v1/2024.acl-long.785","title":"CausalGym: Benchmarking causal interpretability methods on linguistic tasks","year":2024,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"University of Cambridge; Institute for Catastrophic Loss Reduction","keywords":"Interpretability; Benchmarking; Computer science; Natural language processing; Artificial intelligence; Linguistics; Philosophy","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001178763,0.0001616647,0.0001674726,0.0001285423,0.000069515,0.0003534064,0.0006236516,0.00006948369,0.0001442256],"category_scores_gemma":[0.0003505491,0.0001327025,0.00008633696,0.0002902995,0.00003031111,0.0001627766,0.0003527772,0.0003076043,0.00009425826],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000112968,"about_ca_system_score_gemma":0.00009996235,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001223094,"about_ca_topic_score_gemma":0.00001598211,"domain_scores_codex":[0.9982972,0.0001832101,0.0002890575,0.0007087833,0.0002271074,0.0002946965],"domain_scores_gemma":[0.9983483,0.0006784492,0.00002341003,0.0008011772,0.00004892668,0.00009974048],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000002419795,0.00002504671,0.0001193053,0.00006220901,0.00002716414,0.00006590001,0.001838475,0.0004959443,0.0006325925,0.4776755,0.0003378856,0.5187176],"study_design_scores_gemma":[0.00003790349,0.00006162748,0.0001344221,0.00008170913,0.000007592042,0.00001382026,0.000008623651,0.9630731,0.001324538,0.02666615,0.008417862,0.0001725922],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002833321,0.0002247594,0.9542888,0.0004697332,0.003432943,0.0001038244,8.544769e-7,0.0006349938,0.03801075],"genre_scores_gemma":[0.6198032,0.000002283423,0.3792241,0.0003257731,0.0002517164,0.000008526185,7.067906e-7,0.000008375401,0.0003753167],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9625772,"threshold_uncertainty_score":0.5411451,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03729222069651673,"score_gpt":0.3692485109035606,"score_spread":0.3319562902070439,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}