{"id":"W4417091950","doi":"10.48550/arxiv.2504.13151","title":"MIB: A Mechanistic Interpretability Benchmark","year":2025,"lang":"en","type":"preprint","venue":"UvA-DARE (University of Amsterdam)","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Azrieli Foundation; European Commission; Open Philanthropy Project","keywords":"Interpretability; Benchmark (surveying); Task (project management); Variable (mathematics); Causal model; Artificial neural network; Task analysis","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03831223,0.003563829,0.002254481,0.006589959,0.001722436,0.007730949,0.005067684,0.006277566,0.005874184],"category_scores_gemma":[0.1496385,0.0008714645,0.002466562,0.003803004,0.004056177,0.01129231,0.005733562,0.006517576,0.001917551],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003944631,"about_ca_system_score_gemma":0.002460825,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00308858,"about_ca_topic_score_gemma":0.003707013,"domain_scores_codex":[0.9706024,0.0147345,0.001963293,0.003259351,0.008348599,0.001091871],"domain_scores_gemma":[0.8920966,0.06966,0.006736608,0.01872578,0.01032415,0.002456878],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004171191,0.002041114,0.0237817,0.005044351,0.002025734,0.0003796551,0.0008584265,0.2512579,0.01824193,0.1934592,0.06665543,0.4320833],"study_design_scores_gemma":[0.0003936707,0.002010194,0.008253467,0.000785368,0.0002871682,0.000408079,0.0004915947,0.6624835,0.01809853,0.2814422,0.02512565,0.0002205595],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.113983,0.02025158,0.7896214,0.01458406,0.001691645,0.001101888,0.008000421,0.01018353,0.04058261],"genre_scores_gemma":[0.5448306,0.001930175,0.4283175,0.002278455,0.000625794,0.001591665,0.01383243,0.002223834,0.004369632],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.03831223,"threshold_uncertainty_score":0.2026169,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01612328950733706,"score_gpt":0.2312647162251226,"score_spread":0.2151414267177855,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}