{"id":"W7116210972","doi":"","title":"From Isolation to Entanglement: When Do Interpretability Methods Identify and Disentangle Known Concepts?","year":2025,"lang":"","type":"article","venue":"arXiv (Cornell University)","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"International Max Planck Research School for Advanced Methods in Process and Systems Engineering; International Max Planck Research School for Environmental, Cellular and Molecular Microbiology; McGill University","keywords":"Interpretability; Disjoint sets; Independence (probability theory); Linear subspace; Affect (linguistics); Feature (linguistics); Isolation (microbiology)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02787208,0.002693699,0.001960749,0.003378462,0.001289377,0.006685731,0.001792576,0.00290392,0.003827577],"category_scores_gemma":[0.1564567,0.001092033,0.001503372,0.001875368,0.008097594,0.02725491,0.008399798,0.006989874,0.0009173414],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001682812,"about_ca_system_score_gemma":0.001057753,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001167025,"about_ca_topic_score_gemma":0.0009995978,"domain_scores_codex":[0.9832993,0.009336964,0.0008195946,0.003136012,0.002791899,0.0006162452],"domain_scores_gemma":[0.8949164,0.0781645,0.008455736,0.01277935,0.003962364,0.001721659],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002731926,0.0005034946,0.05408532,0.0013441,0.001452836,0.00040775,0.007165468,0.05819477,0.01958913,0.2106999,0.005509888,0.6383155],"study_design_scores_gemma":[0.0001079021,0.0005376827,0.01501875,0.0004031242,0.0002484599,0.0002776663,0.00172436,0.2358188,0.01208813,0.7275392,0.006032282,0.0002035977],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1397889,0.00230491,0.8437737,0.004123506,0.0002094159,0.0002083655,0.0003313951,0.0008114472,0.008448316],"genre_scores_gemma":[0.8564627,0.0006803205,0.1389331,0.0008067461,0.0002670741,0.0003061297,0.0004860453,0.0006230004,0.001434934],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02787208,"threshold_uncertainty_score":0.1474034,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06445692341868224,"score_gpt":0.2979039952585593,"score_spread":0.2334470718398771,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}