{"id":"W4415967295","doi":"10.48550/arxiv.2510.18148","title":"Extracting Rule-based Descriptions of Attention Features in Transformers","year":2025,"lang":"","type":"preprint","venue":"ArXiv.org","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Interpretability; Security token; Transformer; Interpretation (philosophy); Word (group theory); Value (mathematics); Pattern recognition (psychology)","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006167517,0.0006646534,0.0003921246,0.001342656,0.0002779576,0.001657101,0.0008957445,0.0007721364,0.004148924],"category_scores_gemma":[0.006426376,0.0004913988,0.001137833,0.0008239764,0.0008407516,0.003291026,0.0008252807,0.001217417,0.001781856],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001186335,"about_ca_system_score_gemma":0.001179821,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003925851,"about_ca_topic_score_gemma":0.006522684,"domain_scores_codex":[0.9993597,0.00009192949,0.00006887131,0.0001521827,0.0002393964,0.00008803955],"domain_scores_gemma":[0.9980991,0.001037634,0.0001331923,0.0003213979,0.0003473122,0.00006143587],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005878313,0.0001948262,0.02834575,0.0009700426,0.0001422892,0.004160537,0.002169568,0.1794076,0.0896011,0.1881381,0.01729955,0.4889826],"study_design_scores_gemma":[0.00004735327,0.0001064213,0.003667754,0.0000725631,0.0001110426,0.0007380312,0.0003547338,0.7967904,0.04597184,0.1383207,0.01376539,0.00005387189],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1142753,0.0002168568,0.8603933,0.0003676643,0.00008901219,0.0002690002,0.005371481,0.009086823,0.009930444],"genre_scores_gemma":[0.7959797,0.0002147166,0.1940234,0.000171759,0.00003513944,0.000187162,0.004840289,0.001134215,0.003413615],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004148924,"threshold_uncertainty_score":0.01387954,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07306531986248128,"score_gpt":0.317219122067993,"score_spread":0.2441538022055117,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}