{"id":"W4415967295","doi":"10.48550/arxiv.2510.18148","title":"Extracting Rule-based Descriptions of Attention Features in Transformers","year":2025,"lang":"","type":"preprint","venue":"ArXiv.org","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Interpretability; Security token; Transformer; Interpretation (philosophy); Word (group theory); Value (mathematics); Pattern recognition (psychology)","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.001710222,0.0007130374,0.0009087251,0.00123067,0.0004422103,0.000255089,0.002065383,0.0007154457,0.0001485795],"category_scores_gemma":[0.0004987186,0.0008380156,0.0006775933,0.002114796,0.0003688339,0.00113687,0.0005043449,0.001926714,0.0001197715],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005496085,"about_ca_system_score_gemma":0.00127494,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003531802,"about_ca_topic_score_gemma":0.001877596,"domain_scores_codex":[0.9940549,0.0005064478,0.00194576,0.001683849,0.0007205498,0.001088532],"domain_scores_gemma":[0.9963424,0.0006314308,0.0008165148,0.001392909,0.0006123849,0.0002043259],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.000331441,0.002345877,0.641705,0.002486432,0.0002773008,0.0001573296,0.009980448,0.07214008,0.1085219,0.01129319,0.0001839858,0.150577],"study_design_scores_gemma":[0.0009098484,0.000289782,0.6143907,0.005947463,0.0002362059,0.00001320546,0.00439966,0.1015277,0.265218,0.004466878,0.0008810121,0.001719591],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6931932,0.0005503758,0.2974939,0.001894782,0.00227708,0.0009967221,0.00003060216,0.0001065255,0.003456811],"genre_scores_gemma":[0.9903468,0.0003019072,0.007293563,0.0002527836,0.0001101213,0.000159979,0.00003623969,0.0000305741,0.001468054],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2971536,"threshold_uncertainty_score":0.9994071,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07306531986248128,"score_gpt":0.317219122067993,"score_spread":0.2441538022055117,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}