{"id":"W4382599796","doi":"10.3389/fcomp.2023.1178450","title":"Self-attention in vision transformers performs perceptual grouping, not attention","year":2023,"lang":"en","type":"article","venue":"Frontiers in Computer Science","topic":"Visual Attention and Saliency Detection","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"Air Force Office of Scientific Research; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Artificial intelligence; Salience (neuroscience); Visual perception; Computer science; Visual attention; Computational model; Transformer; Perception; Salient; Cognitive psychology; Psychology; Neuroscience; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007604735,0.0005112772,0.0004387809,0.0004743103,0.0003192906,0.001202174,0.001029769,0.0007054771,0.002571953],"category_scores_gemma":[0.003192856,0.000302077,0.0006530639,0.0003245256,0.001335922,0.002996384,0.001339442,0.001061327,0.0004014609],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001067513,"about_ca_system_score_gemma":0.0005234967,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002309294,"about_ca_topic_score_gemma":0.001246597,"domain_scores_codex":[0.9995608,0.00008352103,0.00001537112,0.000167735,0.00009137081,0.0000811367],"domain_scores_gemma":[0.9989364,0.0004029319,0.0001535718,0.0002620531,0.0001380852,0.0001069004],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008304839,0.0002706686,0.01656578,0.000544487,0.0003178507,0.0004263453,0.001088343,0.1344207,0.3359085,0.2028216,0.003730254,0.303075],"study_design_scores_gemma":[0.00004409894,0.0004110077,0.01855419,0.00003331024,0.000105887,0.0003280612,0.0001835307,0.7269315,0.07665555,0.1730092,0.003693283,0.0000503975],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5276242,0.0008152491,0.4568492,0.0005190996,0.00008512445,0.00006177252,0.0001110543,0.001504241,0.01243009],"genre_scores_gemma":[0.9789687,0.0001648938,0.01944865,0.000070627,0.00001134534,0.00001649075,0.0000666634,0.00006374505,0.001188879],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.002571953,"threshold_uncertainty_score":0.00860399,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01226093722006374,"score_gpt":0.2660569458484697,"score_spread":0.253796008628406,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}