{"id":"W4385801313","doi":"10.1109/cvprw59228.2023.00496","title":"DeCAtt: Efficient Vision Transformers with Decorrelated Attention Heads","year":2023,"lang":"en","type":"article","venue":"","topic":"Advanced Neural Network Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Transformer; Overfitting; Artificial intelligence; Regularization (linguistics); Machine learning; Computer engineering; Engineering; Artificial neural network; Electrical engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008497038,0.001308195,0.0009620856,0.0006822736,0.0003861941,0.001042961,0.002637787,0.001283332,0.007776643],"category_scores_gemma":[0.00294882,0.0005588197,0.0006652155,0.0005591406,0.0006046137,0.002452812,0.002320718,0.002210906,0.002687797],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001028289,"about_ca_system_score_gemma":0.001660664,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006564949,"about_ca_topic_score_gemma":0.01173867,"domain_scores_codex":[0.9996407,0.00005793824,0.00001825543,0.0001057937,0.0001053972,0.00007196028],"domain_scores_gemma":[0.999441,0.0002079576,0.00003762895,0.0001374212,0.0001150428,0.00006088426],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007954578,0.0005022543,0.001386212,0.0002752919,0.0001419218,0.0002295377,0.0001261845,0.1192099,0.03082382,0.01796777,0.03670434,0.7918374],"study_design_scores_gemma":[0.00008396293,0.0001891102,0.0003113271,0.00001973565,0.0000317233,0.000122713,0.00002821518,0.9668316,0.01657454,0.01122971,0.004557393,0.00001990327],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04197453,0.001341228,0.9198303,0.0005865413,0.0003306111,0.0002594791,0.0005263271,0.02726989,0.007881234],"genre_scores_gemma":[0.6372589,0.000619097,0.3363863,0.001746582,0.000202361,0.0003655642,0.002237777,0.001349387,0.01983395],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007776643,"threshold_uncertainty_score":0.02601546,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.009890696637562682,"score_gpt":0.2666187037926013,"score_spread":0.2567280071550386,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}