{"id":"W4401201671","doi":"10.48550/arxiv.2407.19394","title":"Depth-Wise Convolutions in Vision Transformers for Efficient Training on Small Datasets","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Advanced Neural Network Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Transformer; Training (meteorology); Computer science; Artificial intelligence; Computer vision; Pattern recognition (psychology); Engineering; Geography; Electrical engineering; Voltage","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006905569,0.001244366,0.0007035241,0.0005784564,0.0003297501,0.0009108789,0.002320021,0.0008941282,0.01139876],"category_scores_gemma":[0.002883806,0.0007283234,0.0009683324,0.0007584727,0.0006379602,0.002594122,0.001717808,0.002241455,0.00396955],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001094278,"about_ca_system_score_gemma":0.001440994,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006735649,"about_ca_topic_score_gemma":0.01202161,"domain_scores_codex":[0.999638,0.00005088011,0.00002843983,0.0001204075,0.00009361024,0.00006855572],"domain_scores_gemma":[0.9995599,0.0001399607,0.00003575328,0.0001575299,0.00007469426,0.00003215055],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004397257,0.0002147919,0.002342436,0.0004123708,0.0001831788,0.00026422,0.0001875209,0.2523465,0.04142914,0.03247225,0.02247907,0.6472288],"study_design_scores_gemma":[0.00002940613,0.00007629182,0.0003308913,0.000015618,0.00002462971,0.00012832,0.00002082272,0.9670145,0.01370816,0.01397464,0.004664475,0.00001228359],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02193129,0.0005079239,0.9612579,0.0002611255,0.00008266936,0.0001397563,0.0005610232,0.01140674,0.00385162],"genre_scores_gemma":[0.4530032,0.000649872,0.5331327,0.0005127852,0.00007401411,0.0004479979,0.002793313,0.001230064,0.00815607],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01139876,"threshold_uncertainty_score":0.03813267,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1135221462489228,"score_gpt":0.2439901235274572,"score_spread":0.1304679772785344,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}