{"id":"W4402904188","doi":"10.1109/cvprw63382.2024.00190","title":"Towards Efficient Audio-Visual Learners via Empowering Pre-trained Vision Transformers with Cross-Modal Adaptation","year":2024,"lang":"en","type":"article","venue":"","topic":"Speech and Audio Processing","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Audio visual; Transformer; Adaptation (eye); Modal; Speech recognition; Artificial intelligence; Domain adaptation; Computer vision; Multimedia; Engineering; Psychology; Electrical engineering; Voltage","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001089124,0.001360724,0.0006317089,0.000443356,0.0002282704,0.0008965006,0.001985357,0.001053385,0.005582869],"category_scores_gemma":[0.003686361,0.0004529687,0.0007667131,0.0003581516,0.0005949892,0.002317994,0.002510848,0.002349626,0.003222026],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004274877,"about_ca_system_score_gemma":0.0005780369,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002540505,"about_ca_topic_score_gemma":0.003328786,"domain_scores_codex":[0.9995004,0.0001063079,0.00002008317,0.0002124313,0.00009775374,0.00006299387],"domain_scores_gemma":[0.9992505,0.0003109637,0.00003919993,0.0001763998,0.000160307,0.00006259628],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004683473,0.0003295263,0.001784895,0.0003117136,0.000175542,0.0002034613,0.0003129924,0.1155501,0.1109753,0.007965547,0.007901987,0.7540205],"study_design_scores_gemma":[0.00002801364,0.0001495606,0.0005215106,0.00002616258,0.00004727629,0.0001482515,0.00009615984,0.957059,0.0273378,0.009969701,0.00459201,0.00002445349],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02211924,0.0005755752,0.9671813,0.0001326548,0.0001036939,0.00009198512,0.000149243,0.00652057,0.003125759],"genre_scores_gemma":[0.6100305,0.0005558142,0.3787462,0.000672762,0.0000969114,0.0003127462,0.001098208,0.000942238,0.007544503],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005582869,"threshold_uncertainty_score":0.01867658,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.008765059562367266,"score_gpt":0.3076469750364574,"score_spread":0.2988819154740902,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}