{"id":"W4392903137","doi":"10.1109/icassp48485.2024.10446627","title":"MOMA: Mixture-of-Modality-Adaptations for Transferring Knowledge from Image Models Towards Efficient Audio-Visual Action Recognition","year":2024,"lang":"en","type":"article","venue":"","topic":"Human Pose and Action Recognition","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Modality (human–computer interaction); Encoder; Adaptation (eye); Artificial intelligence; Modalities; Audio visual; Speech recognition; Image (mathematics); Visualization; Computer vision; Multimedia","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007780612,0.001395334,0.0008857572,0.0004764509,0.0002345465,0.0005738307,0.002386059,0.00107089,0.003880347],"category_scores_gemma":[0.002445345,0.0004918243,0.001081057,0.0004598693,0.0005967615,0.001607858,0.001930769,0.001952074,0.00257599],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003599018,"about_ca_system_score_gemma":0.0005195406,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002322411,"about_ca_topic_score_gemma":0.00346521,"domain_scores_codex":[0.9995303,0.00009129677,0.00001838828,0.0001978396,0.000101143,0.00006106801],"domain_scores_gemma":[0.9994468,0.0001708328,0.00003910934,0.0002273545,0.00007150861,0.00004440291],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0003495227,0.0002750814,0.001186306,0.0002076815,0.0002228732,0.0001702014,0.0001534913,0.1099841,0.09035918,0.003891372,0.005823421,0.7873768],"study_design_scores_gemma":[0.00001701509,0.0001416653,0.0007160155,0.00001746432,0.00004978197,0.0001756847,0.0000404124,0.9591007,0.02950043,0.005855521,0.004352985,0.00003242169],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01488812,0.0004720959,0.9766633,0.00009120203,0.00009815372,0.00006785682,0.0001372236,0.005963473,0.001618432],"genre_scores_gemma":[0.4986973,0.0005680276,0.4908528,0.0006134113,0.0001466267,0.0003320451,0.00101694,0.0009197108,0.006853237],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003880347,"threshold_uncertainty_score":0.01298106,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08406708936599298,"score_gpt":0.3390201958727369,"score_spread":0.2549531065067439,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}