{"id":"W4392903137","doi":"10.1109/icassp48485.2024.10446627","title":"MOMA: Mixture-of-Modality-Adaptations for Transferring Knowledge from Image Models Towards Efficient Audio-Visual Action Recognition","year":2024,"lang":"en","type":"article","venue":"","topic":"Human Pose and Action Recognition","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Modality (human–computer interaction); Encoder; Adaptation (eye); Artificial intelligence; Modalities; Audio visual; Speech recognition; Image (mathematics); Visualization; Computer vision; Multimedia","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002558915,0.0001637985,0.0001779231,0.0002625978,0.000167299,0.0002389048,0.0001867456,0.0000993791,0.00009261124],"category_scores_gemma":[0.00001821487,0.0001558838,0.0002072253,0.0003655973,0.00003180029,0.0009777611,0.00003319758,0.0001400974,0.00009652664],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008226668,"about_ca_system_score_gemma":0.0001299103,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001200388,"about_ca_topic_score_gemma":0.00007604242,"domain_scores_codex":[0.9986945,0.00006245294,0.00036046,0.0004686008,0.000202599,0.0002113778],"domain_scores_gemma":[0.9992728,0.0001885937,0.00004557901,0.0001741879,0.0002415762,0.00007724978],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003536751,0.0004500902,7.007004e-7,0.0002413725,0.0001217211,0.000004244631,0.004068294,0.00118912,0.04929886,0.0253145,0.0007619002,0.9185138],"study_design_scores_gemma":[0.0003379766,0.00008672038,0.000102917,0.0001124312,0.00005171432,0.000004203095,0.0001488747,0.8287469,0.110695,0.05913751,0.0003797812,0.0001960122],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1185402,0.0001931911,0.8737866,0.0003987798,0.000904514,0.0003515651,0.00009728751,0.0004653347,0.005262539],"genre_scores_gemma":[0.9659027,0.00003026222,0.03320898,0.00004249075,0.000231459,0.0001154733,0.0001410119,0.00001801274,0.0003095933],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9183178,"threshold_uncertainty_score":0.6356757,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08406708936599298,"score_gpt":0.3390201958727369,"score_spread":0.2549531065067439,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}