{"id":"W4417422982","doi":"10.1109/iccv51701.2025.00630","title":"Zero-AVSR: Zero-Shot Audio-Visual Speech Recognition with LLMs by Learning Language-Agnostic Speech Representations","year":2025,"lang":"en","type":"article","venue":"","topic":"Speech and Audio Processing","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Ministry of Science and ICT, South Korea; National Research Foundation of Korea","keywords":"Speech corpus; Language model; Training set; Speech synthesis; Speech processing; Speech analytics; Speech technology; Voice activity detection","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000315573,0.0002514273,0.0002473024,0.0002639034,0.000418322,0.0005414521,0.0005518799,0.0001027499,0.0001776289],"category_scores_gemma":[0.0003205544,0.0002193031,0.00006877624,0.001140106,0.00007790979,0.0009083195,0.0002163441,0.0003875041,0.0003017525],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006962042,"about_ca_system_score_gemma":0.0001742508,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002038244,"about_ca_topic_score_gemma":0.00005944606,"domain_scores_codex":[0.9979011,0.0001051028,0.0003332776,0.0007363146,0.0004356798,0.0004884725],"domain_scores_gemma":[0.9987867,0.0003202184,0.0001457519,0.0004152304,0.0002029112,0.0001292155],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00004371192,0.000283583,0.006546482,0.00009052422,0.000108493,0.0002968174,0.0009324644,0.0002246548,0.07219589,0.000266221,0.03974975,0.8792614],"study_design_scores_gemma":[0.001341095,0.0002784507,0.001193455,0.0003008985,0.00006176853,0.0001813685,0.0007513956,0.003467005,0.9859489,0.003380106,0.002486073,0.0006095277],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2795239,0.000325287,0.666555,0.001826209,0.0002654692,0.0003146152,0.000004470287,0.0008022176,0.05038283],"genre_scores_gemma":[0.7978309,0.00005288453,0.1721576,0.00139256,0.00008377648,0.00004222092,0.0001056464,0.00002973314,0.02830466],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.913753,"threshold_uncertainty_score":0.8942922,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01428954594590552,"score_gpt":0.2913103697040624,"score_spread":0.2770208237581568,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}