{"id":"W4417422982","doi":"10.1109/iccv51701.2025.00630","title":"Zero-AVSR: Zero-Shot Audio-Visual Speech Recognition with LLMs by Learning Language-Agnostic Speech Representations","year":2025,"lang":"en","type":"article","venue":"","topic":"Speech and Audio Processing","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Ministry of Science and ICT, South Korea; National Research Foundation of Korea","keywords":"Speech corpus; Language model; Training set; Speech synthesis; Speech processing; Speech analytics; Speech technology; Voice activity detection","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00112805,0.0008725338,0.000825964,0.0005611174,0.0002426741,0.0007552789,0.001722332,0.0009411004,0.003730946],"category_scores_gemma":[0.002700101,0.0003302607,0.0007551231,0.0003446003,0.0006867537,0.001433481,0.001990628,0.001325799,0.002345131],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003280049,"about_ca_system_score_gemma":0.0007267185,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002070715,"about_ca_topic_score_gemma":0.003413253,"domain_scores_codex":[0.9992012,0.000187569,0.00003446208,0.0003144661,0.0001791485,0.00008312314],"domain_scores_gemma":[0.999215,0.0003423801,0.0000553987,0.0001806825,0.0001511784,0.00005534701],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005505357,0.0002224781,0.0009898932,0.0002837034,0.0001116872,0.0002117218,0.0002086079,0.0760171,0.07355111,0.008233537,0.004669027,0.8349506],"study_design_scores_gemma":[0.00003787144,0.0002835508,0.0006743325,0.00002691735,0.00003892513,0.0002425854,0.00006911613,0.9437333,0.04196409,0.007980478,0.004911842,0.00003709034],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02365961,0.0004141651,0.9685835,0.0001500542,0.0001172254,0.0000716765,0.0002021155,0.004392019,0.002409694],"genre_scores_gemma":[0.5108364,0.0004450191,0.4744992,0.0005495166,0.0001627036,0.0002044197,0.001992619,0.0005851064,0.01072502],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003730946,"threshold_uncertainty_score":0.01248127,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01428954594590552,"score_gpt":0.2913103697040624,"score_spread":0.2770208237581568,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}