{"id":"W4408355320","doi":"10.1109/icassp49660.2025.10888977","title":"LAVViT: Latent Audio-Visual Vision Transformers for Speaker Verification","year":2025,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Computer Research Institute of Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Audio visual; Transformer; Speech recognition; Speaker verification; Computer vision; Artificial intelligence; Speaker recognition; Multimedia; Engineering; Electrical engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001488533,0.00138523,0.0007398521,0.0008062372,0.0003968893,0.00106341,0.002557722,0.001082358,0.01547162],"category_scores_gemma":[0.00400746,0.0006280728,0.0009440108,0.0004421146,0.0006348324,0.002386557,0.002771659,0.002152217,0.008137542],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007728655,"about_ca_system_score_gemma":0.001135374,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005236831,"about_ca_topic_score_gemma":0.009830005,"domain_scores_codex":[0.9993773,0.0001307572,0.00002575975,0.0002233795,0.0001542399,0.0000885166],"domain_scores_gemma":[0.9994042,0.0001905466,0.00003647781,0.0001888515,0.0001221132,0.00005786474],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001098008,0.0002623433,0.00242952,0.0003720247,0.0002350608,0.0001914533,0.0001720258,0.05038613,0.0633001,0.01256251,0.04073308,0.8282577],"study_design_scores_gemma":[0.0001291281,0.0002779859,0.001157261,0.00005430549,0.00006615249,0.0002880551,0.00006358414,0.9076065,0.05694883,0.016671,0.01668743,0.00004976129],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02068306,0.0008450643,0.921994,0.0002469615,0.0003081307,0.0002303507,0.001731668,0.04958403,0.004376817],"genre_scores_gemma":[0.5247533,0.0004900699,0.4441499,0.0006346804,0.0001666978,0.0003992966,0.01210782,0.001884599,0.01541366],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01547162,"threshold_uncertainty_score":0.05175769,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01425759948865119,"score_gpt":0.30255972333916,"score_spread":0.2883021238505089,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}