{"id":"W4388878581","doi":"10.1007/978-3-031-48312-7_2","title":"Audio-Visual Speaker Verification via Joint Cross-Attention","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Speech and Audio Processing","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Computer science; Leverage (statistics); Modal; Speech recognition; Audio visual; Concatenation (mathematics); Feature (linguistics); Artificial intelligence; Joint (building); Modalities; Pattern recognition (psychology); Multimedia","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.001102486,0.0005071888,0.0004540927,0.0008948914,0.0004351753,0.001394783,0.002192108,0.0003443002,0.00002084098],"category_scores_gemma":[0.0001183023,0.0004842075,0.0001557211,0.001106518,0.0006153929,0.001127106,0.001060727,0.0007389992,0.0004458483],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003277241,"about_ca_system_score_gemma":0.0004438348,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001767947,"about_ca_topic_score_gemma":0.00003664266,"domain_scores_codex":[0.9955379,0.0000279904,0.0006448338,0.001802912,0.001234856,0.0007515315],"domain_scores_gemma":[0.9977912,0.0001453448,0.0004279691,0.001120522,0.0003317763,0.0001832004],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000004784497,0.00003098363,0.0002970068,0.00006800801,0.00001211404,0.00008426596,0.0002690237,0.006199284,0.006071283,0.002147869,0.00003575004,0.9847797],"study_design_scores_gemma":[0.0008278619,0.000392002,0.01375648,0.001344419,0.0000243393,0.000280434,3.813629e-7,0.6173516,0.05916456,0.3026785,0.001840359,0.00233905],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0006952669,0.0001462528,0.992824,0.0006016315,0.003121604,0.00028573,0.000002812732,0.0004789902,0.001843696],"genre_scores_gemma":[0.3404886,0.0001118878,0.6492437,0.00249489,0.00205723,0.000033218,0.00005948017,0.0001692953,0.005341678],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9824406,"threshold_uncertainty_score":0.999761,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02622826901695002,"score_gpt":0.2813863992700888,"score_spread":0.2551581302531388,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}