{"id":"W4406134357","doi":"10.1109/iccv51701.2025.00156","title":"AVTrustBench: Assessing and Enhancing Reliability and Robustness in Audio-Visual LLMs","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Digital Rights Management and Security","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; Kootenay Association for Science & Technology; University of Toronto","funders":"","keywords":"Robustness (evolution); Audio visual; Reliability (semiconductor); Computer science; Reliability engineering; Multimedia; Engineering; Chemistry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003959345,0.001630478,0.0007168821,0.0008110483,0.0004322479,0.001484839,0.002043144,0.001981202,0.005139338],"category_scores_gemma":[0.02653905,0.0004089378,0.0007111467,0.0003347254,0.001016949,0.002208544,0.002471822,0.002054867,0.001769798],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001099137,"about_ca_system_score_gemma":0.001323262,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005034809,"about_ca_topic_score_gemma":0.005614821,"domain_scores_codex":[0.9973215,0.001015024,0.0001786988,0.0006216251,0.0006266093,0.0002366113],"domain_scores_gemma":[0.990132,0.006907572,0.0005127334,0.001320524,0.0008205417,0.0003065886],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002197338,0.0005663984,0.006864656,0.001165732,0.0002373228,0.0003635041,0.00054143,0.5945134,0.03385282,0.003923108,0.01336545,0.3424089],"study_design_scores_gemma":[0.00008312735,0.0004188565,0.00110503,0.00006399242,0.00002922509,0.0001045403,0.0001170169,0.9764663,0.01482276,0.004700719,0.002055689,0.00003275621],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3979132,0.003957215,0.5442811,0.001310052,0.0005615275,0.0004803847,0.00204853,0.03520031,0.01424765],"genre_scores_gemma":[0.8835518,0.0003079223,0.1090206,0.0003831449,0.00005513885,0.0002308095,0.002136546,0.0007934656,0.003520614],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005139338,"threshold_uncertainty_score":0.02093923,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01555325027566546,"score_gpt":0.2816976066732604,"score_spread":0.266144356397595,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}