{"id":"W4416183485","doi":"10.1109/mipr67560.2025.00054","title":"Mitigating Image Captioning Hallucinations in Vision-Language Models","year":2025,"lang":"","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Closed captioning; Retraining; Normalization (sociology); Inference; Reliability (semiconductor); Adaptation (eye); Test data; Reduction (mathematics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003063197,0.001649974,0.001285465,0.0006704799,0.0005323151,0.001460403,0.002049033,0.001550648,0.001372982],"category_scores_gemma":[0.01447438,0.0006373823,0.001072607,0.0006424424,0.001057933,0.002786414,0.002802892,0.003011009,0.000857294],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008946741,"about_ca_system_score_gemma":0.001128945,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005198171,"about_ca_topic_score_gemma":0.006057022,"domain_scores_codex":[0.9982551,0.0005702623,0.00009333028,0.0004329612,0.0004682452,0.0001801673],"domain_scores_gemma":[0.994769,0.002331114,0.0004498716,0.001142664,0.001073413,0.0002339456],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009177374,0.0003289923,0.002876805,0.0003333241,0.0002852735,0.0004759515,0.0004422762,0.3381471,0.04701713,0.003912783,0.00719515,0.5980675],"study_design_scores_gemma":[0.00001710442,0.0001148358,0.00047356,0.00001119723,0.00003372852,0.0001379623,0.00003892327,0.9788167,0.01712686,0.002294713,0.0009136621,0.00002066292],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06794984,0.001000424,0.924809,0.0004689782,0.0001829043,0.0001092658,0.0001720716,0.003821183,0.00148625],"genre_scores_gemma":[0.7079877,0.0007985175,0.2831698,0.0009641578,0.0002232722,0.0001769889,0.001088634,0.000573929,0.005016962],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005198171,"threshold_uncertainty_score":0.01619995,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.009100288445900715,"score_gpt":0.3229183336735161,"score_spread":0.3138180452276154,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}