{"id":"W4416817077","doi":"10.1016/j.compbiomed.2025.111347","title":"Med-VCD: Mitigating hallucination for medical large vision language models through visual contrastive decoding","year":2025,"lang":"en","type":"article","venue":"Computers in Biology and Medicine","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Decoding methods; Context (archaeology); Redundancy (engineering); Hallucinating; Modalities; Language model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002507565,0.001440455,0.000899587,0.0008881658,0.000436931,0.001474064,0.002143882,0.001522409,0.003967533],"category_scores_gemma":[0.01446191,0.0005243716,0.0009972863,0.0005802684,0.00118083,0.001614486,0.003044472,0.002672095,0.00183538],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008953472,"about_ca_system_score_gemma":0.001638573,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004182747,"about_ca_topic_score_gemma":0.008472282,"domain_scores_codex":[0.9987251,0.0005483917,0.00008400623,0.0002768717,0.0002821707,0.00008341548],"domain_scores_gemma":[0.9946942,0.003977099,0.0002037543,0.0005208371,0.0004486974,0.0001554518],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0008447999,0.000213555,0.002483659,0.0006096654,0.0002103382,0.0004929698,0.0004003639,0.294466,0.02307531,0.01491653,0.02209013,0.6401966],"study_design_scores_gemma":[0.00005496648,0.00008480216,0.0001331809,0.00002486568,0.00002120469,0.0001246204,0.00003852834,0.9753804,0.0103817,0.01107142,0.00266517,0.00001920454],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01514168,0.0007394457,0.9743066,0.0007038564,0.0001510724,0.0001031116,0.0005409524,0.006901031,0.001412255],"genre_scores_gemma":[0.4228874,0.0005880341,0.5627951,0.001684122,0.0002939709,0.0003537512,0.00345963,0.001491368,0.006446707],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004182747,"threshold_uncertainty_score":0.0132727,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01191472068424883,"score_gpt":0.4048271390593892,"score_spread":0.3929124183751403,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}