{"id":"W4415214274","doi":"10.1016/j.neucom.2025.131217","title":"Sycophancy in vision-language models: A systematic analysis and an inference-time mitigation framework","year":2025,"lang":"en","type":"article","venue":"Neurocomputing","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"National Key Research and Development Program of China; Youth Innovation Promotion Association of the Chinese Academy of Sciences; Chinese Academy of Sciences; National Natural Science Foundation of China; Canadian Anesthesiologists' Society","keywords":"Sensitivity (control systems); Inference; Filter (signal processing); Trustworthiness; Decoding methods; Polarity (international relations); Path (computing); Mechanism (biology)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01088533,0.001946107,0.002150865,0.001745179,0.001453671,0.002627781,0.003300519,0.00271826,0.003820783],"category_scores_gemma":[0.04523825,0.001704502,0.00217191,0.001203401,0.003345971,0.007105169,0.005416322,0.006620509,0.0006573315],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001981311,"about_ca_system_score_gemma":0.004188988,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00616896,"about_ca_topic_score_gemma":0.006326125,"domain_scores_codex":[0.9944602,0.00273133,0.0002706788,0.001088692,0.001051495,0.0003975341],"domain_scores_gemma":[0.9586277,0.03494882,0.001645561,0.002561527,0.001851245,0.0003650526],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004460052,0.00023335,0.004473268,0.0008500931,0.0005282739,0.00102826,0.0006310016,0.3122064,0.006228454,0.5291067,0.007836946,0.1364312],"study_design_scores_gemma":[0.00001621433,0.00007386543,0.0004739956,0.00006848918,0.00009862291,0.0001551222,0.0000494948,0.8528903,0.00205637,0.1425247,0.001564608,0.0000282192],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.009165446,0.001306526,0.9861691,0.001252247,0.00008694299,0.00007052525,0.0001024242,0.0002887919,0.001558015],"genre_scores_gemma":[0.687545,0.004856156,0.2886508,0.001455902,0.001250702,0.0005889939,0.0007822866,0.0008493123,0.01402077],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01088533,"threshold_uncertainty_score":0.05756783,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.007109514855624758,"score_gpt":0.3227593656927562,"score_spread":0.3156498508371314,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}