{"id":"W4415428234","doi":"10.3233/faia251306","title":"Assessing and Improving the Multilingual Visual Word Sense Disambiguation Ability of Vision-Language Models","year":2025,"lang":"en","type":"book-chapter","venue":"Frontiers in artificial intelligence and applications","topic":"Speech and dialogue systems","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Generalization; Task (project management); Set (abstract data type); Generative grammar; Lemma (botany); Word (group theory); Word-sense disambiguation; Generative model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002072222,0.001618738,0.0007025925,0.001040978,0.0003342843,0.002086612,0.001135161,0.001125966,0.003965335],"category_scores_gemma":[0.006695508,0.0004006147,0.0006107925,0.0006475658,0.0004505878,0.002651508,0.002109853,0.00139145,0.002214737],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006905142,"about_ca_system_score_gemma":0.0007818085,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00461965,"about_ca_topic_score_gemma":0.006114622,"domain_scores_codex":[0.9991842,0.0002838863,0.00003885003,0.0002971111,0.0001289979,0.00006693997],"domain_scores_gemma":[0.9976725,0.001615901,0.00008229356,0.0002644717,0.0002639025,0.0001009708],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005742293,0.0004616049,0.007959006,0.0004584027,0.0004185034,0.0002120114,0.0004330843,0.1217683,0.03492916,0.005356404,0.009307667,0.8181216],"study_design_scores_gemma":[0.0000439207,0.0003950824,0.002856172,0.00004475639,0.0001396112,0.0001957005,0.000305147,0.9546966,0.03080371,0.006733461,0.003718292,0.0000674365],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.5944492,0.006242536,0.3673971,0.0009267833,0.0004070521,0.0001719397,0.001217929,0.01145427,0.01773321],"genre_scores_gemma":[0.8560997,0.0009524884,0.1357298,0.000213043,0.00007362598,0.00007162993,0.002026003,0.0004576359,0.004376222],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.00461965,"threshold_uncertainty_score":0.01326543,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03518871009626741,"score_gpt":0.3228200114855175,"score_spread":0.2876313013892501,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}