{"id":"W4405140008","doi":"10.1007/978-981-96-0917-8_17","title":"Vision Language Models are blind","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Artificial intelligence; Natural language processing; Computer vision; Speech recognition","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001053534,0.0008928317,0.0009846973,0.0006185666,0.0006196969,0.003626584,0.001090565,0.002154596,0.01176862],"category_scores_gemma":[0.007967066,0.0009047221,0.0007072523,0.0006039462,0.002212717,0.007022109,0.001787771,0.004178441,0.00887177],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008019052,"about_ca_system_score_gemma":0.0009963608,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003286597,"about_ca_topic_score_gemma":0.001925527,"domain_scores_codex":[0.9991515,0.0001917652,0.0000384048,0.0002536678,0.0002738297,0.00009089545],"domain_scores_gemma":[0.997641,0.001177571,0.0001322138,0.0005835134,0.0003866042,0.00007913913],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001881658,0.00006083706,0.0004346289,0.0003064238,0.00009196869,0.0001897301,0.0003003688,0.0195,0.007345468,0.554504,0.1023503,0.3147281],"study_design_scores_gemma":[0.00001488818,0.00002032444,0.000237909,0.00005420249,0.00002905866,0.0002607308,0.00007634924,0.07755405,0.005437205,0.8718761,0.0444014,0.0000376514],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01234385,0.009024541,0.8123169,0.01423516,0.002171679,0.00003412558,0.001010885,0.006089141,0.1427738],"genre_scores_gemma":[0.5426603,0.01175323,0.1411215,0.006926325,0.002330933,0.0001190296,0.002564536,0.00281165,0.2897125],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01176862,"threshold_uncertainty_score":0.03936994,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01704194164196904,"score_gpt":0.2930850625544102,"score_spread":0.2760431209124412,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}