{"id":"W6910422042","doi":"10.48448/a52t-ff08","title":"Benchmarking Vision Language Models for Cultural Understanding","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute; Université de Montréal","funders":"","keywords":"Benchmarking; Set (abstract data type); Cultural diversity; Benchmark (surveying); Cultural background; Comprehension; Foundation (evidence)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001364598,0.0005338191,0.0004438827,0.00163562,0.0003291531,0.0008107609,0.001023038,0.0002910048,0.0005654712],"category_scores_gemma":[0.0001041728,0.0004303543,0.0001671822,0.001645073,0.001143779,0.0006677592,0.0003901833,0.0004018545,0.001680721],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001513763,"about_ca_system_score_gemma":0.0004144352,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002414512,"about_ca_topic_score_gemma":0.0007826958,"domain_scores_codex":[0.9959971,0.00003251151,0.0003790381,0.00136919,0.001243702,0.0009785154],"domain_scores_gemma":[0.9986601,0.0000777754,0.0002806494,0.0006383577,0.0001039541,0.0002391441],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001553501,0.00005256052,0.000001493363,0.0003050983,0.000060904,0.0000267005,0.001598965,0.0007809433,0.00688801,0.1159012,0.8716971,0.002671468],"study_design_scores_gemma":[0.0007811011,0.0002453492,6.25079e-7,0.002522185,0.0002214446,0.00004011059,0.00697082,0.7777327,0.0002888561,0.05487559,0.1548474,0.001473826],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.00004891244,0.002943292,0.1216069,0.0002357017,0.002668697,0.001406947,0.0007636735,0.001781921,0.868544],"genre_scores_gemma":[0.1525116,0.00009615936,0.07999009,0.000299629,0.003532318,0.000114683,0.0006656341,0.003492795,0.7592971],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.7769517,"threshold_uncertainty_score":0.9998148,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08421030401067177,"score_gpt":0.3806526856515964,"score_spread":0.2964423816409246,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}