{"id":"W4405023785","doi":"10.1109/cvpr52734.2025.01822","title":"All Languages Matter: Evaluating LMMs on Culturally Diverse 100 Languages","year":2025,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Benchmark (surveying); Computer science; Resource (disambiguation); Artificial intelligence; Data science; Natural language processing; Geography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009059776,0.002738687,0.0008991589,0.001654204,0.001243134,0.00317345,0.002998526,0.002697508,0.01326929],"category_scores_gemma":[0.03964505,0.0004644523,0.002091633,0.001359031,0.001309208,0.006450656,0.00482202,0.002974109,0.004732336],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002836028,"about_ca_system_score_gemma":0.001988583,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01488939,"about_ca_topic_score_gemma":0.01972989,"domain_scores_codex":[0.9919738,0.005558912,0.0004115989,0.0009990414,0.0008070339,0.0002496353],"domain_scores_gemma":[0.9796007,0.01563732,0.0004731032,0.001896822,0.001697536,0.0006944206],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.003465353,0.004297025,0.05457202,0.005314319,0.001426144,0.0006875534,0.005699627,0.2129634,0.007737758,0.02461318,0.1490443,0.5301793],"study_design_scores_gemma":[0.0007190063,0.002259319,0.02346869,0.001565847,0.0005554869,0.0005114658,0.008331732,0.8339854,0.01179717,0.05305268,0.06338827,0.0003649579],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.748957,0.005668902,0.1189962,0.004846936,0.001163704,0.003296845,0.03083269,0.01392399,0.07231385],"genre_scores_gemma":[0.7991775,0.0009772932,0.1181768,0.002652762,0.0001750972,0.003484264,0.0647361,0.001201069,0.009419125],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01488939,"threshold_uncertainty_score":0.04791325,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02232824207649547,"score_gpt":0.3794272945488076,"score_spread":0.3570990524723121,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}