{"id":"W7077920206","doi":"10.48448/k2yz-re54","title":"A Massive-Scale Benchmark for Multilingual and Multicultural Visual Question Answering on Global Cuisines","year":2025,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology; University of Toronto; McGill University","funders":"","keywords":"Benchmark (surveying); Question answering; Knowledge base; Multiculturalism; Adversarial system; Questions and answers","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0003822719,0.00030836,0.0002707416,0.0001653845,0.0003031491,0.0002698966,0.0008305662,0.0002079083,0.00002606135],"category_scores_gemma":[0.0005052406,0.0002551428,0.00005382627,0.0005440213,0.0004726558,0.0001847928,0.0004101826,0.0001567388,0.000006587421],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009788141,"about_ca_system_score_gemma":0.0002780643,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001532012,"about_ca_topic_score_gemma":0.0001226045,"domain_scores_codex":[0.997954,0.00002531778,0.000232361,0.000969578,0.0003559117,0.0004628902],"domain_scores_gemma":[0.9989926,0.00009525985,0.0001753056,0.0003541145,0.000250283,0.0001324207],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001399559,0.00110496,0.0032695,0.001814919,0.0001841725,0.0001011525,0.001937244,0.00549622,0.01601967,0.08735371,0.07100214,0.8115764],"study_design_scores_gemma":[0.00188162,0.0005069526,0.001858648,0.00162405,0.00005826532,0.00004493438,0.0004423768,0.8008795,0.009395772,0.006095328,0.175641,0.001571604],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.01228757,0.00101665,0.3831117,0.009638468,0.005934995,0.002840183,0.0002429307,0.001952337,0.5829752],"genre_scores_gemma":[0.4378921,0.00007005946,0.2930414,0.001043088,0.00116574,0.0001037719,0.0001214081,0.00004369104,0.2665187],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.8100048,"threshold_uncertainty_score":0.9999901,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01030860831526746,"score_gpt":0.3008500997751254,"score_spread":0.2905414914598579,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}