{"id":"W7082275164","doi":"10.48448/3yrn-9195","title":"Global MMLU: Understanding and Addressing Cultural and Linguistic Biases in Multilingual Evaluation","year":2025,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology; McGill University","funders":"","keywords":"Cultural diversity; Annotation; Benchmark (surveying); Culturally appropriate; Cultural bias; Language proficiency; Cultural competence; Translation (biology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05141301,0.00301476,0.001351663,0.005107014,0.001569125,0.007698386,0.002885147,0.001880656,0.008470332],"category_scores_gemma":[0.1712451,0.0006895373,0.001395541,0.003494531,0.002108926,0.01025083,0.009975891,0.003390673,0.003361984],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002620451,"about_ca_system_score_gemma":0.003528059,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005772543,"about_ca_topic_score_gemma":0.01028102,"domain_scores_codex":[0.9342782,0.05267834,0.003035791,0.003749845,0.005496739,0.0007610833],"domain_scores_gemma":[0.8862115,0.07560313,0.003466368,0.01616188,0.01680585,0.001751313],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001430657,0.0006525616,0.03146835,0.005343674,0.001410589,0.0004066619,0.009534436,0.0360626,0.01179043,0.05145529,0.1069396,0.7435051],"study_design_scores_gemma":[0.0006167044,0.002068563,0.02656025,0.004799238,0.001012749,0.001160384,0.01164318,0.4565555,0.04298964,0.2286999,0.2230045,0.0008895198],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1128795,0.007077142,0.7868649,0.006715962,0.00115739,0.00161911,0.008332997,0.02293862,0.05241432],"genre_scores_gemma":[0.6229731,0.001201416,0.3447619,0.001896786,0.0002895106,0.002148555,0.01550216,0.005833289,0.005393259],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.05141301,"threshold_uncertainty_score":0.2719012,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1505566938512948,"score_gpt":0.3835747598386859,"score_spread":0.2330180659873912,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}