{"id":"W7106790711","doi":"10.48448/sk2n-f313","title":"MUG-Eval: A Proxy Evaluation Framework for Multilingual Generation Capabilities in Any Language","year":2025,"lang":"","type":"other","venue":"Open MIND","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Conversation; Proxy (statistics); Task (project management); Key (lock); Computational linguistics; Quality (philosophy); Language model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02120305,0.003863424,0.001772089,0.006857874,0.001490778,0.006050022,0.003678939,0.002805772,0.008105219],"category_scores_gemma":[0.07247128,0.0008189315,0.00171323,0.002943827,0.001883053,0.008237632,0.008194422,0.002636102,0.005058503],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002292437,"about_ca_system_score_gemma":0.00367357,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01021832,"about_ca_topic_score_gemma":0.01266515,"domain_scores_codex":[0.9690652,0.01872105,0.001868287,0.003671495,0.005745804,0.0009280927],"domain_scores_gemma":[0.9689951,0.01629832,0.001947691,0.006717411,0.004930964,0.001110622],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.002502804,0.001000665,0.02931825,0.002536555,0.001152319,0.0006047412,0.001646164,0.144514,0.01670062,0.0362127,0.09444883,0.6693624],"study_design_scores_gemma":[0.0003167725,0.00103261,0.008192066,0.0004778376,0.0002043036,0.0004760991,0.0008425206,0.85718,0.02782966,0.0562005,0.0468778,0.0003697327],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"other","genre_scores_codex":[0.0952889,0.005296605,0.7196298,0.00160692,0.00100242,0.00141365,0.01701931,0.1245476,0.03419494],"genre_scores_gemma":[0.6059909,0.0006303901,0.3452208,0.0008246274,0.0002089804,0.001384023,0.03219808,0.008222813,0.005319356],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.02120305,"threshold_uncertainty_score":0.1121337,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09042435503935342,"score_gpt":0.4343194324993454,"score_spread":0.343895077459992,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}