{"id":"W4416033845","doi":"10.18653/v1/2025.wmt-1.69","title":"MSLC25: Metric Performance on Low-Quality Machine Translation, Empty Strings, and Language Variants","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Metric (unit); Machine translation; Focus (optics); Variety (cybernetics); Translation (biology); Range (aeronautics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01664453,0.005851497,0.003616322,0.01074411,0.00301707,0.005195859,0.003563904,0.003733767,0.007634688],"category_scores_gemma":[0.06121928,0.000610425,0.002117844,0.009803848,0.002232453,0.005688042,0.004470584,0.003301732,0.01015341],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003311214,"about_ca_system_score_gemma":0.003294626,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01561491,"about_ca_topic_score_gemma":0.02495846,"domain_scores_codex":[0.9716738,0.01122066,0.003557996,0.004432943,0.007545131,0.001569457],"domain_scores_gemma":[0.9580573,0.01718909,0.00175223,0.009672921,0.01130741,0.002021042],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.004363091,0.001336448,0.02300856,0.005303952,0.002043115,0.0007769254,0.001123438,0.05262688,0.01839289,0.006737727,0.5179511,0.3663357],"study_design_scores_gemma":[0.0023417,0.005543481,0.07417456,0.001302135,0.001090407,0.004116704,0.0030323,0.451327,0.120615,0.03871478,0.2963772,0.001364623],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.531623,0.03295628,0.1003179,0.004228229,0.008233773,0.001952668,0.140906,0.1111932,0.06858901],"genre_scores_gemma":[0.5201012,0.001824038,0.1204256,0.001327673,0.0009905437,0.001374992,0.3260254,0.01333306,0.01459749],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01664453,"threshold_uncertainty_score":0.08802581,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01252710956702142,"score_gpt":0.3015938198665457,"score_spread":0.2890667102995243,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}