{"id":"W4389519548","doi":"10.18653/v1/2023.wmt-1.51","title":"Results of WMT23 Metrics Shared Task: Metrics Might Be Guilty but References Are Not Innocent","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"Bundesministerium für Bildung und Forschung; Deutsche Forschungsgemeinschaft","keywords":"George (robot); Task (project management); Computer science; Machine translation; Artificial intelligence; Engineering; Systems engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001128449,0.0002400533,0.0003875327,0.001474322,0.0001142387,0.0002306055,0.002690587,0.000188017,0.0000166285],"category_scores_gemma":[0.003570525,0.000180535,0.0000972916,0.01063797,0.00006556102,0.0006149308,0.001316365,0.0002907685,0.00003575919],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008650504,"about_ca_system_score_gemma":0.0001229126,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003544574,"about_ca_topic_score_gemma":0.00005492077,"domain_scores_codex":[0.9970638,0.00009309404,0.0006613233,0.0006791117,0.00105504,0.0004476061],"domain_scores_gemma":[0.997099,0.0005347713,0.0004689703,0.001085115,0.0006886204,0.0001235525],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.000183364,0.0005456716,0.002406985,0.0005905981,0.0001802519,0.0004381169,0.00291785,0.000022872,0.03115334,0.1074668,0.6577345,0.1963596],"study_design_scores_gemma":[0.001175661,0.0004025512,0.006003549,0.0002614614,0.00004025282,0.00001939584,0.000443822,0.03090821,0.917798,0.02066169,0.02116918,0.00111623],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06831805,0.01124584,0.8189359,0.05185109,0.003615078,0.002302864,0.001982699,0.01695777,0.02479076],"genre_scores_gemma":[0.618795,0.0001477333,0.3777364,0.0005703432,0.00004541205,0.00001504674,0.00005537015,0.00001550169,0.002619109],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8866447,"threshold_uncertainty_score":0.7362003,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07787230308748376,"score_gpt":0.3156563963070204,"score_spread":0.2377840932195366,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}