{"id":"W7131990533","doi":"","title":"Results of the WMT21 metrics shared task: evaluating metrics with expert-based human evaluations on TED and news domain","year":2021,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Grantová Agentura České Republiky","keywords":"Robustness (evolution); Task (project management); Annotation; Domain (mathematical analysis); Task analysis; Quality (philosophy); Machine translation","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02775997,0.004742843,0.002536234,0.005205954,0.001832988,0.003482931,0.002015265,0.003699048,0.005338296],"category_scores_gemma":[0.08903466,0.0005651286,0.001931909,0.002817136,0.001751331,0.004408782,0.005956547,0.002259095,0.004952998],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002041552,"about_ca_system_score_gemma":0.00180166,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008300084,"about_ca_topic_score_gemma":0.01034305,"domain_scores_codex":[0.9505237,0.02900882,0.004593099,0.006185222,0.008165775,0.001523501],"domain_scores_gemma":[0.9146718,0.04130284,0.004485306,0.01351249,0.02114609,0.004881539],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.008299509,0.005772996,0.04588085,0.006738405,0.003077286,0.001427014,0.009889858,0.0329796,0.07152028,0.002845123,0.1860143,0.6255548],"study_design_scores_gemma":[0.004911943,0.02086455,0.3432255,0.001274412,0.001944329,0.004810916,0.01044615,0.2507396,0.1397009,0.01271896,0.2068832,0.002479401],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8332656,0.006765197,0.06847443,0.001749059,0.002049766,0.00292009,0.02710854,0.01695537,0.04071198],"genre_scores_gemma":[0.834196,0.0006075224,0.07572305,0.0008350801,0.0005829863,0.003356528,0.0630923,0.003664007,0.01794258],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.97224,"threshold_uncertainty_score":0.1468105,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05253733671464236,"score_gpt":0.3546825522455395,"score_spread":0.3021452155308971,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}