{"id":"W3120179416","doi":"10.18653/v1/2020.wmt-1.99","title":"Extended Study on Using Pretrained Language Models and YiSi-1 for Machine Translation Evaluation","year":2020,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Machine translation; Computer science; Task (project management); Artificial intelligence; Natural language processing; Translation (biology); Evaluation of machine translation; BLEU; Language model; Quality (philosophy); Embedding; Machine learning; Example-based machine translation; Machine translation software usability","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004150269,0.00009833244,0.0001049021,0.00005725382,0.000069326,0.0001022534,0.0002093565,0.00003652156,0.000004527363],"category_scores_gemma":[0.00006791883,0.00007908267,0.00002232969,0.0001788973,0.000008312392,0.0004939612,0.0000370833,0.00007121835,3.127858e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002190601,"about_ca_system_score_gemma":0.00002559497,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001901584,"about_ca_topic_score_gemma":0.000005324326,"domain_scores_codex":[0.9991062,0.00006394807,0.0001363019,0.0003176939,0.0002710268,0.000104851],"domain_scores_gemma":[0.9996132,0.00005961132,0.00004788447,0.000160154,0.00007297566,0.00004621241],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000115534,0.0001859428,0.00003262236,0.00006328236,0.00003059131,0.000005550303,0.0304365,0.0008134826,0.04837855,0.01458122,0.00004845096,0.9053082],"study_design_scores_gemma":[0.0006225656,0.0003587856,0.0000336213,0.000009282931,0.00001768676,0.000001448652,0.0001543664,0.9777681,0.007627087,0.0133048,0.000001690641,0.000100628],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0359192,0.0009033586,0.960719,0.0008133482,0.00002097474,0.001087494,0.000002798611,0.0003743359,0.0001594269],"genre_scores_gemma":[0.6315781,4.609083e-7,0.3680821,0.0002881202,0.00001872153,0.00001955207,0.000003187753,0.000005548191,0.000004204248],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9769546,"threshold_uncertainty_score":0.3224898,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1102848588545824,"score_gpt":0.3752472339290164,"score_spread":0.264962375074434,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}