{"id":"W4397012280","doi":"10.1017/nlp.2024.7","title":"A survey of context in neural machine translation and its evaluation","year":2024,"lang":"en","type":"article","venue":"Natural language processing.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"Dublin City University; Science Foundation Ireland","keywords":"Machine translation; Computer science; Artificial intelligence; Evaluation of machine translation; Context (archaeology); Natural language processing; Terminology; Machine translation software usability; Consistency (knowledge bases); Example-based machine translation; Sentence; Computer-assisted translation; Task (project management); Paragraph; Transfer-based machine translation; Machine learning; Linguistics; World Wide Web; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"simulation_or_modeling","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"},{"model":"gpt","categories":[],"domain":null,"study_design":"design_other","genre":"review","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"}],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001200429,0.0001674724,0.0002080248,0.0003216718,0.00004143562,0.0002081295,0.0003713108,0.0001043749,0.000008807088],"category_scores_gemma":[0.0003116757,0.000133637,0.00003076506,0.001012987,0.00003459365,0.001116576,0.00007800641,0.000374032,0.000001623841],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005292219,"about_ca_system_score_gemma":0.0001272299,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003744379,"about_ca_topic_score_gemma":0.000523493,"domain_scores_codex":[0.9984742,0.0001774378,0.0003103892,0.0004106246,0.0004265146,0.0002008712],"domain_scores_gemma":[0.9993268,0.0001632666,0.00008707403,0.0001549779,0.0002292766,0.00003864651],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001414486,0.00001347461,0.0002605296,0.000240383,0.000004058089,0.00001853542,0.002735178,0.000003252864,0.001215373,0.00008019738,0.00001011634,0.9954048],"study_design_scores_gemma":[0.0002483814,0.00003079389,0.002776106,0.0002585694,0.000009357004,0.0000274343,0.00003319751,0.9948944,0.001461541,0.00009091802,0.00001190488,0.0001573541],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"review","genre_gemma":"empirical","genre_scores_codex":[0.01872984,0.9748766,0.005152482,0.0003949656,0.0001534186,0.0002889692,0.00001109746,0.0003181414,0.0000745157],"genre_scores_gemma":[0.9890822,0.00003369208,0.01066072,0.00009984415,0.0000217572,0.0000174472,0.00003615488,0.00001410868,0.00003405881],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9952474,"threshold_uncertainty_score":0.5449557,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03055695989947995,"score_gpt":0.3326759585021715,"score_spread":0.3021189986026915,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}