{"id":"W2980941473","doi":"10.18653/v1/w19-5351","title":"Linguistic Evaluation of German-English Machine Translation Using a Test Suite","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Banting and Best Diabetes Centre, University of Toronto; Bundesministerium für Bildung und Forschung","keywords":"German; Valency; Computer science; Punctuation; Linguistics; Natural language processing; Test suite; Verb; Suite; Artificial intelligence; Test (biology); Machine translation; Test case; History","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007834214,0.00239099,0.001302722,0.003758762,0.0011136,0.001596806,0.001392596,0.001456811,0.002780336],"category_scores_gemma":[0.02627442,0.0005204774,0.001078683,0.003210723,0.0007597571,0.001048341,0.001996496,0.000916367,0.002278431],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009987653,"about_ca_system_score_gemma":0.0008827948,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004754002,"about_ca_topic_score_gemma":0.004606049,"domain_scores_codex":[0.9836451,0.01106154,0.001571858,0.001176884,0.001968185,0.0005764691],"domain_scores_gemma":[0.9659515,0.02061754,0.001228496,0.003173117,0.00784123,0.001188021],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.008546036,0.008467501,0.06841766,0.004475452,0.00319482,0.006294451,0.007796585,0.1372405,0.1492227,0.003433829,0.05204334,0.5508671],"study_design_scores_gemma":[0.002812764,0.01528534,0.2161755,0.0004044883,0.001934233,0.005400334,0.004048725,0.4141931,0.2742147,0.002547866,0.06248926,0.0004937606],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9722394,0.0009734125,0.01384534,0.0002771346,0.0001348727,0.0002628252,0.003981008,0.004220817,0.004065224],"genre_scores_gemma":[0.9264615,0.0004513755,0.0261347,0.0001666273,0.0001176407,0.0004322609,0.04061571,0.00128164,0.00433851],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.007834214,"threshold_uncertainty_score":0.04143178,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05688717792736218,"score_gpt":0.3638439371814072,"score_spread":0.306956759254045,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}