{"id":"W3210115313","doi":"10.18653/v1/2021.wnut-1.23","title":"Noisy UGC Translation at the Character Level: Revisiting Open-Vocabulary Capabilities and Robustness of Char-Based Models","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian Nautical Research Society","funders":"Agence Nationale de la Recherche","keywords":"Computer science; Robustness (evolution); Machine translation; Artificial intelligence; Vocabulary; Natural language processing; Character (mathematics); Translation (biology); Bridging (networking); Machine learning; Linguistics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006174397,0.001376831,0.001425867,0.00101434,0.0009223059,0.003425514,0.002261016,0.002284196,0.003490718],"category_scores_gemma":[0.039645,0.0006678066,0.000912555,0.0009204605,0.002255016,0.006421829,0.002964218,0.003287986,0.002399383],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001149682,"about_ca_system_score_gemma":0.001311671,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006744715,"about_ca_topic_score_gemma":0.004539257,"domain_scores_codex":[0.9963676,0.001971397,0.0002163036,0.0007679512,0.0004608505,0.0002159471],"domain_scores_gemma":[0.978478,0.01523271,0.0008153083,0.003503706,0.001594011,0.0003762445],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007285483,0.0002122488,0.003855291,0.0005402726,0.0002379087,0.0004038902,0.0009449346,0.8095265,0.01847692,0.02723718,0.00348842,0.134348],"study_design_scores_gemma":[0.00001106203,0.00006484594,0.0002029283,0.0000283575,0.00001631625,0.0000509593,0.00005046778,0.9829516,0.003037186,0.0128828,0.0006872107,0.00001632658],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2380891,0.001832832,0.7402427,0.002292547,0.0003427516,0.0001537023,0.0006312804,0.005270686,0.01114448],"genre_scores_gemma":[0.9285393,0.0004655039,0.06518655,0.0005022848,0.0001452466,0.0001237554,0.00101923,0.0007301639,0.003287913],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006744715,"threshold_uncertainty_score":0.03265375,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07650691870858589,"score_gpt":0.2976728666687269,"score_spread":0.221165947960141,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}