{"id":"W4307751734","doi":"10.3390/app122111038","title":"Framework for Handling Rare Word Problems in Neural Machine Translation System Using Multi-Word Expressions","year":2022,"lang":"en","type":"article","venue":"Applied Sciences","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"ca_institutions":"Brandon University","funders":"","keywords":"Computer science; Artificial intelligence; Natural language processing; Machine translation; Fluency; Word (group theory); Test set; Set (abstract data type); Word embedding; Vocabulary; Translation (biology); Embedding; Linguistics; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008961157,0.0001427003,0.0001779622,0.0002548788,0.001068867,0.000237056,0.001270304,0.00005336983,0.000005455578],"category_scores_gemma":[0.00003492571,0.0001210734,0.0000444862,0.001161518,0.00009860758,0.0004073975,0.0002863659,0.0003088355,3.722639e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009815065,"about_ca_system_score_gemma":0.00008196189,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004828751,"about_ca_topic_score_gemma":0.00001890756,"domain_scores_codex":[0.9983498,0.00006761238,0.0002852775,0.0005442626,0.0003986747,0.0003543837],"domain_scores_gemma":[0.999282,0.0002504236,0.0001552331,0.0002387373,0.00002531515,0.00004830839],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000050861,0.0002356816,0.001451504,0.0003919366,0.00001170122,0.00001834371,0.0245852,0.114006,0.1019464,0.3046705,0.00001559065,0.4526163],"study_design_scores_gemma":[0.0002402092,0.00003127392,0.00003718634,0.0001167718,0.000003566955,0.0000103826,0.0006038653,0.97658,0.003053465,0.01906008,0.00005794642,0.0002052321],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0195377,0.004323568,0.9744623,0.0002585025,0.000233287,0.0007238581,0.000005572258,0.0004118293,0.00004335021],"genre_scores_gemma":[0.5239695,6.01184e-7,0.4757726,0.00006136004,0.00001656501,0.0001694754,0.000001553467,0.000005042595,0.000003260519],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.862574,"threshold_uncertainty_score":0.8220968,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0708010812538504,"score_gpt":0.327040812905348,"score_spread":0.2562397316514976,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}