{"id":"W2104554525","doi":"10.5539/cis.v8n1p119","title":"Augmenting Performance of SMT Models by Deploying Fine Tokenization of the Text and Part-of-Speech Tag","year":2015,"lang":"en","type":"article","venue":"Computer and Information Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Natural Science Foundation of China","keywords":"Computer science; Machine translation; Natural language processing; Lexical analysis; Artificial intelligence; Phrase; Word (group theory); Language model; Set (abstract data type); Translation (biology); Speech recognition; Linguistics; Programming language","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002437413,0.001944174,0.001402953,0.001116233,0.0006028273,0.001698041,0.00139343,0.001068257,0.004381843],"category_scores_gemma":[0.008163886,0.000610441,0.001032137,0.001248994,0.0003978832,0.003157269,0.0012225,0.001822444,0.008651255],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006761125,"about_ca_system_score_gemma":0.001562084,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008812862,"about_ca_topic_score_gemma":0.01456239,"domain_scores_codex":[0.9985327,0.000547965,0.00009819328,0.0004078889,0.0002955187,0.0001177643],"domain_scores_gemma":[0.9953721,0.002650311,0.0001583988,0.0009022566,0.0008157999,0.0001011636],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0007090512,0.0003504462,0.006352907,0.0006099947,0.0004409988,0.0002761408,0.0002663641,0.1612198,0.04637344,0.001381003,0.01043413,0.7715857],"study_design_scores_gemma":[0.00003218782,0.0003412144,0.001915884,0.00003869388,0.0001531662,0.0002336923,0.0001168455,0.9591455,0.02664091,0.002495683,0.008824253,0.0000619572],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.225921,0.004932093,0.7140635,0.001572671,0.00104445,0.0003795075,0.00186817,0.03664044,0.0135781],"genre_scores_gemma":[0.6982049,0.001518769,0.2813706,0.0006751649,0.0002937517,0.000207589,0.0051884,0.001750421,0.01079047],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008812862,"threshold_uncertainty_score":0.01752317,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01804727061539576,"score_gpt":0.2396531086832067,"score_spread":0.2216058380678109,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}