{"id":"W2104554525","doi":"10.5539/cis.v8n1p119","title":"Augmenting Performance of SMT Models by Deploying Fine Tokenization of the Text and Part-of-Speech Tag","year":2015,"lang":"en","type":"article","venue":"Computer and Information Science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Natural Science Foundation of China","keywords":"Computer science; Machine translation; Natural language processing; Lexical analysis; Artificial intelligence; Phrase; Word (group theory); Language model; Set (abstract data type); Translation (biology); Speech recognition; Linguistics; Programming language","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006412912,0.00005680957,0.00008999632,0.0001106059,0.00009093892,0.00009130609,0.0005193573,0.0000212152,2.680598e-7],"category_scores_gemma":[0.0000453851,0.00003939118,0.00001001133,0.0006440291,0.0002066484,0.006338776,0.0004804268,0.00004810112,1.621474e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001142303,"about_ca_system_score_gemma":0.00007391811,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000008334231,"about_ca_topic_score_gemma":1.821999e-7,"domain_scores_codex":[0.9991598,0.00001250783,0.000289259,0.00009138894,0.0003516563,0.00009541677],"domain_scores_gemma":[0.9990876,0.0000217226,0.0002820268,0.0001874119,0.0003842579,0.00003694798],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000009584597,0.00001792776,0.004988281,0.0002554378,0.000003991092,6.759146e-8,0.009202654,0.002119675,0.004920539,0.03222838,0.0004710317,0.9457824],"study_design_scores_gemma":[0.0001195653,0.00005895682,0.0004494245,0.00008717634,0.000001477201,0.000006853031,0.000018366,0.9029777,0.09525816,0.0008179473,0.0001500839,0.00005429051],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3462477,0.0002296504,0.6530181,0.00007311541,0.00008786052,0.00008175921,0.000001288545,0.00002978981,0.0002307964],"genre_scores_gemma":[0.8458399,0.00001899693,0.1540659,0.00006334278,0.000006244292,0.000001046285,6.690706e-7,9.551687e-7,0.000002905862],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9457281,"threshold_uncertainty_score":0.459546,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01804727061539576,"score_gpt":0.2396531086832067,"score_spread":0.2216058380678109,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}