{"id":"W2407793169","doi":"","title":"Training a Perceptron with Global and Local Features for Chinese Word Segmentation","year":2008,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Conditional random field; Segmentation; Artificial intelligence; Word (group theory); Computer science; Character (mathematics); Text segmentation; Natural language processing; Perceptron; Pattern recognition (psychology); Speech recognition; Artificial neural network; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003773972,0.00133123,0.001674861,0.001505095,0.0009031111,0.001042193,0.001368039,0.001590478,0.003935998],"category_scores_gemma":[0.005022627,0.0009864842,0.001281496,0.001829033,0.0006935347,0.00338302,0.001082181,0.001906889,0.001990892],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00098127,"about_ca_system_score_gemma":0.001404923,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007724361,"about_ca_topic_score_gemma":0.01058017,"domain_scores_codex":[0.9982116,0.0005762361,0.0001378669,0.0006745875,0.0001880969,0.0002115349],"domain_scores_gemma":[0.9973979,0.00166422,0.0001321154,0.0002189721,0.0004476786,0.0001392032],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00140307,0.0004227044,0.009645985,0.000312677,0.0004071301,0.0003456685,0.000365083,0.1732311,0.02388433,0.00248738,0.0081343,0.7793607],"study_design_scores_gemma":[0.0000357865,0.0001325329,0.001341403,0.00001429511,0.00006840856,0.00005768005,0.0000525493,0.9894013,0.005842485,0.002113733,0.0009170203,0.00002286709],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1515858,0.0009211142,0.8291192,0.0003619209,0.0002390037,0.0001324505,0.0003830734,0.01408193,0.003175455],"genre_scores_gemma":[0.6710196,0.0002474257,0.3204025,0.0003746762,0.0001275637,0.0002086417,0.001458046,0.0004548822,0.005706614],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007724361,"threshold_uncertainty_score":0.01995891,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01506532431074484,"score_gpt":0.290412212100892,"score_spread":0.2753468877901472,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}