{"id":"W1517708896","doi":"10.1109/nlpke.2005.1598738","title":"Improved Estimation for Unsupervised Part-of-Speech Tagging","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Hidden Markov model; Computer science; Word (group theory); Speech recognition; Artificial intelligence; Simple (philosophy); Maximum-entropy Markov model; Pattern recognition (psychology); Markov model; Natural language processing; Machine learning; Markov chain; Mathematics; Variable-order Markov model","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001811301,0.00007520836,0.00009472478,0.00006738295,0.00004981761,0.0000717705,0.0003959874,0.00004438596,0.000005445057],"category_scores_gemma":[0.00004890731,0.00006068605,0.00003960887,0.0001894024,0.00001601917,0.000377856,0.00006760938,0.00004032427,0.000001135054],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001773035,"about_ca_system_score_gemma":0.00002762988,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00009978558,"about_ca_topic_score_gemma":0.000008909893,"domain_scores_codex":[0.9993824,0.000009644873,0.0001820639,0.0001826845,0.0001001241,0.0001431295],"domain_scores_gemma":[0.9994832,0.00006484473,0.00007743137,0.0002457657,0.0001117089,0.00001708127],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000006495863,0.00003739678,0.00003446201,0.00008531774,0.000004790259,0.000001506682,0.00007712264,0.00003538732,0.09698216,0.3086698,0.002699553,0.591366],"study_design_scores_gemma":[0.0001378319,0.00003092991,0.000008211937,0.00001379481,0.000002274188,0.000002378279,0.000001943454,0.4741393,0.4540645,0.07062756,0.0008872642,0.00008393188],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00128672,0.000162229,0.9964101,0.0006518647,0.00007540116,0.000241238,0.000001307108,0.0005920047,0.0005791925],"genre_scores_gemma":[0.2097679,4.063563e-7,0.7898107,0.00009728347,0.00003925739,0.00002430809,0.000005548137,0.000004796279,0.0002498398],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.5912821,"threshold_uncertainty_score":0.2474705,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01191686440937055,"score_gpt":0.2657707391658634,"score_spread":0.2538538747564928,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}