{"id":"W1517708896","doi":"10.1109/nlpke.2005.1598738","title":"Improved Estimation for Unsupervised Part-of-Speech Tagging","year":2006,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Hidden Markov model; Computer science; Word (group theory); Speech recognition; Artificial intelligence; Simple (philosophy); Maximum-entropy Markov model; Pattern recognition (psychology); Markov model; Natural language processing; Machine learning; Markov chain; Mathematics; Variable-order Markov model","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001951111,0.001107533,0.00105642,0.000579176,0.0005116274,0.0009957985,0.001644712,0.001177068,0.002074499],"category_scores_gemma":[0.009100508,0.0006302596,0.0007246341,0.0009166822,0.0006022772,0.002042552,0.001331064,0.001514697,0.004136621],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000464373,"about_ca_system_score_gemma":0.0008953345,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003587349,"about_ca_topic_score_gemma":0.006664454,"domain_scores_codex":[0.9985082,0.0006126124,0.00007106485,0.0003812937,0.0003183852,0.0001084439],"domain_scores_gemma":[0.9953687,0.002567409,0.000211286,0.001118754,0.0006626061,0.00007120997],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002307509,0.0002148906,0.003895465,0.000155243,0.0001506799,0.0001970151,0.0003106787,0.4860075,0.04407575,0.01897202,0.006422882,0.4393672],"study_design_scores_gemma":[0.000005733769,0.00001568512,0.0004346239,0.000005531097,0.00001248384,0.00005339277,0.00001003863,0.9822037,0.008501667,0.00709017,0.001647319,0.00001962643],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.004852135,0.00004875906,0.9933105,0.00003358558,0.00002654336,0.00001137315,0.00005423864,0.001309347,0.0003536106],"genre_scores_gemma":[0.1865736,0.0001722808,0.8078691,0.0001419483,0.00007935363,0.0001322913,0.001051738,0.000778162,0.003201479],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003587349,"threshold_uncertainty_score":0.01031858,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01191686440937055,"score_gpt":0.2657707391658634,"score_spread":0.2538538747564928,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}