{"id":"W157143106","doi":"","title":"A Hierarchical EM Approach to Word Segmentation.","year":2001,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Morpheme; Word (group theory); Artificial intelligence; Computer science; Segmentation; Lexicon; Text segmentation; Natural language processing; Character (mathematics); Speech recognition; Minimum description length; Hidden Markov model; Pattern recognition (psychology); Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001251889,0.00007716383,0.00007158171,0.0001019767,0.00006021951,0.000193985,0.0007063384,0.00003444514,0.00001809466],"category_scores_gemma":[0.00002683632,0.00005950007,0.00002367095,0.0005669478,0.0000107177,0.0003037295,0.0002323151,0.0001048386,0.00005161873],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002916425,"about_ca_system_score_gemma":0.00002069854,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001099124,"about_ca_topic_score_gemma":0.000002922071,"domain_scores_codex":[0.9992169,0.0000235997,0.0001042305,0.0002709998,0.0002023541,0.0001818828],"domain_scores_gemma":[0.9995334,0.00002183308,0.00001793763,0.0002944939,0.00003724284,0.00009516178],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000008697453,0.0001218872,0.0001630784,0.00000915405,0.000006500647,0.00002076043,0.00196847,0.00001175899,0.003010698,0.258991,0.009508232,0.7261798],"study_design_scores_gemma":[0.001677568,0.0005803286,0.002987936,0.0001266523,0.00002152194,0.001089116,0.0008242716,0.1494329,0.1347392,0.6589387,0.04675751,0.002824325],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.002906844,0.0001141537,0.9818735,0.001214513,0.00004019676,0.0001360832,1.564564e-7,0.0008591288,0.01285541],"genre_scores_gemma":[0.1283101,0.000001765458,0.8678077,0.002075564,0.00003504518,0.00002692928,0.00000121045,0.000004379409,0.00173732],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.7233554,"threshold_uncertainty_score":0.2426343,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01821088677380965,"score_gpt":0.2825513774056881,"score_spread":0.2643404906318785,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}