{"id":"W1982365650","doi":"10.3115/1706543.1706551","title":"Computing word similarity and identifying cognates with pair hidden Markov models","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":45,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Hidden Markov model; Computer science; Word (group theory); Similarity (geometry); Focus (optics); Artificial intelligence; Markov chain; Markov model; Identification (biology); Maximum-entropy Markov model; Natural language processing; Variation (astronomy); Speech recognition; Variable-order Markov model; Pattern recognition (psychology); Machine learning; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003480821,0.0001505397,0.0001506793,0.00009166472,0.0001723269,0.0004231828,0.0005574948,0.00006189963,0.000007052092],"category_scores_gemma":[0.00001830396,0.0001120856,0.00002122126,0.0002582506,0.00005424553,0.00127457,0.000487458,0.0002023258,0.000002738289],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003184877,"about_ca_system_score_gemma":0.00002800847,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005516541,"about_ca_topic_score_gemma":0.0000510857,"domain_scores_codex":[0.9988958,0.00003916711,0.0001637326,0.0003963734,0.0002385526,0.0002663924],"domain_scores_gemma":[0.9993804,0.00009213275,0.0000744163,0.0002857732,0.00009041338,0.00007693002],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001076652,0.00004272055,0.001652051,0.00006700882,0.00002219248,0.00003540912,0.001156668,0.00006152292,0.0006175248,0.0393461,0.0008374774,0.9561505],"study_design_scores_gemma":[0.000338103,0.00004492539,0.0004855155,0.0001850502,0.00001082367,0.000106549,0.00005717711,0.9155584,0.01884135,0.06374294,0.0001826998,0.0004464672],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02495215,0.001854871,0.9689703,0.001574896,0.00002523918,0.0001240634,4.669332e-7,0.001255311,0.001242726],"genre_scores_gemma":[0.4923362,0.00000517297,0.5070822,0.0004010496,0.00002476337,0.000001466353,5.348638e-7,0.000005625526,0.0001429596],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9557041,"threshold_uncertainty_score":0.457072,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02002455700352461,"score_gpt":0.2656999985354302,"score_spread":0.2456754415319056,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}