{"id":"W1982365650","doi":"10.3115/1706543.1706551","title":"Computing word similarity and identifying cognates with pair hidden Markov models","year":2005,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":45,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Hidden Markov model; Computer science; Word (group theory); Similarity (geometry); Focus (optics); Artificial intelligence; Markov chain; Markov model; Identification (biology); Maximum-entropy Markov model; Natural language processing; Variation (astronomy); Speech recognition; Variable-order Markov model; Pattern recognition (psychology); Machine learning; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002241612,0.0009152924,0.001251844,0.003221982,0.001010027,0.001738694,0.001350763,0.001659409,0.003687646],"category_scores_gemma":[0.01234602,0.0005664296,0.001117155,0.001922067,0.0007237055,0.004952183,0.002440155,0.001164988,0.002006792],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005565364,"about_ca_system_score_gemma":0.001017163,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002360893,"about_ca_topic_score_gemma":0.00295972,"domain_scores_codex":[0.9974889,0.0009032932,0.0001977657,0.0007752315,0.0004870725,0.0001478378],"domain_scores_gemma":[0.9954773,0.002688856,0.0004399696,0.0006799394,0.0005131595,0.000200814],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002584996,0.00080549,0.03273079,0.0007673424,0.0005834218,0.0008233525,0.001400765,0.03817395,0.05792073,0.02499796,0.007697479,0.8315138],"study_design_scores_gemma":[0.00008848837,0.0003494764,0.004600266,0.00005303723,0.0001735055,0.0006162976,0.0003048533,0.9016777,0.03821868,0.04958495,0.004219357,0.0001133992],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1011482,0.0003857838,0.8839238,0.0001994002,0.00008880029,0.0002070703,0.0007066983,0.01096842,0.002371821],"genre_scores_gemma":[0.4206362,0.000152646,0.5752787,0.0001369308,0.00006359742,0.0002240028,0.001888025,0.0004314791,0.001188339],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003687646,"threshold_uncertainty_score":0.01233643,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02002455700352461,"score_gpt":0.2656999985354302,"score_spread":0.2456754415319056,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}