{"id":"W2250657762","doi":"","title":"Clustering Semantically Equivalent Words into Cognate Sets in Multilingual Lists","year":2011,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":75,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Cognate; Computer science; Word (group theory); Natural language processing; Artificial intelligence; Cluster analysis; ENCODE; Similarity (geometry); Mathematics; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005560224,0.0005695625,0.0006533322,0.004696778,0.001738115,0.00188989,0.0007348501,0.0007286106,0.002911472],"category_scores_gemma":[0.004602815,0.0002187139,0.0007170598,0.002789084,0.0007935276,0.002337492,0.001492426,0.0005670051,0.001444871],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006985156,"about_ca_system_score_gemma":0.0007808064,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00275768,"about_ca_topic_score_gemma":0.005118226,"domain_scores_codex":[0.99902,0.0002412279,0.0001215835,0.0003179921,0.0002085863,0.00009062799],"domain_scores_gemma":[0.9982802,0.0007126956,0.0002310999,0.0002473218,0.0004465655,0.00008209787],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00156088,0.0007453299,0.05212946,0.0006532887,0.0003047655,0.0007462393,0.004584885,0.01613797,0.07219601,0.02292736,0.005800176,0.8222137],"study_design_scores_gemma":[0.0002122887,0.001209968,0.09342878,0.0003976263,0.0007932283,0.003291013,0.01098898,0.5445625,0.1081356,0.1989639,0.03758973,0.0004264285],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.6440349,0.0005587795,0.3426976,0.0002769597,0.00008798077,0.000386093,0.001239294,0.002006413,0.008712041],"genre_scores_gemma":[0.7397361,0.0002271199,0.2533893,0.0001342227,0.00005821024,0.0001856866,0.003948232,0.0001899702,0.002131151],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.004696778,"threshold_uncertainty_score":0.009739876,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03354109724348471,"score_gpt":0.3110713321578363,"score_spread":0.2775302349143516,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}