{"id":"W2027097848","doi":"10.3115/1626516.1626533","title":"Creating a comparative dictionary of Totonac-Tepehua","year":2007,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Wenner-Gren Foundation","keywords":"Cognate; Computer science; Language family; Identification (biology); Natural language processing; Artificial intelligence; Indigenous; Linguistics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002820485,0.00006610561,0.0001082003,0.00009222118,0.00005624204,0.00002218184,0.0003907497,0.00003495325,0.00002137187],"category_scores_gemma":[0.00002610752,0.00005228416,0.00002968709,0.000322899,0.00003772836,0.0003143064,0.0001426436,0.00008546263,0.000004842393],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002329575,"about_ca_system_score_gemma":0.00002531151,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000493817,"about_ca_topic_score_gemma":0.000009997159,"domain_scores_codex":[0.9993374,0.00001357936,0.0001788924,0.0001539378,0.000186686,0.0001295117],"domain_scores_gemma":[0.9994413,0.0001404051,0.00008073622,0.000196239,0.0001060428,0.00003531982],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00001745972,0.0001242955,0.002847268,0.00003215641,0.00002218083,0.00003115139,0.003839738,0.000006126585,0.06446791,0.8815173,0.001716579,0.04537782],"study_design_scores_gemma":[0.0001605197,0.0001188365,0.00330075,0.00006155129,0.000003541089,0.00004378217,0.0002139387,0.008893157,0.9482997,0.03806173,0.0006429087,0.0001995485],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01600016,0.0005276165,0.9499087,0.00007606344,0.00004614115,0.0000739607,4.018293e-7,0.0004257434,0.0329412],"genre_scores_gemma":[0.5305609,5.628931e-7,0.4691336,0.00007570869,0.00001269471,0.000001303888,4.658964e-7,0.000001371871,0.0002134153],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8838318,"threshold_uncertainty_score":0.2132086,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02095664878535249,"score_gpt":0.3178112487715095,"score_spread":0.296854599986157,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}