{"id":"W4385565328","doi":"10.18653/v1/2023.americasnlp-1.15","title":"Finding words that aren’t there: Using word embeddings to improve dictionary search for low-resource languages","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Indigenous; Word (group theory); Natural language processing; Resource (disambiguation); Artificial intelligence; Linguistics; Bilingual dictionary; Natural language; Biology; Philosophy; Ecology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001131445,0.001372362,0.001657704,0.003107882,0.001072477,0.001872697,0.001766741,0.001483164,0.004011364],"category_scores_gemma":[0.007039282,0.0007422454,0.0009315916,0.003566145,0.000852105,0.007859856,0.002811527,0.001555594,0.00377361],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003851064,"about_ca_system_score_gemma":0.0009833772,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004073185,"about_ca_topic_score_gemma":0.0116705,"domain_scores_codex":[0.9987558,0.0003429695,0.0001977333,0.0003656461,0.000207838,0.0001299326],"domain_scores_gemma":[0.9961659,0.001798945,0.0002996778,0.0007883878,0.0007557358,0.0001913274],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001269532,0.0005409982,0.01309246,0.0009091321,0.0002977814,0.0006746175,0.001185985,0.0114481,0.0282033,0.008725562,0.04733793,0.8863145],"study_design_scores_gemma":[0.0006827816,0.001342822,0.006151731,0.000445677,0.0006519862,0.002101848,0.007061817,0.7619556,0.03183786,0.1223494,0.06515003,0.0002685093],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3923486,0.009499157,0.5631265,0.002241693,0.001612509,0.0003612065,0.005690669,0.01562214,0.009497481],"genre_scores_gemma":[0.5455276,0.001977557,0.4260068,0.001050049,0.0003052113,0.0001917325,0.01673149,0.001920681,0.006288973],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004073185,"threshold_uncertainty_score":0.01341933,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0331132668861957,"score_gpt":0.3380253507598908,"score_spread":0.3049120838736951,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}