{"id":"W7011771313","doi":"","title":"A new type of typology: improving the utility of linguistic typology in natural language processing via continuous data representations","year":2024,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"McGill University","keywords":"Typology; Natural language; Linguistic typology; Deep linguistic processing; Type (biology); Language identification","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01773831,0.001172109,0.001851742,0.006639398,0.002219675,0.0130345,0.003723649,0.002800117,0.007283539],"category_scores_gemma":[0.09952279,0.0008301028,0.0023943,0.01174568,0.006416534,0.03020083,0.009376844,0.007952227,0.002069891],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001764438,"about_ca_system_score_gemma":0.002852955,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001896034,"about_ca_topic_score_gemma":0.00176005,"domain_scores_codex":[0.9814419,0.01024442,0.001620319,0.003852133,0.002410801,0.0004303966],"domain_scores_gemma":[0.9024956,0.05749489,0.004017694,0.02918897,0.005420358,0.001382428],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0005200009,0.0004727294,0.02287062,0.001016305,0.0003365107,0.0002465887,0.004693865,0.01397513,0.004199794,0.3353394,0.01980577,0.5965233],"study_design_scores_gemma":[0.0001059845,0.0003102951,0.004640603,0.0006442998,0.0001508702,0.0004530111,0.003214499,0.1735529,0.005105399,0.7600018,0.05157704,0.0002431627],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02289812,0.001038289,0.9624018,0.004081323,0.0005745623,0.0002959942,0.001543559,0.001246326,0.005919995],"genre_scores_gemma":[0.1795963,0.00113435,0.8107633,0.00131956,0.0006618553,0.0007288064,0.00279864,0.0007079649,0.002289241],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01773831,"threshold_uncertainty_score":0.09381026,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02451388020496683,"score_gpt":0.324075701020664,"score_spread":0.2995618208156972,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}