{"id":"W7106799280","doi":"10.48448/dwkn-zp74","title":"Less is More: The Effectiveness of Compact Typological Language Representations","year":2025,"lang":"","type":"other","venue":"Open MIND","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Feature (linguistics); Pipeline (software); Feature selection; Space (punctuation); Typology; Linguistic typology; Selection (genetic algorithm)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002876148,0.00111818,0.001058891,0.001732431,0.000908775,0.003303196,0.001714568,0.001550875,0.008787511],"category_scores_gemma":[0.0182649,0.0003509223,0.0009086723,0.002719349,0.0009853796,0.006406978,0.003612593,0.002007326,0.004516993],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006659409,"about_ca_system_score_gemma":0.001085576,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003389071,"about_ca_topic_score_gemma":0.004519777,"domain_scores_codex":[0.9978257,0.0008266006,0.0001532262,0.0006361902,0.0003929572,0.000165314],"domain_scores_gemma":[0.9925911,0.003175956,0.0005022549,0.002635178,0.0009252966,0.0001701392],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008241533,0.0004234047,0.01366167,0.000337854,0.0002194993,0.0003084642,0.000649573,0.05927638,0.01072087,0.02383008,0.03930533,0.8504426],"study_design_scores_gemma":[0.0001787982,0.0002897031,0.007722597,0.0001369993,0.0001056363,0.000392966,0.001045333,0.7873681,0.01652467,0.1462497,0.03985213,0.0001334816],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3098158,0.001387096,0.6351057,0.003681485,0.0004265215,0.0001506855,0.008593789,0.01954774,0.02129111],"genre_scores_gemma":[0.6782117,0.0003303589,0.2932647,0.000833024,0.0001494578,0.0002193854,0.01790789,0.00170916,0.00737432],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008787511,"threshold_uncertainty_score":0.02939713,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06218217010853384,"score_gpt":0.4204429004209905,"score_spread":0.3582607303124567,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}