{"id":"W1572771042","doi":"10.1007/978-3-540-30586-6_89","title":"Automatic Language Identification Using Multivariate Analysis","year":2005,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"ca_institutions":"BC Research (Canada)","funders":"","keywords":"Computer science; Identification (biology); Multivariate statistics; Character (mathematics); Artificial intelligence; Language identification; Curse of dimensionality; Dimensionality reduction; Natural language processing; Natural language; Machine learning; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007199391,0.00103877,0.000986871,0.002547568,0.0008131686,0.001789521,0.0009305382,0.0006189956,0.01252484],"category_scores_gemma":[0.00251187,0.0004292682,0.001251181,0.002274024,0.0005480771,0.002244716,0.001802808,0.001367619,0.01102992],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003260321,"about_ca_system_score_gemma":0.0007606547,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001023554,"about_ca_topic_score_gemma":0.001435396,"domain_scores_codex":[0.9991236,0.0002137083,0.00005982641,0.0002204137,0.0002558703,0.0001265754],"domain_scores_gemma":[0.998827,0.0005005172,0.00009510497,0.0002232271,0.0003006959,0.0000534736],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002093199,0.00007011511,0.0008728939,0.0001474976,0.00005024817,0.0001452766,0.0001429323,0.004259808,0.05329613,0.01354396,0.01030263,0.9169591],"study_design_scores_gemma":[0.00005160396,0.0001680306,0.005603171,0.00009756979,0.0001756429,0.001383825,0.0005005333,0.7326322,0.1226227,0.08641984,0.05017269,0.0001723437],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01032557,0.0003574234,0.9779273,0.0001742606,0.0001083539,0.0000407887,0.0005396325,0.007189722,0.003336892],"genre_scores_gemma":[0.1857179,0.0008396233,0.7891427,0.0001694205,0.0002426297,0.0002006689,0.003523741,0.002150082,0.01801335],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01252484,"threshold_uncertainty_score":0.0418998,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01789968486689885,"score_gpt":0.2843767554908084,"score_spread":0.2664770706239096,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}