{"id":"W4385572818","doi":"10.18653/v1/2022.emnlp-main.128","title":"AfroLID: A Neural Language Identification Tool for African Languages","year":2022,"lang":"en","type":"article","venue":"","topic":"Authorship Attribution and Profiling","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Advanced Micro Devices","keywords":"Computer science; Domain (mathematical analysis); Identification (biology); Natural language processing; Set (abstract data type); Cover (algebra); Artificial intelligence; Exploit; Language identification; Natural language; Programming language; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00180814,0.00121622,0.0005742416,0.003986715,0.0009952963,0.001554157,0.001426511,0.001060866,0.01433393],"category_scores_gemma":[0.007219915,0.0004062756,0.0008179242,0.001541238,0.0004746787,0.003102521,0.003583293,0.001461606,0.01044192],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000674888,"about_ca_system_score_gemma":0.001389164,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005347307,"about_ca_topic_score_gemma":0.01222241,"domain_scores_codex":[0.9989303,0.0002411444,0.0001267943,0.0003188246,0.0002718379,0.0001110665],"domain_scores_gemma":[0.9978291,0.001052127,0.0001671977,0.0004151531,0.0004280297,0.0001083577],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004163014,0.0002421466,0.01384628,0.001058632,0.0001454574,0.0005007813,0.001199797,0.006982852,0.01833751,0.004469713,0.140123,0.8126776],"study_design_scores_gemma":[0.0002444244,0.0004095107,0.02064799,0.000577952,0.0001582605,0.002955642,0.003184123,0.5470566,0.0947416,0.02834777,0.3013426,0.0003335278],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1408201,0.002380594,0.503893,0.001467874,0.001040774,0.0008119686,0.0479851,0.2693668,0.03223384],"genre_scores_gemma":[0.3573421,0.0006389169,0.5509309,0.0007500783,0.0001168495,0.001005248,0.06057357,0.003678045,0.02496437],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01433393,"threshold_uncertainty_score":0.04795176,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02260801631974186,"score_gpt":0.3036981982625557,"score_spread":0.2810901819428138,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}