{"id":"W6891700532","doi":"10.48448/fjkm-b086","title":"AfroLM: A Self-Active Learning-based Multilingual Pretrained Language Model for 23 African Languages","year":2022,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Language model; Context (archaeology); Code (set theory); Training set; Natural language; Language identification; Downstream (manufacturing); Code-switching","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001758466,0.0008303634,0.00078646,0.002124222,0.0007244649,0.000202751,0.001917682,0.0003236705,0.002834737],"category_scores_gemma":[0.001533529,0.0008205187,0.000256007,0.002058358,0.001094046,0.0002167151,0.0004235978,0.001077404,0.0003556451],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001267706,"about_ca_system_score_gemma":0.003092631,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005332676,"about_ca_topic_score_gemma":0.002189296,"domain_scores_codex":[0.9941427,0.0002147645,0.0004871254,0.001847593,0.001858594,0.001449199],"domain_scores_gemma":[0.9970626,0.0003689825,0.0008271456,0.001069044,0.0002639731,0.0004082862],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00212902,0.006014114,0.0003960872,0.001587787,0.001389733,0.0005829114,0.1179015,0.4585008,0.06005681,0.002784349,0.3035479,0.0451089],"study_design_scores_gemma":[0.001985682,0.0003832415,0.000009874677,0.00006005098,0.0001546931,0.000007148486,0.01075412,0.9293417,0.0008382042,0.00004460048,0.05540299,0.001017699],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.01131853,0.002523932,0.05179544,0.0008328704,0.001640756,0.01619888,0.02423566,0.02111616,0.8703378],"genre_scores_gemma":[0.4800203,0.000007517052,0.08929037,0.0004584194,0.0007948006,0.001022649,0.002820363,0.003176046,0.4224095],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.4708409,"threshold_uncertainty_score":0.9994246,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02007904407409088,"score_gpt":0.3213358809207112,"score_spread":0.3012568368466204,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}