{"id":"W6976084273","doi":"10.60692/2crmm-aaz77","title":"AfroLM: A Self-Active Learning-based Multilingual Pretrained Language Model for 23 African Languages","year":2022,"lang":"en","type":"article","venue":"Greater South Information System","topic":"Nanotechnology research and applications","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Language model; Context (archaeology); Code (set theory); Training set; Language identification; Natural language; Code-switching; Downstream (manufacturing)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002079392,0.000130555,0.0001474885,0.0002730602,0.0002565691,0.00004324973,0.0001973683,0.00005826986,0.00001473919],"category_scores_gemma":[0.00002321755,0.0001302504,0.00006518305,0.0002550874,0.00001630148,0.0001495384,0.0000433217,0.0002379167,0.00005336714],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001959131,"about_ca_system_score_gemma":0.00004715223,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000003387596,"about_ca_topic_score_gemma":6.17438e-7,"domain_scores_codex":[0.9991361,0.00002983767,0.000248821,0.0001091127,0.000198162,0.0002779153],"domain_scores_gemma":[0.9995596,0.00001859005,0.00007466858,0.0002202398,0.00006381507,0.00006304],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001200379,0.000008551438,0.0007268838,0.0005280481,0.0001068362,0.000003487105,0.2023086,0.7938911,0.0001979925,0.0001431732,0.0003591371,0.001606187],"study_design_scores_gemma":[0.0008449013,0.00003526974,0.00008221825,0.00000803872,0.00001236771,0.000005018145,0.05166057,0.9425917,0.004112907,7.935266e-7,0.0005077567,0.000138516],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7818349,0.00001308635,0.2116944,0.00007322578,0.00006506259,0.001310482,0.001156582,0.002583192,0.001269044],"genre_scores_gemma":[0.9963699,4.603294e-8,0.001444502,0.00002333984,0.00002392248,0.001732245,0.0001620199,0.00002164459,0.0002223737],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.214535,"threshold_uncertainty_score":0.5311458,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01695083142444189,"score_gpt":0.2339535317920627,"score_spread":0.2170027003676208,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}