{"id":"W3176169354","doi":"10.18653/v1/2021.acl-long.551","title":"ARBERT &amp; MARBERT: Deep Bidirectional Transformers for Arabic","year":2021,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":352,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Compute Canada","keywords":"Arabic; Transformer; Computer science; Natural language processing; Linguistics; Artificial intelligence; Speech recognition; Electrical engineering; Engineering; Philosophy; Voltage","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004288811,0.001277789,0.0005017632,0.001025157,0.0008145044,0.002055159,0.00127468,0.0008686268,0.04826551],"category_scores_gemma":[0.0015774,0.0007814677,0.000946492,0.000791222,0.0005474907,0.006130095,0.002037643,0.00175423,0.01833704],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007569564,"about_ca_system_score_gemma":0.0007547188,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005249097,"about_ca_topic_score_gemma":0.01340077,"domain_scores_codex":[0.9997951,0.00004894977,0.00001536807,0.00005947737,0.00005079709,0.00003029001],"domain_scores_gemma":[0.9996636,0.0001275425,0.00001589097,0.00008676078,0.00008072062,0.00002564035],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0006389768,0.0001025436,0.0007182117,0.0004742463,0.00008044076,0.000461668,0.0005039183,0.01285183,0.01372084,0.09503891,0.2871666,0.5882418],"study_design_scores_gemma":[0.0002031984,0.0001132504,0.0005828761,0.0001980522,0.0001029239,0.0004963652,0.0004440203,0.3194001,0.05625177,0.2882971,0.3338153,0.00009497983],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01623176,0.002260812,0.8129079,0.001796991,0.000931973,0.00015036,0.007943547,0.1135448,0.04423179],"genre_scores_gemma":[0.3332583,0.001724971,0.5817866,0.000833377,0.0002358029,0.0002392809,0.01438654,0.01209677,0.05543839],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.04826551,"threshold_uncertainty_score":0.1614642,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02856709349934125,"score_gpt":0.2625137650090602,"score_spread":0.233946671509719,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}