{"id":"W2970485137","doi":"10.18653/v1/w19-4637","title":"No Army, No Navy: BERT Semi-Supervised Learning of Arabic Dialects","year":2019,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":36,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Compute Canada","keywords":"Porting; Computer science; Arabic; Artificial intelligence; Natural language processing; Identification (biology); Navy; Macro; Supervised learning; Task (project management); F1 score; Set (abstract data type); Machine learning; Engineering; Linguistics; History; Artificial neural network; Programming language","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002558251,0.0001469268,0.0002236063,0.0001071392,0.0000484187,0.00008850424,0.0009775271,0.0000951994,0.0002058534],"category_scores_gemma":[0.0001875888,0.0001176373,0.0000709939,0.0003652012,0.00002940758,0.000525572,0.0003558549,0.0002811948,0.0006006914],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003109577,"about_ca_system_score_gemma":0.00006982312,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00009201145,"about_ca_topic_score_gemma":0.000002371099,"domain_scores_codex":[0.9987887,0.00005953284,0.0002128803,0.0003646701,0.0003053551,0.0002689289],"domain_scores_gemma":[0.9989693,0.000122902,0.0001032612,0.0004894102,0.0002535318,0.00006161151],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00005453372,0.0001956662,0.0142782,0.0005576964,0.00006027229,0.00003575016,0.001746644,0.00006150171,0.8358752,0.05901367,0.00656009,0.08156082],"study_design_scores_gemma":[0.001235197,0.001072973,0.0008943402,0.0005645384,0.00001821863,0.00004484482,0.00003313307,0.2759763,0.6867069,0.0119221,0.02028576,0.001245679],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2502022,0.002750248,0.5454324,0.0008415648,0.001496232,0.0009482545,0.00000167313,0.004269034,0.1940584],"genre_scores_gemma":[0.7004496,0.00001209435,0.2874792,0.0003248396,0.00005149957,0.000005589912,0.000002014077,0.0000125866,0.01166254],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.4502474,"threshold_uncertainty_score":0.7720873,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.007084348724772917,"score_gpt":0.2356253151227733,"score_spread":0.2285409663980004,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}