{"id":"W3205231356","doi":"10.18653/v1/2022.findings-acl.135","title":"Morphosyntactic Tagging with Pre-trained Language Models for Arabic and its Dialects","year":2022,"lang":"en","type":"article","venue":"Findings of the Association for Computational Linguistics: ACL 2022","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"York University; New York University Abu Dhabi","keywords":"Transformer; Arabic; Computer science; Modern Standard Arabic; Natural language processing; Artificial intelligence; Language model; Training set; Resource (disambiguation); Linguistics; Engineering; Voltage","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007076722,0.0001272026,0.0001852497,0.0001321492,0.0005454268,0.00009623699,0.0006181544,0.0000414701,0.000003314238],"category_scores_gemma":[0.002809485,0.0001109158,0.00008221796,0.0003407768,0.00001501793,0.0001271208,0.0003182719,0.0001819536,1.99851e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003197362,"about_ca_system_score_gemma":0.0001520891,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001199651,"about_ca_topic_score_gemma":0.000002460421,"domain_scores_codex":[0.9986441,0.00005614358,0.0002474711,0.0002889278,0.0005473695,0.0002159645],"domain_scores_gemma":[0.9976488,0.00110271,0.0005146643,0.0001397391,0.0005615843,0.00003253887],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001614651,0.0001367514,0.00112823,0.0002697315,0.0001930484,0.000002536809,0.004446344,0.10071,0.001440132,0.8883649,0.002470359,0.0006765378],"study_design_scores_gemma":[0.0008755852,0.0002191018,0.0005516494,0.00004449783,0.00006236912,0.000007341939,0.00006195276,0.7444204,0.002867074,0.2495728,0.001070178,0.0002470729],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06648106,0.001376306,0.9234148,0.002836395,0.001339674,0.002644379,0.001037249,0.000498022,0.0003721483],"genre_scores_gemma":[0.8595336,0.000001028737,0.1393979,0.0002580985,0.0000943464,0.0001371775,0.00006717026,0.00001789391,0.0004927909],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.7930526,"threshold_uncertainty_score":0.4523014,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.008924129542520672,"score_gpt":0.2522943333195564,"score_spread":0.2433702037770357,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}