{"id":"W3205231356","doi":"10.18653/v1/2022.findings-acl.135","title":"Morphosyntactic Tagging with Pre-trained Language Models for Arabic and its Dialects","year":2022,"lang":"en","type":"article","venue":"Findings of the Association for Computational Linguistics: ACL 2022","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"York University; New York University Abu Dhabi","keywords":"Transformer; Arabic; Computer science; Modern Standard Arabic; Natural language processing; Artificial intelligence; Language model; Training set; Resource (disambiguation); Linguistics; Engineering; Voltage","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001590058,0.002446197,0.0008345883,0.001559214,0.001009802,0.002342101,0.001837591,0.001221626,0.006922849],"category_scores_gemma":[0.006235406,0.0007230107,0.001233521,0.001292635,0.0006431009,0.003510089,0.001912848,0.002592636,0.009834396],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001289223,"about_ca_system_score_gemma":0.001387838,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01410134,"about_ca_topic_score_gemma":0.0216637,"domain_scores_codex":[0.998902,0.000272595,0.00008983855,0.0005129444,0.0001233097,0.00009931322],"domain_scores_gemma":[0.9966467,0.00165068,0.0001111057,0.0007681766,0.0007044837,0.0001188327],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001080288,0.0005390086,0.01312763,0.001027456,0.000710432,0.0009175782,0.001414559,0.1035979,0.05795325,0.003359026,0.03956905,0.7767038],"study_design_scores_gemma":[0.000159145,0.0003236316,0.008412139,0.0002231477,0.0004567699,0.0008174787,0.001291554,0.8382786,0.09793703,0.01012375,0.04172909,0.0002476256],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.4244402,0.004453862,0.4374241,0.001193125,0.001477309,0.0004399508,0.01598232,0.09113926,0.02344993],"genre_scores_gemma":[0.6342845,0.0012259,0.3018605,0.0006494557,0.0001262914,0.0002666144,0.04681075,0.003675458,0.01110055],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01410134,"threshold_uncertainty_score":0.0280385,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.008924129542520672,"score_gpt":0.2522943333195564,"score_spread":0.2433702037770357,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}