{"id":"W4385573074","doi":"10.18653/v1/2022.emnlp-demos.32","title":"Camelira: An Arabic Multi-Dialect Morphological Disambiguator","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"York University; New York University Abu Dhabi","keywords":"Arabic; Computer science; Component (thermodynamics); Modern Standard Arabic; Natural language processing; Identification (biology); Interface (matter); Artificial intelligence; Linguistics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003282071,0.0001232442,0.0001320185,0.00008725389,0.0003042573,0.0001264899,0.001557006,0.00004204574,0.0003077412],"category_scores_gemma":[0.0000674159,0.00009723237,0.0000517871,0.0004316205,0.00004139368,0.0004234425,0.0008833586,0.000330332,0.00002040973],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008178503,"about_ca_system_score_gemma":0.00005217786,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001229079,"about_ca_topic_score_gemma":0.00001371121,"domain_scores_codex":[0.9986541,0.0001567499,0.0001514409,0.0004617836,0.0003051449,0.0002708017],"domain_scores_gemma":[0.9992303,0.00003895398,0.0000525666,0.0005475653,0.00003699117,0.00009366311],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00007015852,0.002460987,0.004021533,0.00003944259,0.00004591994,0.002589163,0.003780567,0.0002450408,0.1362596,0.3773024,0.02277598,0.4504093],"study_design_scores_gemma":[0.003178046,0.004471244,0.006215889,0.00002684109,0.00003898554,0.003066229,0.0006633623,0.6171095,0.1462329,0.1896181,0.02535632,0.004022601],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0444704,0.0008001138,0.9504418,0.0009827913,0.0002707725,0.0001910072,0.000004216799,0.002355886,0.0004829743],"genre_scores_gemma":[0.569836,0.000001190348,0.4283977,0.001248297,0.00002440006,0.00004995918,0.00000360606,0.000005959876,0.0004328981],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.6168644,"threshold_uncertainty_score":0.3965021,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02902459647331002,"score_gpt":0.2951499694393067,"score_spread":0.2661253729659967,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}