{"id":"W4385572949","doi":"10.18653/v1/2022.wanlp-1.49","title":"On The Arabic Dialects’ Identification: Overcoming Challenges of Geographical Similarities Between Arabic dialects and Imbalanced Datasets","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Windsor","funders":"Vector Institute; Compute Canada","keywords":"Arabic; Preprocessor; Computer science; Natural language processing; Artificial intelligence; Macro; Identification (biology); Task (project management); Entropy (arrow of time); Linguistics; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006673339,0.001421105,0.001860083,0.004477846,0.002004574,0.003689146,0.002218428,0.001859855,0.0008039242],"category_scores_gemma":[0.02017775,0.0003644301,0.0007908956,0.004846173,0.000996838,0.004630406,0.004648727,0.001804794,0.001045537],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008412564,"about_ca_system_score_gemma":0.001947501,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009067715,"about_ca_topic_score_gemma":0.01266425,"domain_scores_codex":[0.9937152,0.002257201,0.0004807068,0.001657928,0.001417851,0.0004710756],"domain_scores_gemma":[0.9891251,0.004734174,0.0008569593,0.002304852,0.002502386,0.000476526],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001829646,0.0009806274,0.06689723,0.0007865318,0.0007377783,0.001153588,0.002818382,0.03419963,0.01749421,0.01024441,0.06233197,0.800526],"study_design_scores_gemma":[0.0002370976,0.0003236672,0.05805616,0.0002403195,0.0004228649,0.001211378,0.01271303,0.8090285,0.01622524,0.04637772,0.05502341,0.0001405994],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.663888,0.01001334,0.2947344,0.009354462,0.001629494,0.0006421959,0.008050979,0.003446027,0.00824116],"genre_scores_gemma":[0.8413025,0.00163204,0.137272,0.001110503,0.0008567667,0.0002122166,0.01429121,0.000232687,0.003090165],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.009067715,"threshold_uncertainty_score":0.03529239,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02296168059963013,"score_gpt":0.2617181005923733,"score_spread":0.2387564199927432,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}