{"id":"W4402671205","doi":"10.18653/v1/2024.arabicnlp-1.4","title":"Exploiting Dialect Identification in Automatic Dialectal Text Normalization","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"York University; New York University Abu Dhabi","keywords":"Computer science; Normalization (sociology); Natural language processing; Identification (biology); Artificial intelligence; Speech recognition; Biology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001415817,0.001390657,0.0008071055,0.001471642,0.0007501474,0.001848858,0.001111676,0.001120435,0.004479496],"category_scores_gemma":[0.004249761,0.0004595937,0.00118481,0.001038635,0.0007062312,0.002353629,0.001417613,0.001993794,0.006547916],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009148158,"about_ca_system_score_gemma":0.001064043,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008580999,"about_ca_topic_score_gemma":0.01324372,"domain_scores_codex":[0.9988186,0.0003370139,0.00007980014,0.0005573483,0.0001148956,0.00009234212],"domain_scores_gemma":[0.9981173,0.0008505358,0.00008555451,0.0004002472,0.0004724834,0.00007388903],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0008579729,0.0004022521,0.01091099,0.0004305254,0.0002669824,0.0006511821,0.001249353,0.05079474,0.08316084,0.004083489,0.02742554,0.819766],"study_design_scores_gemma":[0.00007102273,0.0002362891,0.008562129,0.00006059066,0.0001040312,0.0009313544,0.0006758432,0.8822383,0.06959412,0.01035234,0.02706147,0.0001124301],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3118158,0.002775527,0.6272297,0.0009140573,0.001093333,0.0005229094,0.005818528,0.03824314,0.01158705],"genre_scores_gemma":[0.6495091,0.0006651371,0.3210439,0.0005228559,0.0001770472,0.0003097582,0.01444543,0.001658813,0.01166791],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008580999,"threshold_uncertainty_score":0.01706213,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01044162544620815,"score_gpt":0.2726622949550494,"score_spread":0.2622206695088413,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}