{"id":"W4402671393","doi":"10.18653/v1/2024.arabicnlp-1.80","title":"Arabic Train at NADI 2024 shared task: LLMs’ Ability to Translate Arabic Dialects into Modern Standard Arabic","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Arabic; Task (project management); Modern Standard Arabic; Computer science; Arabic languages; Natural language processing; Linguistics; History; Engineering; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003534959,0.001514228,0.0009196501,0.0007524961,0.001074982,0.001940253,0.001492574,0.001987435,0.009372615],"category_scores_gemma":[0.01632792,0.0003039356,0.0006251741,0.0006117859,0.0006504506,0.003727657,0.003589181,0.002467097,0.007892421],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001147907,"about_ca_system_score_gemma":0.001849279,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01214248,"about_ca_topic_score_gemma":0.01762598,"domain_scores_codex":[0.9979482,0.0009218167,0.0001405645,0.0005451566,0.0002650895,0.0001790835],"domain_scores_gemma":[0.993391,0.003297047,0.0002378366,0.001606279,0.000805222,0.0006626273],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.004077401,0.002072843,0.02822529,0.001510266,0.0004821714,0.001147865,0.004462447,0.03471915,0.0505622,0.00451331,0.1918861,0.676341],"study_design_scores_gemma":[0.001179739,0.004284585,0.05928546,0.0004231901,0.0004617332,0.002185184,0.008565768,0.5724886,0.1423727,0.02542769,0.1826356,0.0006898503],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8713946,0.001935458,0.03091636,0.002141059,0.0009131412,0.0005117212,0.01356577,0.01789925,0.0607227],"genre_scores_gemma":[0.9306645,0.0001639705,0.03069461,0.0007217292,0.00007555525,0.0003138653,0.02267311,0.0005087536,0.0141838],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01214248,"threshold_uncertainty_score":0.03135455,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01127465623214217,"score_gpt":0.2841371138451353,"score_spread":0.2728624576129931,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}