{"id":"W4389518309","doi":"10.18653/v1/2023.arabicnlp-1.6","title":"TARJAMAT: Evaluation of Bard and ChatGPT on Machine Translation of Ten Arabic Varieties","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Arabic; Constraint (computer-aided design); Computer science; Linguistics; Modern Standard Arabic; Natural language processing; Machine translation; Artificial intelligence; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004124805,0.001858508,0.0009536931,0.001260873,0.001022326,0.002237093,0.00195458,0.001911446,0.006691444],"category_scores_gemma":[0.01314776,0.0003997672,0.0008445,0.001214528,0.0007594929,0.002888975,0.00223264,0.002179028,0.006152085],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00128661,"about_ca_system_score_gemma":0.001606937,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01010578,"about_ca_topic_score_gemma":0.01218877,"domain_scores_codex":[0.9970791,0.001339646,0.0003015627,0.0007847313,0.0003540429,0.0001408697],"domain_scores_gemma":[0.9943903,0.003182361,0.0001589807,0.0008212471,0.001060465,0.0003866384],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.005698754,0.001966319,0.01473822,0.003878973,0.001003003,0.0009959965,0.002284807,0.07538374,0.02477821,0.003345747,0.09411094,0.7718154],"study_design_scores_gemma":[0.001636851,0.004230402,0.02288095,0.0006590227,0.000580684,0.001123076,0.00386665,0.8230035,0.05380603,0.006390956,0.08147902,0.0003428487],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7895118,0.008234525,0.07424331,0.001846805,0.00198974,0.001204685,0.0190683,0.07474978,0.02915109],"genre_scores_gemma":[0.8003255,0.00135327,0.1148871,0.0008575559,0.000163445,0.000758705,0.06828087,0.002330492,0.01104306],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01010578,"threshold_uncertainty_score":0.02238512,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03261906517732031,"score_gpt":0.3079726270094847,"score_spread":0.2753535618321644,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}