{"id":"W4389519409","doi":"10.18653/v1/2023.emnlp-main.16","title":"GPTAraEval: A Comprehensive Evaluation of ChatGPT on Arabic NLP","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":48,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Arabic; Modern Standard Arabic; Computer science; Natural language processing; Focus (optics); Transformative learning; Artificial intelligence; Software deployment; Linguistics; Psychology; Software engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005579563,0.002486705,0.001185196,0.002236767,0.001349973,0.002550181,0.003387009,0.002276455,0.009981016],"category_scores_gemma":[0.02611868,0.000513944,0.00135744,0.001727732,0.0009579686,0.004853565,0.003713819,0.003225981,0.005967899],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001812755,"about_ca_system_score_gemma":0.002290177,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02208818,"about_ca_topic_score_gemma":0.02474077,"domain_scores_codex":[0.9948778,0.002564279,0.0003634567,0.001068671,0.0009400328,0.000185685],"domain_scores_gemma":[0.9858972,0.009670503,0.0002416765,0.001860785,0.001714599,0.0006152497],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.002748484,0.001444921,0.01504144,0.006717127,0.001452568,0.001908281,0.003869497,0.138748,0.0159722,0.006727955,0.2698854,0.5354841],"study_design_scores_gemma":[0.0006084316,0.001046873,0.009988479,0.0005108398,0.0003445393,0.001001994,0.002196362,0.8498735,0.01436188,0.008014932,0.1118326,0.0002196337],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.3780051,0.009069833,0.239457,0.00467458,0.002465642,0.002963556,0.05967126,0.2539926,0.04970032],"genre_scores_gemma":[0.5803208,0.001988632,0.2246479,0.001470516,0.0002852589,0.001548692,0.1680907,0.008238233,0.0134093],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02208818,"threshold_uncertainty_score":0.04391921,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.136394988307044,"score_gpt":0.3458940014193512,"score_spread":0.2094990131123072,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}