{"id":"W6891761579","doi":"10.48448/gyjr-wh13","title":"Dolphin: A Challenging and Diverse Benchmark for Arabic NLG","year":2023,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Benchmark (surveying); Arabic; Feature (linguistics); Matching (statistics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007992018,0.0002850014,0.0002805056,0.0009077983,0.0003321142,0.0003961257,0.001962909,0.0001850808,0.00001931332],"category_scores_gemma":[0.0002728504,0.0002415805,0.00004976551,0.001027619,0.0004884656,0.0005584563,0.0009348117,0.0002462058,0.00002152245],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009801342,"about_ca_system_score_gemma":0.0002122126,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00009260508,"about_ca_topic_score_gemma":0.0001969893,"domain_scores_codex":[0.9977347,0.00001564528,0.0001989951,0.0009723111,0.0005336415,0.0005446597],"domain_scores_gemma":[0.998751,0.0001356507,0.000200265,0.0006503445,0.0001127604,0.0001499847],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000005823176,0.00007478972,0.00006070845,0.0004531282,0.00004449331,0.00007935835,0.001211036,0.00000664584,0.0008102804,0.3148733,0.09278559,0.5895948],"study_design_scores_gemma":[0.001873077,0.0007610411,0.0001094135,0.003563106,0.0001210474,0.0001714662,0.0005599015,0.3145424,0.002666957,0.2873796,0.383885,0.004366991],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00002729519,0.008004371,0.9488562,0.002376399,0.001836563,0.001202457,0.00004547945,0.00774671,0.02990456],"genre_scores_gemma":[0.007065575,0.0004785905,0.9213993,0.000431296,0.0004929432,0.00008409371,0.000009731215,0.0003566342,0.06968187],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.5852278,"threshold_uncertainty_score":0.9851366,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02330809533530947,"score_gpt":0.3025879152959176,"score_spread":0.2792798199606081,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}