{"id":"W6891697581","doi":"10.48448/qt4q-cq81","title":"Dolphin: A Challenging and Diverse Benchmark for Arabic NLG","year":2023,"lang":"en","type":"other","venue":"Open MIND","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Benchmark (surveying); Modular design; Arabic; Generalization; Set (abstract data type); Modern Standard Arabic; Range (aeronautics); Test set","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009465149,0.003774884,0.001350172,0.006431325,0.003227587,0.00473181,0.004677709,0.0037642,0.01596703],"category_scores_gemma":[0.03402899,0.0006213547,0.001608483,0.005950714,0.002058948,0.005854897,0.006121919,0.002947983,0.01334295],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003255861,"about_ca_system_score_gemma":0.003684534,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02256027,"about_ca_topic_score_gemma":0.02574071,"domain_scores_codex":[0.9903107,0.004174372,0.0009474184,0.001830738,0.002191902,0.0005448568],"domain_scores_gemma":[0.9857868,0.006123545,0.0004591472,0.003492516,0.003248618,0.0008893627],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00118334,0.001021489,0.006636972,0.002872298,0.0003677368,0.0007539089,0.001372812,0.06632726,0.005068377,0.01851538,0.5918011,0.3040794],"study_design_scores_gemma":[0.0009872027,0.0007047392,0.008053734,0.0009330256,0.0001714535,0.001362735,0.002405443,0.3593013,0.02503039,0.04924685,0.5514991,0.0003040302],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.2121589,0.01685802,0.1821702,0.01099841,0.005790262,0.004280703,0.237815,0.1296337,0.2002948],"genre_scores_gemma":[0.2076302,0.00207997,0.214322,0.002209355,0.0004807454,0.002411058,0.5385732,0.009535555,0.02275803],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.02256027,"threshold_uncertainty_score":0.05341506,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08378304741677689,"score_gpt":0.3435293216621338,"score_spread":0.2597462742453569,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}