{"id":"W6891761579","doi":"10.48448/gyjr-wh13","title":"Dolphin: A Challenging and Diverse Benchmark for Arabic NLG","year":2023,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Benchmark (surveying); Arabic; Feature (linguistics); Matching (statistics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003262056,0.002720085,0.001215625,0.005255064,0.002892916,0.003538584,0.002918715,0.0037886,0.02843759],"category_scores_gemma":[0.01694636,0.0003813342,0.001056603,0.003733199,0.001500703,0.005402143,0.004518057,0.002317956,0.01881558],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002047062,"about_ca_system_score_gemma":0.002716144,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01840353,"about_ca_topic_score_gemma":0.02169298,"domain_scores_codex":[0.996213,0.001351842,0.0004795897,0.0008344668,0.0008509182,0.0002702979],"domain_scores_gemma":[0.9922606,0.003703032,0.0001794195,0.001322697,0.001949428,0.0005847646],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001281757,0.0006109299,0.0023194,0.003006304,0.0001982931,0.001376837,0.0009063162,0.02503001,0.008220386,0.01448519,0.6148122,0.3277524],"study_design_scores_gemma":[0.00110079,0.0005718056,0.005203351,0.0007616332,0.0001476,0.002406166,0.003723969,0.2865066,0.03296023,0.04556105,0.6208304,0.0002263834],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"other","genre_gemma":"methods","genre_scores_codex":[0.2243636,0.01180581,0.1250088,0.009393361,0.004667368,0.00272868,0.216003,0.1735853,0.2324441],"genre_scores_gemma":[0.2626445,0.002368399,0.2185384,0.00233078,0.0004627533,0.001156782,0.4664156,0.01053693,0.03554588],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.02843759,"threshold_uncertainty_score":0.09513319,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02330809533530947,"score_gpt":0.3025879152959176,"score_spread":0.2792798199606081,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}