{"id":"W3098826124","doi":"10.1109/taslp.2024.3426331","title":"Overview of the Ninth Dialog System Technology Challenge: DSTC9","year":2024,"lang":"en","type":"article","venue":"IEEE/ACM Transactions on Audio Speech and Language Processing","topic":"Speech and dialogue systems","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Dialog box; Computer science; Task (project management); Dialog system; Domain (mathematical analysis); Set (abstract data type); Human–computer interaction; Natural language processing; Artificial intelligence; Multimedia; World Wide Web; Programming language; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03080283,0.002898243,0.002978173,0.005515925,0.00398542,0.009346928,0.006008571,0.00557578,0.01565765],"category_scores_gemma":[0.03673344,0.001251074,0.00216709,0.004708015,0.001986367,0.01435743,0.01121078,0.009952625,0.01968193],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.007614073,"about_ca_system_score_gemma":0.01986665,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03072348,"about_ca_topic_score_gemma":0.03364677,"domain_scores_codex":[0.9757339,0.007372221,0.00187402,0.002924839,0.0101971,0.001897966],"domain_scores_gemma":[0.9620299,0.007450946,0.000788357,0.006023554,0.0179785,0.005728812],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000523538,0.0008331122,0.001576546,0.002630282,0.000170345,0.0001781029,0.0007799809,0.005151553,0.008058876,0.01313639,0.6867496,0.2802117],"study_design_scores_gemma":[0.0001335889,0.0007101234,0.002898722,0.0007498506,0.00008492627,0.000423705,0.0007527162,0.0209187,0.01147035,0.009456193,0.9522365,0.0001644565],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"other","genre_scores_codex":[0.04497108,0.06785179,0.5012492,0.06292696,0.03608163,0.01072802,0.1014316,0.0634568,0.1113029],"genre_scores_gemma":[0.06093185,0.01709005,0.3751514,0.00799458,0.003268339,0.005946643,0.4592398,0.005860714,0.06451669],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.03080283,"threshold_uncertainty_score":0.1629029,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02095451411627129,"score_gpt":0.2734351245595172,"score_spread":0.2524806104432459,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}