{"id":"W2963903950","doi":"10.18653/v1/d16-1230","title":"How NOT To Evaluate Your Dialogue System: An Empirical Study of Unsupervised Evaluation Metrics for Dialogue Response Generation","year":2016,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":916,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; McGill University","funders":"","keywords":"Computer science; Empirical research; Artificial intelligence; Machine learning; Natural language processing; Statistics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0751058,0.001401588,0.001135208,0.002512992,0.00169474,0.002574272,0.00162826,0.001556051,0.0009687002],"category_scores_gemma":[0.3490035,0.0004293195,0.0006807425,0.002088597,0.001969258,0.004241251,0.002724957,0.002848998,0.00101577],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002045282,"about_ca_system_score_gemma":0.001492706,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003303829,"about_ca_topic_score_gemma":0.003641384,"domain_scores_codex":[0.8514024,0.1207503,0.006361726,0.008328507,0.01186841,0.001288628],"domain_scores_gemma":[0.4495643,0.4645077,0.0219516,0.02744881,0.03308805,0.003439408],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.005577881,0.003631657,0.2433226,0.003833797,0.00165635,0.0003542115,0.02150789,0.03878148,0.02918665,0.007494257,0.0266714,0.6179819],"study_design_scores_gemma":[0.0006611232,0.01004224,0.3679879,0.001184213,0.0008384131,0.001690172,0.008597869,0.4965291,0.05874176,0.01984653,0.03302925,0.0008514684],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8622093,0.002277613,0.1209213,0.0007072926,0.0002733945,0.001157392,0.001408417,0.002561749,0.008483575],"genre_scores_gemma":[0.9556538,0.0001480679,0.03950766,0.0002346297,0.00007561212,0.0008589956,0.001730532,0.0006894632,0.001101123],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9248942,"threshold_uncertainty_score":0.3972022,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2550579332141903,"score_gpt":0.3930564024462317,"score_spread":0.1379984692320415,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}