{"id":"W2948710720","doi":"10.18653/v1/p19-1004","title":"Do Neural Dialog Systems Use the Conversation History Effectively? An Empirical Study","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal; Université de Montréal","funders":"Nvidia","keywords":"Dialog box; Computer science; Generative grammar; Flexibility (engineering); Shuffling; Context (archaeology); Artificial intelligence; Transformer; Code (set theory); Conversation; Dialog system; Natural language processing; Machine learning; Human–computer interaction; Programming language; Communication; World Wide Web; Psychology; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01623791,0.000837775,0.0006917346,0.001128496,0.0008965142,0.001795089,0.001361798,0.001422132,0.002228109],"category_scores_gemma":[0.09303408,0.0008065788,0.0006391177,0.001022259,0.001655809,0.005492673,0.001634543,0.00268525,0.001163648],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001287231,"about_ca_system_score_gemma":0.0004678395,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00468525,"about_ca_topic_score_gemma":0.005356996,"domain_scores_codex":[0.9894404,0.00742089,0.0005098988,0.001536682,0.0007961684,0.0002959506],"domain_scores_gemma":[0.8796498,0.09958354,0.00511376,0.01099752,0.003533814,0.00112152],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.005842248,0.002387303,0.3785306,0.002741591,0.002145136,0.0006613047,0.008887505,0.2161473,0.01896477,0.007882627,0.01444667,0.3413631],"study_design_scores_gemma":[0.0002525989,0.00156905,0.2239786,0.0002393251,0.0004995986,0.001062291,0.002946074,0.7368421,0.01080963,0.0138364,0.007729324,0.0002349983],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9753044,0.00239788,0.01571604,0.0007664411,0.00006127753,0.0001282563,0.001249425,0.0002958234,0.004080488],"genre_scores_gemma":[0.9941855,0.0003123327,0.003395483,0.00008804819,0.00003400393,0.0000495076,0.001346799,0.000049146,0.0005391579],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01623791,"threshold_uncertainty_score":0.08587533,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1060491787075014,"score_gpt":0.3092665205169398,"score_spread":0.2032173418094383,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}