{"id":"W2734029172","doi":"10.48550/arxiv.1703.05423","title":"End-to-end optimization of goal-driven and visually grounded dialogue systems Harm de Vries","year":2017,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Speech and dialogue systems","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Task (project management); Utterance; Artificial intelligence; Context (archaeology); Reinforcement learning; Sequence (biology); Object (grammar); Human–computer interaction","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0004223371,0.0003156983,0.0005378033,0.0003487998,0.0002331345,0.0004182011,0.001718724,0.0003360405,0.000005575249],"category_scores_gemma":[0.0001152932,0.0003591714,0.0001319455,0.0002404259,0.000182804,0.0004988787,0.001645774,0.0002571959,0.00001830339],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002070269,"about_ca_system_score_gemma":0.00037394,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002840117,"about_ca_topic_score_gemma":0.0001974683,"domain_scores_codex":[0.9980183,0.0002289062,0.0002928209,0.0009320887,0.0001510894,0.0003768124],"domain_scores_gemma":[0.9974852,0.0001343325,0.0005402858,0.00127852,0.0002703059,0.0002913554],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00008572081,0.00008945385,0.006235668,0.0004048025,0.0001763941,0.0002191088,0.001456462,0.8608514,0.0002889851,0.1295512,0.0003102333,0.0003306269],"study_design_scores_gemma":[0.001114445,0.0002172204,0.00621612,0.0005019591,0.0001214554,0.00002771726,0.0002627704,0.9839988,0.0003707425,0.005958834,0.0004172492,0.0007926585],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1197158,0.0001124152,0.8750533,0.00005551005,0.001226628,0.0005794823,0.00006329181,0.0001534389,0.003040204],"genre_scores_gemma":[0.9944073,0.0001499614,0.004594823,0.00002636923,0.0001425583,0.000003195059,0.00004249881,0.00001903973,0.0006142335],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8746915,"threshold_uncertainty_score":0.999886,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05763941532819348,"score_gpt":0.2097574543190631,"score_spread":0.1521180389908696,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}