{"id":"W2808007596","doi":"10.65109/miek6215","title":"Training Dialogue Systems With Human Advice","year":2018,"lang":"en","type":"preprint","venue":"","topic":"Speech and dialogue systems","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Microsoft (Canada)","funders":"Horizon 2020 Framework Programme; European Commission","keywords":"Reinforcement learning; Computer science; Perspective (graphical); Bridge (graph theory); Encoding (memory); Point (geometry); Reinforcement; Artificial intelligence; Advice (programming); Human–computer interaction; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001959424,0.0008619681,0.0008723102,0.0003054595,0.0003339572,0.00083522,0.001094645,0.001165113,0.00358033],"category_scores_gemma":[0.01021244,0.0004940571,0.0002825508,0.0001716686,0.0005959115,0.001215445,0.001372532,0.001742148,0.001174649],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004459859,"about_ca_system_score_gemma":0.0007865847,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001740034,"about_ca_topic_score_gemma":0.001758947,"domain_scores_codex":[0.9984558,0.0007191705,0.00007391683,0.0003902887,0.0002591061,0.0001017193],"domain_scores_gemma":[0.9949914,0.003882711,0.0001784208,0.0004512834,0.000324164,0.0001720739],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008018253,0.0007693662,0.001756947,0.0004284451,0.00009044103,0.0002313781,0.00100403,0.4170336,0.0367423,0.00657682,0.004219295,0.5303456],"study_design_scores_gemma":[0.00007797276,0.0001375335,0.0002065565,0.00001530295,0.00001385828,0.0000355461,0.0000413175,0.9869071,0.007588747,0.002886582,0.00207673,0.00001272435],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0916096,0.0004150729,0.8949174,0.0002753998,0.00009874385,0.0001526027,0.0000450792,0.007847193,0.004638851],"genre_scores_gemma":[0.7603661,0.0001369801,0.235429,0.0001915838,0.00004524662,0.0001950917,0.0001188358,0.0002272121,0.003289996],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00358033,"threshold_uncertainty_score":0.01197737,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05333686751493118,"score_gpt":0.2733405274845878,"score_spread":0.2200036599696566,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}