{"id":"W4388623348","doi":"10.1109/ro-man57019.2023.10309622","title":"Adapting a Teachable Robot’s Dialog Responses using Reinforcement Learning in Teaching Conversation","year":2023,"lang":"en","type":"article","venue":"","topic":"Social Robot Interaction and HRI","field":"Psychology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"Australian Research Council","keywords":"Reinforcement learning; Conversation; Dialog box; Computer science; Robot; Human–computer interaction; Selection (genetic algorithm); Task (project management); Artificial intelligence; Gaze; Gesture; Robotics; Social cue; Psychology; Engineering; World Wide Web; Cognitive psychology; Communication","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001800473,0.0006735364,0.0005429361,0.0003041159,0.0003049402,0.0006041578,0.001043695,0.0007984317,0.001557025],"category_scores_gemma":[0.006567839,0.0002598012,0.0003121482,0.0001603624,0.0006635212,0.0006419001,0.000679622,0.0007847968,0.0004661808],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005322333,"about_ca_system_score_gemma":0.0007194637,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002171876,"about_ca_topic_score_gemma":0.002004408,"domain_scores_codex":[0.9990656,0.0005261841,0.00003413143,0.0001872267,0.0001208219,0.00006610318],"domain_scores_gemma":[0.996863,0.002295381,0.0002281489,0.0001836898,0.0002759449,0.0001538609],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0009104758,0.001681248,0.008069804,0.000455475,0.0001397469,0.0004085551,0.001965257,0.493018,0.06338392,0.004733969,0.00114235,0.4240912],"study_design_scores_gemma":[0.00007876931,0.0004571263,0.001248662,0.0000167679,0.00002266731,0.00008822149,0.0001066955,0.9846289,0.01068556,0.001588834,0.001047962,0.00002977272],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2879001,0.0002459127,0.7057332,0.0002830602,0.00004117942,0.000417088,0.00003258678,0.002163584,0.003183216],"genre_scores_gemma":[0.8835979,0.00006385868,0.1145921,0.00007136565,0.00001062238,0.0002104351,0.00002434633,0.00004307913,0.001386321],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002171876,"threshold_uncertainty_score":0.009521961,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.125742803229994,"score_gpt":0.4148619474141664,"score_spread":0.2891191441841724,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}