{"id":"W2796085096","doi":"10.1007/978-3-319-89656-4_6","title":"Advice-Based Exploration in Model-Based Reinforcement Learning","year":2018,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"ca_institutions":"Centre for Social Innovation; Vector Institute; University of Toronto","funders":"","keywords":"Reinforcement learning; Satisficing; Computer science; Advice (programming); Robustness (evolution); Convergence (economics); Grid; Artificial intelligence; Operations research; Mathematical optimization; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009662018,0.0006576166,0.0009562142,0.0002899258,0.0003067165,0.0008501934,0.001392867,0.00104812,0.003655578],"category_scores_gemma":[0.004286531,0.0004889423,0.0005037368,0.0004442394,0.001024336,0.00118418,0.001350407,0.001827019,0.0004055896],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007163897,"about_ca_system_score_gemma":0.0006849151,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002748895,"about_ca_topic_score_gemma":0.002291393,"domain_scores_codex":[0.9995215,0.0002073303,0.00002538263,0.00007008334,0.0001308916,0.00004481832],"domain_scores_gemma":[0.9985141,0.001155387,0.00006518456,0.0000951766,0.0001132431,0.00005686724],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001144551,0.00005801309,0.0002563703,0.0001319772,0.0000321198,0.00004760781,0.00008594605,0.8493829,0.001374064,0.06387893,0.001620108,0.0830176],"study_design_scores_gemma":[0.000008543843,0.00001682929,0.00002136449,0.000006297943,0.000003309481,0.000007562715,0.000002153563,0.9805722,0.0001795444,0.01886423,0.000314964,0.000002904569],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01119602,0.0007559029,0.9825369,0.000179599,0.0000752644,0.00002638698,0.00002229967,0.0003076601,0.004900085],"genre_scores_gemma":[0.8023399,0.0008450404,0.1877841,0.0001270998,0.0001001708,0.0002152257,0.00008464736,0.0001629633,0.008340823],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003655578,"threshold_uncertainty_score":0.01222914,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02666271158748201,"score_gpt":0.2551386000725917,"score_spread":0.2284758884851097,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}