{"id":"W2198357760","doi":"10.1609/aaai.v29i1.9702","title":"Reward Shaping for Model-Based Bayesian Reinforcement Learning","year":2015,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Defense Acquisition Program Administration; Agency for Defense Development; National Research Foundation of Korea; Ministry of Science, ICT and Future Planning; National Research Foundation","keywords":"Reinforcement learning; Benchmark (surveying); Machine learning; Computer science; Artificial intelligence; Heuristic; Bayesian probability; Function (biology); Bayes' theorem; Domain (mathematical analysis); Bayesian inference; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003449043,0.00118107,0.001467882,0.0006258523,0.0004506247,0.001269438,0.001750599,0.00150083,0.00333999],"category_scores_gemma":[0.01523477,0.0005842118,0.0005463179,0.0005367351,0.00179552,0.00198319,0.001952069,0.002384395,0.0005436678],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001987776,"about_ca_system_score_gemma":0.001847793,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002638406,"about_ca_topic_score_gemma":0.00211786,"domain_scores_codex":[0.9983397,0.0008420249,0.00006443527,0.0002205832,0.0003854898,0.0001478419],"domain_scores_gemma":[0.995797,0.002966649,0.0003543791,0.0002940409,0.0003758431,0.00021217],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00007261695,0.00005476117,0.0002847785,0.00007880075,0.0000225943,0.00003570752,0.00006095665,0.9078116,0.0006953048,0.06605547,0.0009487285,0.02387858],"study_design_scores_gemma":[0.00001155639,0.00001828018,0.00002228946,0.000008070245,0.000002841245,0.000005671557,0.000002849924,0.9739354,0.0001236706,0.02559595,0.0002688837,0.000004465441],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.008320238,0.000275053,0.9875627,0.0002847443,0.00002942972,0.00004517261,0.0000427924,0.0003273454,0.003112639],"genre_scores_gemma":[0.8441398,0.0003688222,0.1518679,0.0003007174,0.00005033469,0.000316631,0.000124386,0.0001480994,0.002683328],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003449043,"threshold_uncertainty_score":0.01824045,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1686025736564232,"score_gpt":0.3233494496799389,"score_spread":0.1547468760235158,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}