{"id":"W2232305727","doi":"10.1609/aaai.v29i1.9700","title":"Tighter Value Function Bounds for Bayesian Reinforcement Learning","year":2015,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Defense Acquisition Program Administration; Agency for Defense Development; National Research Foundation of Korea; Ministry of Science, ICT and Future Planning; National Research Foundation","keywords":"Reinforcement learning; Bayesian probability; Computer science; Heuristic; Bellman equation; Bayes' theorem; Function (biology); Perspective (graphical); Value (mathematics); Artificial intelligence; Machine learning; Mathematical optimization; Focus (optics); Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01395135,0.002583914,0.003270631,0.002541065,0.001215896,0.004452459,0.002986605,0.003480006,0.006231288],"category_scores_gemma":[0.08731733,0.001537704,0.001555141,0.001774031,0.004484632,0.00834672,0.005587924,0.009845499,0.001179353],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005310373,"about_ca_system_score_gemma":0.003541018,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003519701,"about_ca_topic_score_gemma":0.00331677,"domain_scores_codex":[0.9921227,0.00381482,0.0003591272,0.0009361127,0.002098082,0.0006692118],"domain_scores_gemma":[0.9574349,0.03578346,0.001769579,0.002082909,0.002142334,0.0007868977],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00009496891,0.0001035388,0.0004532874,0.0001529561,0.00005200364,0.00005051314,0.0001685249,0.7046476,0.0008358313,0.2631941,0.001642742,0.02860386],"study_design_scores_gemma":[0.00001548075,0.00002156187,0.00004822771,0.00004347095,0.000007056229,0.00000914058,0.00001004166,0.8635418,0.0003026066,0.1354702,0.0005196395,0.00001080988],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.004064534,0.0004512745,0.9918521,0.0003956958,0.00004184445,0.00004397385,0.00003831208,0.00020476,0.002907458],"genre_scores_gemma":[0.5231126,0.001300437,0.4685656,0.0009903042,0.0002831211,0.0008425387,0.0003262791,0.0007196479,0.003859593],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01395135,"threshold_uncertainty_score":0.07378268,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09005187973659068,"score_gpt":0.2974173093581666,"score_spread":0.2073654296215759,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}