{"id":"W2232305727","doi":"10.1609/aaai.v29i1.9700","title":"Tighter Value Function Bounds for Bayesian Reinforcement Learning","year":2015,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Defense Acquisition Program Administration; Agency for Defense Development; National Research Foundation of Korea; Ministry of Science, ICT and Future Planning; National Research Foundation","keywords":"Reinforcement learning; Bayesian probability; Computer science; Heuristic; Bellman equation; Bayes' theorem; Function (biology); Perspective (graphical); Value (mathematics); Artificial intelligence; Machine learning; Mathematical optimization; Focus (optics); Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000985991,0.0002484176,0.0002587067,0.0001521361,0.0002923829,0.0004435041,0.001605539,0.000109096,0.00003091224],"category_scores_gemma":[0.0007631306,0.000193086,0.0001428436,0.0004950477,0.0001664303,0.0006254836,0.0003756727,0.0003458889,0.00008334719],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001190814,"about_ca_system_score_gemma":0.0001745006,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002004679,"about_ca_topic_score_gemma":0.000001282775,"domain_scores_codex":[0.9977155,0.00002227625,0.0006261173,0.0004627515,0.0007334446,0.0004399586],"domain_scores_gemma":[0.9979054,0.0001000794,0.0005235491,0.0003193779,0.001001055,0.0001505937],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009852436,0.00003183675,0.0001576861,0.00003464258,0.0000215581,1.177395e-7,0.001229539,0.1484189,0.002865925,0.8351511,0.0005020636,0.01148811],"study_design_scores_gemma":[0.00006951249,0.0009121821,0.00002922158,0.0001198493,0.00002042549,0.000002113591,0.0004101837,0.8505645,0.0692981,0.07585861,0.002480649,0.000234618],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002718333,0.0000090677,0.9727134,0.001836529,0.0008908436,0.0005671081,3.618715e-7,0.0001390439,0.02112539],"genre_scores_gemma":[0.9876659,0.000007756005,0.009073542,0.0002934582,0.0001260947,0.0000555337,0.000001447327,0.00001829931,0.002757954],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9849476,"threshold_uncertainty_score":0.7873818,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09005187973659068,"score_gpt":0.2974173093581666,"score_spread":0.2073654296215759,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}