{"id":"W2236244207","doi":"10.1561/2200000049","title":"Bayesian Reinforcement Learning: A Survey","year":2015,"lang":"en","type":"article","venue":"Foundations and Trends® in Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":223,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Machine learning; Reinforcement learning; Artificial intelligence; Bayesian inference; Bayesian probability; Variable-order Bayesian network; Prior probability; Inference; Bellman equation; Algorithm; Mathematical optimization; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004878199,0.001806153,0.002751805,0.002339058,0.0006884105,0.003515098,0.002922326,0.002894475,0.006918041],"category_scores_gemma":[0.01509955,0.001307841,0.001279284,0.004347405,0.002017614,0.004233444,0.001999738,0.003736906,0.002949983],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002382739,"about_ca_system_score_gemma":0.003210425,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005603018,"about_ca_topic_score_gemma":0.00400937,"domain_scores_codex":[0.9968479,0.001323011,0.0002122427,0.0004120325,0.001097947,0.0001068429],"domain_scores_gemma":[0.9904035,0.007986934,0.000220332,0.0004286645,0.0008227439,0.0001377627],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00006478524,0.0001390015,0.001281765,0.002274508,0.0001592148,0.00006810849,0.0001853246,0.05350136,0.0003125357,0.3446048,0.01871744,0.5786912],"study_design_scores_gemma":[0.00006125388,0.00009670688,0.0009556268,0.001401176,0.00009139439,0.0003002932,0.00009830664,0.2135358,0.0005544247,0.5736824,0.2091278,0.00009489838],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"review","genre_scores_codex":[0.001584646,0.2261449,0.7485257,0.003841139,0.0005675985,0.0001080122,0.0002607502,0.000406917,0.01856046],"genre_scores_gemma":[0.09047037,0.5075034,0.3839407,0.001897743,0.003980792,0.0006423852,0.000834936,0.0004903558,0.0102394],"genre_candidate":"review","genre_consensus":null,"teacher_disagreement_score":0.006918041,"threshold_uncertainty_score":0.02579868,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0465390705497981,"score_gpt":0.3022379049415592,"score_spread":0.2556988343917611,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}