{"id":"W2739747865","doi":"10.24963/ijcai.2017/290","title":"Constrained Bayesian Reinforcement Learning via Approximate Linear Programming","year":2017,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo; Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Defense Acquisition Program Administration; Korea Advanced Institute of Science and Technology; Agency for Defense Development; Ministry of Science, ICT and Future Planning","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Bayesian probability; Mathematical optimization; Machine learning; Linear programming; Bayesian optimization; State (computer science); Algorithm; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001533661,0.001241564,0.001700083,0.0004756278,0.0004098463,0.001083001,0.001472417,0.001389802,0.002913157],"category_scores_gemma":[0.00678737,0.0006641957,0.0005069654,0.0005539567,0.001426023,0.001374524,0.001648482,0.002070822,0.0005200035],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001251837,"about_ca_system_score_gemma":0.001821155,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004415961,"about_ca_topic_score_gemma":0.003986261,"domain_scores_codex":[0.9988099,0.000502071,0.00003888959,0.0001964515,0.0003383625,0.0001142911],"domain_scores_gemma":[0.9970986,0.002080781,0.0002643339,0.000157139,0.0002793433,0.000119871],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003727329,0.00003634176,0.0002034535,0.00004248676,0.00001818689,0.00003043451,0.00002819224,0.9646149,0.0003111048,0.0159298,0.0005175091,0.01823025],"study_design_scores_gemma":[0.000006533191,0.000009992384,0.0000129976,0.000002987681,0.00000151587,0.000004043203,0.000001798179,0.9925936,0.00007139673,0.007162987,0.0001306104,0.000001677171],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.005258357,0.000121025,0.9928069,0.0001460792,0.00001289746,0.0000285987,0.00002092451,0.0002477248,0.001357393],"genre_scores_gemma":[0.7571067,0.0002378352,0.2376364,0.0002958451,0.0000535802,0.0003898371,0.0001664195,0.0001475398,0.003965807],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004415961,"threshold_uncertainty_score":0.009745419,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0208781681538385,"score_gpt":0.2726533437933096,"score_spread":0.2517751756394711,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}