{"id":"W136985510","doi":"10.5555/2034396.2034517","title":"Escaping local optima in POMDP planning as inference","year":2011,"lang":"en","type":"article","venue":"Adaptive Agents and Multi-Agents Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Inference; Partially observable Markov decision process; Local optimum; Reinforcement learning; Computer science; Mathematical optimization; Observable; Greedy algorithm; Artificial intelligence; Machine learning; Mathematics; Markov chain; Markov model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002925021,0.0009257593,0.001625636,0.0006072966,0.000615276,0.0009409456,0.001495914,0.001274782,0.001348385],"category_scores_gemma":[0.008670349,0.0007916964,0.0007210283,0.0006155971,0.002620418,0.001698337,0.002133942,0.002405516,0.0002162767],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001348704,"about_ca_system_score_gemma":0.001464654,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005802807,"about_ca_topic_score_gemma":0.005541426,"domain_scores_codex":[0.9989944,0.0004414437,0.00005151698,0.0001668119,0.0002275489,0.0001181434],"domain_scores_gemma":[0.9957508,0.003364935,0.0002818339,0.000280252,0.0001936061,0.0001286334],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004356256,0.00001877312,0.0002574802,0.00003855765,0.00002473203,0.00003781235,0.0000715155,0.9690899,0.0004309418,0.01735602,0.0001977493,0.012433],"study_design_scores_gemma":[0.00001322325,0.0000148217,0.00002860878,0.000006205497,0.000005860445,0.0000052345,0.000007509139,0.9851624,0.0002569314,0.01436823,0.000126748,0.000004116733],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02155071,0.0002570921,0.9755437,0.0002590087,0.00001845887,0.0000431878,0.0000207886,0.0004005992,0.001906364],"genre_scores_gemma":[0.7663519,0.000245208,0.2313246,0.0001770438,0.00002945833,0.0002071902,0.00004911335,0.0001118733,0.001503646],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005802807,"threshold_uncertainty_score":0.01546919,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1458052012319216,"score_gpt":0.3207535313869368,"score_spread":0.1749483301550152,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}