{"id":"W136985510","doi":"10.5555/2034396.2034517","title":"Escaping local optima in POMDP planning as inference","year":2011,"lang":"en","type":"article","venue":"Adaptive Agents and Multi-Agents Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Inference; Partially observable Markov decision process; Local optimum; Reinforcement learning; Computer science; Mathematical optimization; Observable; Greedy algorithm; Artificial intelligence; Machine learning; Mathematics; Markov chain; Markov model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0004280967,0.0002922819,0.000330724,0.000230624,0.000163234,0.0001825527,0.000709742,0.0001231148,0.00001990259],"category_scores_gemma":[0.00006657723,0.0002723748,0.00004996403,0.0002773002,0.00008966195,0.0006984895,0.0005076737,0.0002657892,0.0001401132],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001034499,"about_ca_system_score_gemma":0.00005725723,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00145943,"about_ca_topic_score_gemma":0.000008596812,"domain_scores_codex":[0.9977815,0.0001756274,0.0005130767,0.0005940566,0.0003895269,0.0005462286],"domain_scores_gemma":[0.9989308,0.00008626519,0.0002449965,0.0004497585,0.00009072117,0.0001974884],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006434666,0.0003519422,0.1371648,0.000172008,0.0002686074,0.0008595563,0.04207514,0.7818159,0.0001561736,0.02546311,0.000440442,0.01116796],"study_design_scores_gemma":[0.0009258892,0.0001893018,0.04901441,0.0003496743,0.000008620755,0.00002189576,0.001382086,0.9471146,0.00006440315,0.0000286407,0.0005655186,0.000334947],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03692525,0.0003475878,0.9556413,0.000009500646,0.0006961336,0.0004958632,0.000001530719,0.0001075585,0.005775293],"genre_scores_gemma":[0.9844389,0.00005264179,0.01444962,0.0001413584,0.00003129439,0.00003574352,0.000003146525,0.00002094336,0.0008263548],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9475136,"threshold_uncertainty_score":0.9999728,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1458052012319216,"score_gpt":0.3207535313869368,"score_spread":0.1749483301550152,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}