{"id":"W1498235675","doi":"","title":"AEMS: an anytime online search algorithm for approximate policy refinement in large POMDPs","year":2007,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":65,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université Laval","funders":"","keywords":"Computer science; Partially observable Markov decision process; Markov decision process; Computation; Task (project management); Mathematical optimization; State space; State (computer science); Function (biology); Bellman equation; Online algorithm; Observable; Algorithm; Markov process; Markov chain; Machine learning; Markov model; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001655794,0.001084854,0.001574345,0.0006587129,0.0004781042,0.000834537,0.002118782,0.001701245,0.004150755],"category_scores_gemma":[0.005930704,0.000595987,0.0008419678,0.0006654413,0.000900804,0.001898689,0.00228919,0.002112753,0.0007250078],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000920685,"about_ca_system_score_gemma":0.002107391,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004158682,"about_ca_topic_score_gemma":0.004425732,"domain_scores_codex":[0.999115,0.0002817795,0.00006446835,0.0001691125,0.0002727166,0.00009680982],"domain_scores_gemma":[0.9978624,0.001467019,0.0001886598,0.0002007838,0.0001805779,0.0001005177],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002651778,0.0001158418,0.000468332,0.0001253698,0.00005539459,0.00008308498,0.0001161707,0.8583562,0.001879685,0.02582916,0.002116335,0.1105892],"study_design_scores_gemma":[0.00003071073,0.00002115738,0.00002519236,0.000005489034,0.000003827522,0.000009544759,0.000005796475,0.9941298,0.000330623,0.005026673,0.0004079865,0.000003296366],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.005223008,0.0001031477,0.9927654,0.00009810719,0.00002808889,0.00005056761,0.00003641071,0.0007508668,0.0009443356],"genre_scores_gemma":[0.3633036,0.0001756418,0.6330306,0.0001831661,0.00004989607,0.0004695028,0.0001986178,0.0002269723,0.002362006],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004158682,"threshold_uncertainty_score":0.01388562,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03203246700207627,"score_gpt":0.3428225774256188,"score_spread":0.3107901104235425,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}