{"id":"W4283800411","doi":"10.1609/aaai.v36i9.21220","title":"Sample-Efficient Iterative Lower Bound Optimization of Deep Reactive Policies for Planning in Continuous MDPs","year":2022,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Machine Learning and Algorithms","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"National Research Foundation Singapore; National Research Foundation","keywords":"Computer science; Mathematical optimization; Variance (accounting); Maximization; Iterative learning control; Upper and lower bounds; Parametric statistics; Sample (material); Markov decision process; Optimization problem; Artificial intelligence; Algorithm; Mathematics; Control (management)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001838112,0.001358534,0.00134121,0.0005416527,0.0004388265,0.0009585877,0.001277563,0.001297332,0.002490437],"category_scores_gemma":[0.00798416,0.0008822725,0.0006692108,0.0004763948,0.001552295,0.001263367,0.001693632,0.002403673,0.0003956459],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001732386,"about_ca_system_score_gemma":0.002800903,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009502151,"about_ca_topic_score_gemma":0.01065838,"domain_scores_codex":[0.9993111,0.0002433826,0.0000323608,0.0001281891,0.0001780659,0.000106786],"domain_scores_gemma":[0.9964557,0.002688547,0.0002576832,0.0001848529,0.0002707441,0.0001424675],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002809929,0.00001582353,0.0001331089,0.00002641124,0.000007952328,0.00001436078,0.00001756069,0.9899143,0.0002001656,0.003090753,0.0002549617,0.006296539],"study_design_scores_gemma":[0.000004371062,0.000008270134,0.00001261973,0.000003522693,0.00000133056,0.000001874578,0.000002663911,0.9978862,0.0001260328,0.001865452,0.00008652793,0.000001147532],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0202543,0.0002391823,0.9760576,0.0002681708,0.00002656497,0.00005166956,0.00005780324,0.0006107566,0.002433853],"genre_scores_gemma":[0.7398497,0.0002319825,0.2560887,0.0002638424,0.00003296255,0.0003455139,0.0002834856,0.0003474547,0.002556292],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009502151,"threshold_uncertainty_score":0.01889366,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04515529792387129,"score_gpt":0.3091709383972863,"score_spread":0.264015640473415,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}