{"id":"W6929030644","doi":"10.48448/sa9g-t219","title":"Sample-Efficient Iterative Lower Bound Optimization of Deep Reactive Policies for Planning in Continuous MDPs","year":2022,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Upper and lower bounds; Gradient descent; Parametric statistics; Markov decision process; Artificial neural network; Optimization problem; Reinforcement learning; Range (aeronautics); Convergence (economics); Quality (philosophy)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001425751,0.0004628433,0.0007093037,0.002793494,0.0003338384,0.0001621052,0.0008906876,0.0001777153,0.001422091],"category_scores_gemma":[0.001409726,0.0004603477,0.00009987527,0.002585707,0.001649781,0.0002256765,0.0003519597,0.0003727397,0.00002055228],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001182696,"about_ca_system_score_gemma":0.0007657545,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001247473,"about_ca_topic_score_gemma":0.0004037413,"domain_scores_codex":[0.9962919,0.000129696,0.0006549967,0.0009837921,0.001135284,0.0008043699],"domain_scores_gemma":[0.9972889,0.0005677039,0.001071245,0.0005656619,0.0003789615,0.0001274593],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002585808,0.000729883,0.0008042499,0.00008800497,0.0000710857,0.00001339939,0.008746757,0.9655089,0.001458402,0.01123964,0.01045859,0.000622547],"study_design_scores_gemma":[0.001915796,0.0007014755,0.0001911376,0.0004566754,0.00007199383,0.00001242906,0.007260949,0.9531413,0.0004355258,0.0004456468,0.03442051,0.0009465553],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.008123397,0.001217274,0.6936544,0.0002105139,0.002080291,0.006912549,0.006338839,0.0006648468,0.2807979],"genre_scores_gemma":[0.5598162,0.00004943476,0.3592698,0.0009936324,0.001387325,0.00171771,0.005285411,0.004500138,0.06698031],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.5516928,"threshold_uncertainty_score":0.9997848,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02553907224511928,"score_gpt":0.319526539367639,"score_spread":0.2939874671225198,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}