{"id":"W4384574988","doi":"10.23952/jano.5.2023.2.01","title":"Multi-step actor-critic framework for reinforcement learning in continuous control","year":2023,"lang":"en","type":"article","venue":"Journal of Applied and Numerical Optimization","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Reinforcement learning; Temporal difference learning; Computer science; Sequence (biology); State (computer science); Control (management); Artificial intelligence; Reinforcement; Optimal control; Action (physics); Machine learning; Mathematical optimization; Mathematics; Algorithm; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005129366,0.0001132001,0.0003017553,0.0002032471,0.00008997731,0.0001215317,0.0002277047,0.00008596822,0.000005993477],"category_scores_gemma":[0.0003259993,0.0001007781,0.00006865932,0.0003780361,0.00002293054,0.0002252519,0.0000535895,0.0002934615,0.00000375469],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004715164,"about_ca_system_score_gemma":0.00003779406,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000001297178,"about_ca_topic_score_gemma":2.960846e-8,"domain_scores_codex":[0.9988026,0.00003076843,0.000506311,0.0001517015,0.0002654127,0.0002432321],"domain_scores_gemma":[0.9989275,0.0004312873,0.0003287215,0.0001087386,0.0001143377,0.00008940976],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006385638,0.00001666699,0.0002004793,0.00001809969,0.00002429647,0.000004404697,0.000308577,0.9882131,0.00007662349,0.005237695,0.00003643178,0.005799809],"study_design_scores_gemma":[0.001611162,0.0002841213,0.0003261968,0.00004563882,0.00001500379,0.000004105963,0.0001056444,0.9965022,0.00003737968,0.0002782046,0.0006795131,0.0001108347],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0003090465,0.00002954669,0.9986778,0.0003875884,0.0001870606,0.000242685,1.579799e-7,0.00005242605,0.000113671],"genre_scores_gemma":[0.6061676,0.00006510926,0.3934723,0.0001707546,0.00005512069,0.00000947877,0.000002235684,0.000009739872,0.00004771585],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.6058586,"threshold_uncertainty_score":0.4109611,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01559883102852876,"score_gpt":0.2672656284476811,"score_spread":0.2516667974191523,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}