{"id":"W2094387729","doi":"10.1016/j.automatica.2009.07.008","title":"Natural actor–critic algorithms","year":2009,"lang":"en","type":"article","venue":"Automatica","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":569,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Temporal difference learning; Reinforcement learning; Computer science; Function approximation; Convergence (economics); Bellman equation; Mathematical proof; Function (biology); Stochastic gradient descent; Variance (accounting); Mathematics; Mathematical optimization; Artificial intelligence; Applied mathematics; Algorithm; Artificial neural network","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00133634,0.0008623444,0.000780511,0.0004885596,0.0005726913,0.000897716,0.001278947,0.001616539,0.0069444],"category_scores_gemma":[0.003892239,0.0005689578,0.0005047329,0.0003782699,0.001213521,0.001250851,0.001012431,0.001551954,0.001602722],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005894248,"about_ca_system_score_gemma":0.0008228104,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0012352,"about_ca_topic_score_gemma":0.002175177,"domain_scores_codex":[0.9994937,0.0001884871,0.00002587997,0.0001567341,0.0001010924,0.00003417404],"domain_scores_gemma":[0.9987341,0.000776974,0.00007584225,0.0001853443,0.000182759,0.00004504412],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001102242,0.0001096189,0.0005981494,0.0002055258,0.00009454824,0.00009158626,0.00009331269,0.6211019,0.003402977,0.165495,0.009209668,0.1994876],"study_design_scores_gemma":[0.00001467423,0.00001632886,0.00005114386,0.000008618609,0.000008045756,0.00002494151,0.000004969598,0.9583105,0.0004703049,0.03864738,0.00243715,0.000005869494],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.004533079,0.0004114238,0.9861299,0.0002477291,0.0001222294,0.00003444306,0.00002717905,0.0005410488,0.007952861],"genre_scores_gemma":[0.4347113,0.0006529115,0.5399045,0.000430031,0.0001886026,0.0003064238,0.0001864661,0.0002860199,0.02333382],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0069444,"threshold_uncertainty_score":0.02323127,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.009214446772915929,"score_gpt":0.2557505181836024,"score_spread":0.2465360714106865,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}