{"id":"W3093980962","doi":"","title":"Incremental Policy Gradients for Online Reinforcement Learning Control","year":2021,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Term (time); Control (management); Artificial intelligence; Econometrics; Machine learning; Economics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001615618,0.001212793,0.00121595,0.0007646118,0.0003856845,0.001227413,0.001457777,0.001142681,0.00382387],"category_scores_gemma":[0.008275854,0.0005470573,0.0005416449,0.0006102683,0.001202881,0.001502105,0.00126458,0.00239873,0.0007245602],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001489993,"about_ca_system_score_gemma":0.001452882,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004533744,"about_ca_topic_score_gemma":0.003478379,"domain_scores_codex":[0.9993568,0.0002204113,0.00003785606,0.0001073167,0.0002164295,0.00006117328],"domain_scores_gemma":[0.9980664,0.001352373,0.0001373633,0.0001195394,0.0002462179,0.00007813468],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009167357,0.00006671449,0.0004528459,0.0001644756,0.00004164793,0.00005548307,0.00006721407,0.841387,0.0009175842,0.07705104,0.002329288,0.07737507],"study_design_scores_gemma":[0.000008632725,0.00001721351,0.00003752719,0.00001052674,0.000003613915,0.00000799464,0.000002217158,0.9770839,0.0002394413,0.02187394,0.0007098595,0.00000518374],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.003201545,0.0005371151,0.9931052,0.0001866323,0.00006635611,0.00004311382,0.00003776562,0.0005665192,0.002255655],"genre_scores_gemma":[0.6635723,0.001101188,0.3262515,0.0003423982,0.0002132339,0.0005132593,0.0002038087,0.0004005058,0.007401835],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004533744,"threshold_uncertainty_score":0.01279217,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01937927747768515,"score_gpt":0.280628564126172,"score_spread":0.2612492866484868,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}