{"id":"W2945943081","doi":"10.24963/ijcai.2019/581","title":"Metatrace Actor-Critic: Online Step-Size Tuning by Meta-gradient Descent for Reinforcement Learning Control","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Hyperparameter; Reinforcement learning; Computer science; Robustness (evolution); Gradient descent; Artificial intelligence; Nonlinear system; Function approximation; Machine learning; Mathematical optimization; Artificial neural network; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001634012,0.001284735,0.001368782,0.0005378613,0.0003974879,0.001159389,0.002234803,0.001542569,0.003250221],"category_scores_gemma":[0.005835801,0.0006576928,0.0006394656,0.0004783887,0.001138181,0.001261087,0.001353295,0.002448103,0.0007543269],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009505824,"about_ca_system_score_gemma":0.001741998,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00327669,"about_ca_topic_score_gemma":0.00406995,"domain_scores_codex":[0.9993507,0.0002299639,0.00003554836,0.000134446,0.0001750312,0.00007433799],"domain_scores_gemma":[0.9982448,0.0009691113,0.0001836217,0.0002395043,0.0002570587,0.0001059213],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009569144,0.00005721249,0.0003905476,0.00006808368,0.00005999501,0.000057363,0.00003885916,0.9501979,0.00170395,0.007413169,0.001568995,0.03834825],"study_design_scores_gemma":[0.00001128085,0.00001641684,0.00001989849,0.0000053419,0.000003664662,0.000007354379,0.000001562289,0.9972408,0.0003790421,0.002012583,0.0002994455,0.000002670839],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00951023,0.0003754173,0.9857002,0.0002070976,0.00006901182,0.00005725544,0.00004064117,0.00174883,0.002291156],"genre_scores_gemma":[0.7356002,0.0002843106,0.2590189,0.0002715742,0.0000675793,0.000302285,0.0001386652,0.0004835258,0.003832859],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.00327669,"threshold_uncertainty_score":0.01087302,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04272323222080878,"score_gpt":0.286499092031481,"score_spread":0.2437758598106722,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}