{"id":"W2971484784","doi":"10.1609/aaai.v34i04.5784","title":"Fixed-Horizon Temporal Difference Methods for Stable Reinforcement Learning","year":2020,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; Huawei Technologies (Canada); University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Vector Institute; Alberta Innovates; DeepMind","keywords":"Reinforcement learning; Bellman equation; Horizon; Temporal difference learning; Function (biology); Time horizon; Value (mathematics); Stability (learning theory); Computer science; Mathematics; Mathematical optimization; Artificial intelligence; Machine learning","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001982418,0.0008347128,0.0008894865,0.0003808095,0.0003855237,0.0009468663,0.001754689,0.001119753,0.003985907],"category_scores_gemma":[0.006930117,0.0004178466,0.0006545017,0.0004359124,0.001279356,0.001366746,0.001311554,0.002301897,0.0005145989],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00130499,"about_ca_system_score_gemma":0.001145038,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003537911,"about_ca_topic_score_gemma":0.002388589,"domain_scores_codex":[0.9994004,0.0002444561,0.0000293931,0.0001051263,0.0001712258,0.00004940934],"domain_scores_gemma":[0.9971412,0.002103798,0.0002088576,0.0001506824,0.000294901,0.0001004927],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00008645369,0.00006586307,0.0004727232,0.0001507296,0.00004877562,0.00005864015,0.00008713765,0.8343075,0.001704011,0.1150358,0.001134484,0.04684788],"study_design_scores_gemma":[0.000008464381,0.00001550128,0.00001843315,0.000008116852,0.000003128749,0.000006025136,0.0000026726,0.9816613,0.0002599637,0.01744183,0.0005709677,0.00000358291],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.002313752,0.0002090468,0.9959947,0.000101753,0.00004352049,0.00001810281,0.00001401472,0.0000832907,0.001221762],"genre_scores_gemma":[0.5148572,0.0006232769,0.47777,0.0003045141,0.0001038721,0.0003541287,0.0001061306,0.0002077484,0.005673142],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003985907,"threshold_uncertainty_score":0.01333421,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.137803606700458,"score_gpt":0.3658360123305238,"score_spread":0.2280324056300658,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}