{"id":"W2998461398","doi":"10.1609/aaai.v34i04.5784","title":"Fixed-Horizon Temporal Difference Methods for Stable Reinforcement Learning","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; Huawei Technologies (Canada); University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates; DeepMind","keywords":"Reinforcement learning; Bellman equation; Horizon; Temporal difference learning; Function (biology); Time horizon; Value (mathematics); Stability (learning theory); Computer science; Function approximation; Mathematics; Mathematical optimization; Artificial intelligence; Machine learning; Artificial neural network","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001892933,0.0008204663,0.0008838018,0.0003736944,0.0003737682,0.000878088,0.001742071,0.001039393,0.003911997],"category_scores_gemma":[0.006435315,0.0003937122,0.0006156992,0.0004134063,0.001264954,0.001303956,0.0012494,0.002160058,0.0004993501],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001253006,"about_ca_system_score_gemma":0.00112703,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003497718,"about_ca_topic_score_gemma":0.002513047,"domain_scores_codex":[0.9994251,0.0002307041,0.00002841491,0.0001029328,0.0001644997,0.00004844601],"domain_scores_gemma":[0.9973502,0.001932046,0.0001981546,0.0001418984,0.0002809783,0.0000966486],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00008757655,0.0000663014,0.0004807408,0.0001504787,0.00004752209,0.00005652081,0.00008397981,0.8471003,0.001722734,0.1000382,0.001092675,0.049073],"study_design_scores_gemma":[0.000008108379,0.00001679905,0.00001926174,0.000008066338,0.000003041904,0.000006071911,0.000002645868,0.9843847,0.0002691226,0.01475631,0.0005224794,0.000003518703],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002498533,0.0001977198,0.9958574,0.00009258442,0.00004087613,0.00001809291,0.00001342891,0.00008956917,0.00119187],"genre_scores_gemma":[0.5456198,0.0005602998,0.4476922,0.0002810085,0.0000886076,0.0003217804,0.00009837484,0.0001868174,0.005151144],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003911997,"threshold_uncertainty_score":0.01308697,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05269488141129239,"score_gpt":0.328042082555781,"score_spread":0.2753472011444886,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}