{"id":"W2593237273","doi":"10.1609/aaai.v32i1.11631","title":"Multi-Step Reinforcement Learning: A Unifying Algorithm","year":2018,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":107,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates; DeepMind","keywords":"Reinforcement learning; Computer science; Backup; Algorithm; TRACE (psycholinguistics); Sampling (signal processing); Focus (optics); Monte Carlo method; Importance sampling; Artificial intelligence; Machine learning; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005125657,0.001089709,0.001682817,0.0008340065,0.0005870996,0.00134367,0.003349214,0.002242376,0.00213215],"category_scores_gemma":[0.0106976,0.0005913439,0.0007992157,0.0006536894,0.001699375,0.003062309,0.002902421,0.003134806,0.0006357672],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001275611,"about_ca_system_score_gemma":0.00221014,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00214132,"about_ca_topic_score_gemma":0.002010247,"domain_scores_codex":[0.997906,0.000637068,0.0001462688,0.0005927667,0.0005471883,0.0001706874],"domain_scores_gemma":[0.9962378,0.00226886,0.0002181543,0.000695585,0.0003746171,0.0002050132],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002277628,0.0002248632,0.001323109,0.0001220436,0.00007502508,0.00007056214,0.0002295248,0.6289505,0.003076306,0.1005753,0.001859557,0.2632655],"study_design_scores_gemma":[0.00002434869,0.00004901222,0.00003607898,0.000009986517,0.000006142607,0.00001603444,0.000004921979,0.9825962,0.0006418874,0.01595887,0.0006500112,0.000006582957],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.004356501,0.0001238667,0.9937963,0.0001732886,0.00002700765,0.00005120847,0.000009462159,0.0004982041,0.0009641481],"genre_scores_gemma":[0.264034,0.0002019632,0.7325341,0.0003017705,0.0000790517,0.0002646954,0.00006869081,0.0002173772,0.002298352],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005125657,"threshold_uncertainty_score":0.02710736,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09247718432083181,"score_gpt":0.3136361361909427,"score_spread":0.2211589518701109,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}