{"id":"W6910226933","doi":"10.48448/x6dd-pp74","title":"The Effect of Multi-step Methods on Overestimation in Deep Reinforcement Learning","year":2020,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Deep learning; Representation (politics); Artificial neural network; Temporal difference learning; Deep neural networks","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01049521,0.001352062,0.001503711,0.0005270583,0.0008257535,0.001363902,0.001781989,0.001607754,0.001548468],"category_scores_gemma":[0.04379568,0.0009674859,0.0008132782,0.000431476,0.001787566,0.002797831,0.002480414,0.004016591,0.0003177377],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001324199,"about_ca_system_score_gemma":0.002097022,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003195748,"about_ca_topic_score_gemma":0.003683098,"domain_scores_codex":[0.9960319,0.001764039,0.0003752999,0.0006018004,0.0009067394,0.0003200347],"domain_scores_gemma":[0.9547471,0.03493365,0.003208602,0.004061813,0.002313191,0.0007356498],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001004027,0.0002473686,0.008499634,0.0003385629,0.0001963667,0.0003712516,0.0004954584,0.8805748,0.00570982,0.0153473,0.001970878,0.08524453],"study_design_scores_gemma":[0.00003438169,0.0001428994,0.0004938759,0.00004013475,0.00002467366,0.00007758238,0.00002938097,0.9905428,0.003817631,0.004393453,0.0003875403,0.00001560343],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1502896,0.002936258,0.8403256,0.001624391,0.00023657,0.0001209628,0.00006878042,0.001792914,0.00260491],"genre_scores_gemma":[0.8829961,0.0004410709,0.1139769,0.000666358,0.00007867772,0.0001506737,0.00007423653,0.0002333936,0.001382734],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01049521,"threshold_uncertainty_score":0.05550456,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0332757155414774,"score_gpt":0.3834781824723748,"score_spread":0.3502024669308974,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}