{"id":"W3072315125","doi":"10.1073/pnas.1907370117","title":"Fast reinforcement learning with generalized policy updates","year":2020,"lang":"en","type":"article","venue":"Proceedings of the National Academy of Sciences","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":82,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Leverage (statistics); Artificial intelligence; Machine learning; Generalization; Exploit; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002715583,0.001553939,0.001827071,0.0005412599,0.0004345684,0.001025653,0.001721316,0.001571733,0.003563254],"category_scores_gemma":[0.009218163,0.0008093948,0.0005356335,0.000562767,0.001591533,0.001755644,0.001957394,0.00249151,0.0005995301],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001248039,"about_ca_system_score_gemma":0.002109666,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006226778,"about_ca_topic_score_gemma":0.006543126,"domain_scores_codex":[0.9989277,0.0004142136,0.0000532678,0.000198384,0.0002398266,0.0001667069],"domain_scores_gemma":[0.9965654,0.002377419,0.0002513748,0.0003039174,0.0003304521,0.0001713961],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001572232,0.00007854772,0.0006312098,0.00008673107,0.00005680752,0.00008720789,0.00005496436,0.9293624,0.0009385308,0.02105166,0.001715263,0.04577947],"study_design_scores_gemma":[0.00002907097,0.00001979594,0.00003474525,0.000004446696,0.000004425606,0.000006427604,0.000002907001,0.9913098,0.0001588232,0.008206262,0.0002203131,0.000003105309],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01779661,0.0003793281,0.9773014,0.0004299117,0.00009849174,0.00009477967,0.00004801386,0.001086153,0.002765268],"genre_scores_gemma":[0.8483953,0.0002864514,0.1466878,0.0003858508,0.0001139669,0.0003860947,0.0001512692,0.0001489903,0.003444324],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006226778,"threshold_uncertainty_score":0.01436156,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03975041601643639,"score_gpt":0.2885275840620797,"score_spread":0.2487771680456433,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}