{"id":"W3214290601","doi":"","title":"Risk-Aware Transfer in Reinforcement Learning using Successor Features","year":2021,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Successor cardinal; Reinforcement learning; Computer science; Generalization; Variance (accounting); Task (project management); Representation (politics); Artificial intelligence; Function (biology); Domain (mathematical analysis); Machine learning; Bellman equation; Robot; Transfer of learning; Risk aversion (psychology); Expected utility hypothesis; Mathematical optimization; Mathematics; Engineering; Economics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002804071,0.0007915259,0.001160771,0.0004584704,0.0002755028,0.0009154764,0.001036101,0.000880347,0.001199805],"category_scores_gemma":[0.0098974,0.0004048721,0.0006199905,0.0003195525,0.001286111,0.001978714,0.001433069,0.001623653,0.0001835653],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009156918,"about_ca_system_score_gemma":0.00088558,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001129361,"about_ca_topic_score_gemma":0.0008464419,"domain_scores_codex":[0.9989605,0.0004090104,0.00005923624,0.0002089661,0.0002482918,0.0001139902],"domain_scores_gemma":[0.9964436,0.002373193,0.0003780989,0.0003332011,0.0002974723,0.0001745282],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001522214,0.0001224725,0.001551438,0.0000629304,0.00005690643,0.00008483862,0.0001012399,0.911739,0.002470179,0.02266725,0.0003661646,0.0606254],"study_design_scores_gemma":[0.00001117617,0.00007319413,0.0001732443,0.00000611247,0.000007347216,0.0000140387,0.000003959144,0.9872144,0.0004515477,0.01191902,0.0001195159,0.000006404525],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0769974,0.0001834361,0.9206902,0.0001973246,0.00002466286,0.00005096096,0.00002802574,0.00032716,0.001500895],"genre_scores_gemma":[0.9570548,0.00007565894,0.04170613,0.00006244193,0.0000196878,0.00008483281,0.0000302881,0.00003634673,0.0009297542],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002804071,"threshold_uncertainty_score":0.01482952,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04860939166662293,"score_gpt":0.19407695932823,"score_spread":0.145467567661607,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}