{"id":"W3214290601","doi":"","title":"Risk-Aware Transfer in Reinforcement Learning using Successor Features","year":2021,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Successor cardinal; Reinforcement learning; Computer science; Generalization; Variance (accounting); Task (project management); Representation (politics); Artificial intelligence; Function (biology); Domain (mathematical analysis); Machine learning; Bellman equation; Robot; Transfer of learning; Risk aversion (psychology); Expected utility hypothesis; Mathematical optimization; Mathematics; Engineering; Economics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002495768,0.0001929818,0.0002072054,0.0002127165,0.0002464995,0.0001358372,0.0006869267,0.0001113259,0.00007403954],"category_scores_gemma":[0.00007621519,0.000229055,0.00009462912,0.001230777,0.00004965949,0.0007372614,0.0002985093,0.0005340592,0.00003643104],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002167822,"about_ca_system_score_gemma":0.0001491804,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001942598,"about_ca_topic_score_gemma":0.00007483006,"domain_scores_codex":[0.9984158,0.0002190121,0.0002068535,0.0005864634,0.0001426525,0.000429258],"domain_scores_gemma":[0.9990379,0.0001118834,0.00008785587,0.0005252749,0.0001254962,0.0001115801],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000113508,0.00001740363,0.0220902,0.00001549212,0.00002562654,0.0004288685,0.0003644016,0.9542177,0.0002135492,0.02245908,0.00001372901,0.0001425741],"study_design_scores_gemma":[0.0007573563,0.00005888426,0.004198643,0.00005047646,0.00002467378,0.00001198061,0.0002355898,0.9924342,0.001312713,0.000187934,0.0004456736,0.0002818553],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2241716,0.00002643495,0.773748,0.00003618108,0.0001451829,0.00008784579,3.564681e-7,0.0001063997,0.001677894],"genre_scores_gemma":[0.9946644,0.0001188109,0.001255271,0.00007855264,0.00002466097,2.460907e-7,0.00000594919,0.00001348149,0.003838673],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7724928,"threshold_uncertainty_score":0.9340593,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04860939166662293,"score_gpt":0.19407695932823,"score_spread":0.145467567661607,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}