{"id":"W2903017298","doi":"10.1609/aaai.v33i01.33014512","title":"State-Augmentation Transformations for Risk-Sensitive Reinforcement Learning","year":2019,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Concordia University","funders":"China Scholarship Council","keywords":"Reinforcement learning; Markov decision process; Markov chain; Successor cardinal; Markov process; Function (biology); Computer science; Bellman equation; Q-learning; State (computer science); Mathematical optimization; Mathematics; Artificial intelligence; Machine learning; Algorithm; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006410653,0.0001962835,0.0002188177,0.0001407509,0.0002937055,0.0002574895,0.0009659635,0.00005952648,0.0000418937],"category_scores_gemma":[0.0003182386,0.0001587961,0.0001353337,0.0004254009,0.00009269619,0.0007219526,0.0001405887,0.0003260182,0.0001531095],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008194862,"about_ca_system_score_gemma":0.00008090899,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002552277,"about_ca_topic_score_gemma":0.000002516349,"domain_scores_codex":[0.9981782,0.00002253432,0.0006107399,0.0003426848,0.0005123499,0.0003335122],"domain_scores_gemma":[0.9981335,0.0001885249,0.0006564946,0.0002163654,0.0007445381,0.00006056068],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00007758974,0.00002987522,0.0003372649,0.0000628662,0.00002953289,4.643773e-8,0.004745534,0.5037624,0.01239164,0.4575954,0.00004490194,0.02092291],"study_design_scores_gemma":[0.00006371517,0.0004105042,0.0001289491,0.0001041119,0.0000127288,7.715722e-7,0.0008143045,0.7164541,0.2682008,0.0135169,0.0001352744,0.0001578651],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06181052,0.000002623533,0.9267079,0.0007961463,0.0003764446,0.001140261,0.000002701255,0.00009323414,0.009070151],"genre_scores_gemma":[0.9931095,0.00004233568,0.005459989,0.0001090901,0.00002145054,0.00004508473,0.000002905004,0.00001219461,0.001197514],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9312989,"threshold_uncertainty_score":0.6475517,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04340228077397301,"score_gpt":0.2852395603095862,"score_spread":0.2418372795356132,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}