{"id":"W2903017298","doi":"10.1609/aaai.v33i01.33014512","title":"State-Augmentation Transformations for Risk-Sensitive Reinforcement Learning","year":2019,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Concordia University","funders":"China Scholarship Council","keywords":"Reinforcement learning; Markov decision process; Markov chain; Successor cardinal; Markov process; Function (biology); Computer science; Bellman equation; Q-learning; State (computer science); Mathematical optimization; Mathematics; Artificial intelligence; Machine learning; Algorithm; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002786979,0.001322722,0.001015404,0.000545535,0.0003345639,0.001058768,0.0009290361,0.001097895,0.003684246],"category_scores_gemma":[0.00740729,0.0004224555,0.001275203,0.0005560073,0.002026192,0.001865475,0.001764606,0.002803002,0.0005938086],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001184205,"about_ca_system_score_gemma":0.001239125,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001248854,"about_ca_topic_score_gemma":0.0007517368,"domain_scores_codex":[0.9982483,0.000883098,0.0001070388,0.0003266089,0.0003315052,0.0001034564],"domain_scores_gemma":[0.9970654,0.002045886,0.0002371498,0.0002861625,0.0002635536,0.0001019807],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009055692,0.00009025913,0.0004244224,0.0001169405,0.0000466068,0.0001234585,0.0001880529,0.7082975,0.002295382,0.2452425,0.001076404,0.04200798],"study_design_scores_gemma":[0.000012918,0.00004737906,0.00004719111,0.00001077203,0.000006294512,0.00001701407,0.000005682979,0.9103338,0.0004909163,0.08820883,0.0008113382,0.000008042922],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.005535051,0.0001488332,0.9915959,0.0001459063,0.00003025226,0.00004564401,0.00003682762,0.0001849346,0.002276546],"genre_scores_gemma":[0.7362184,0.0004673816,0.2570255,0.0002720664,0.00009161309,0.0005958988,0.0002246949,0.0001766478,0.004927811],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003684246,"threshold_uncertainty_score":0.01473916,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04340228077397301,"score_gpt":0.2852395603095862,"score_spread":0.2418372795356132,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}