{"id":"W2950912239","doi":"10.48550/arxiv.1804.05950","title":"State-Augmentation Transformations for Risk-Sensitive Reinforcement Learning","year":2018,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Cardiac electrophysiology and arrhythmias","field":"Medicine","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Concordia University","funders":"","keywords":"Reinforcement learning; Markov decision process; Successor cardinal; Markov chain; Markov process; Function (biology); Bellman equation; State (computer science); Computer science; Q-learning; Mathematical optimization; Mathematics; Artificial intelligence; Machine learning; Statistics; Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002518857,0.001333433,0.001004243,0.0005500341,0.0003357753,0.001044137,0.0008996933,0.001053433,0.003862235],"category_scores_gemma":[0.007184109,0.0004256837,0.001228664,0.0005572448,0.001943466,0.001797471,0.001801562,0.002771948,0.0006201072],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001180306,"about_ca_system_score_gemma":0.001267661,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001284585,"about_ca_topic_score_gemma":0.0007679702,"domain_scores_codex":[0.9983449,0.000784449,0.0001055886,0.0003354735,0.0003202652,0.000109397],"domain_scores_gemma":[0.9971378,0.001987144,0.0002526648,0.0002582696,0.0002550769,0.0001089753],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009929465,0.00009693448,0.0004592389,0.0001250343,0.00004736475,0.0001386052,0.0002052265,0.7266743,0.002422074,0.2243814,0.001179083,0.04417144],"study_design_scores_gemma":[0.00001295907,0.00005087803,0.00004744023,0.00001165005,0.000006634009,0.00001932763,0.000006592371,0.9088781,0.0005116582,0.08962724,0.0008194171,0.000008034856],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.005941492,0.0001535868,0.9910977,0.0001412098,0.00003212925,0.00004756639,0.00004074164,0.000212033,0.002333625],"genre_scores_gemma":[0.7633613,0.0004333877,0.229976,0.0002568788,0.00008447334,0.0005556651,0.0002235192,0.0001932478,0.004915504],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003862235,"threshold_uncertainty_score":0.01332116,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03508921438499553,"score_gpt":0.2123391094610163,"score_spread":0.1772498950760208,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}