{"id":"W3194439889","doi":"10.2139/ssrn.3910498","title":"Robust Risk-Aware Reinforcement Learning","year":2021,"lang":"en","type":"article","venue":"SSRN Electronic Journal","topic":"Risk and Portfolio Optimization","field":"Decision Sciences","cited_by":3,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Reinforcement; Computer science; Artificial intelligence; Psychology; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002190587,0.001162168,0.001816755,0.0004935603,0.0002761534,0.001197304,0.001384067,0.001516619,0.002876435],"category_scores_gemma":[0.01021873,0.0005703051,0.0004940764,0.0003769924,0.001022886,0.00119166,0.001826012,0.001790747,0.0005544347],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007594964,"about_ca_system_score_gemma":0.001256997,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002092293,"about_ca_topic_score_gemma":0.001429091,"domain_scores_codex":[0.9987593,0.0004346889,0.00005435101,0.0002703092,0.0002879331,0.0001935084],"domain_scores_gemma":[0.9955782,0.00289127,0.0004569638,0.0003260349,0.0005384613,0.0002090828],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001485006,0.00007678241,0.0004093802,0.00005333918,0.00005456108,0.00004629505,0.00001865314,0.9633145,0.001085596,0.009263583,0.0008798903,0.02464889],"study_design_scores_gemma":[0.00001193543,0.0000298751,0.00005011399,0.00000356187,0.000005394362,0.000008022199,0.000001363214,0.9960866,0.0001783195,0.003536979,0.00008480152,0.000003120318],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03022638,0.0003029043,0.9644083,0.0003050969,0.00006016412,0.00004801755,0.00005334894,0.0004922478,0.004103469],"genre_scores_gemma":[0.9541908,0.000104551,0.04287712,0.0001101798,0.00004416686,0.00006966788,0.00006371696,0.00006231364,0.002477528],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002876435,"threshold_uncertainty_score":0.01158506,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04207220615069272,"score_gpt":0.312367694612759,"score_spread":0.2702954884620662,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}