{"id":"W4225482307","doi":"10.1111/mafi.12388","title":"Reinforcement learning with dynamic convex risk measures","year":2023,"lang":"en","type":"article","venue":"Mathematical Finance","topic":"Risk and Portfolio Optimization","field":"Decision Sciences","cited_by":23,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Computer science; Mathematical optimization; Flexibility (engineering); Q-learning; Dynamic programming; Artificial neural network; Obstacle; Dynamic risk measure; Convex optimization; Artificial intelligence; Value at risk; Regular polygon; Risk management; Mathematics; Economics; Finance","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00284066,0.000951478,0.001022552,0.0004193404,0.0003089905,0.001066926,0.001237053,0.001258017,0.001576647],"category_scores_gemma":[0.00852531,0.000537931,0.0005706743,0.0002980872,0.001706657,0.001209101,0.001223797,0.001762511,0.0001957604],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001494084,"about_ca_system_score_gemma":0.001420535,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003756509,"about_ca_topic_score_gemma":0.002101271,"domain_scores_codex":[0.9987632,0.0006290898,0.00005192387,0.0001761074,0.0002793058,0.0001004234],"domain_scores_gemma":[0.996578,0.002354683,0.0003480569,0.0001874087,0.0003974316,0.0001343423],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00001809484,0.00001796721,0.0001743058,0.00001446658,0.0000212511,0.00002836076,0.00001773191,0.9716746,0.0003614553,0.02288896,0.0001946488,0.004588174],"study_design_scores_gemma":[0.000005283698,0.00000808326,0.00001735757,0.000002232536,0.000001810621,0.000002618341,9.506018e-7,0.9933356,0.0000885332,0.006455567,0.00007999253,0.000001932961],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01308793,0.0001061631,0.9844701,0.0002857027,0.00002376413,0.00002780553,0.00001352927,0.00009850888,0.001886493],"genre_scores_gemma":[0.8894097,0.0001544808,0.1071424,0.00018276,0.00005375816,0.000148868,0.00003846491,0.00004714437,0.002822396],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003756509,"threshold_uncertainty_score":0.01502299,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04896239240589748,"score_gpt":0.3373794269646531,"score_spread":0.2884170345587557,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}