{"id":"W4404351472","doi":"10.1145/3677052.3698668","title":"EX-DRL: Hedging Against Heavy Losses with EXtreme Distributional Reinforcement Learning","year":2024,"lang":"en","type":"article","venue":"","topic":"Risk and Portfolio Optimization","field":"Decision Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University; University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Reinforcement; Environmental science; Artificial intelligence; Engineering; Structural engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001012047,0.0001415396,0.0001561398,0.0002091095,0.000242179,0.0007700258,0.0002241442,0.00004608641,0.001477684],"category_scores_gemma":[0.000308748,0.00008660796,0.00007356599,0.0009190902,0.00007289797,0.0005918065,0.00006946302,0.0001595358,0.000647867],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006051962,"about_ca_system_score_gemma":0.0001507592,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000238342,"about_ca_topic_score_gemma":0.00001283549,"domain_scores_codex":[0.9976704,0.00005758097,0.0004411406,0.0004150353,0.001160516,0.0002552981],"domain_scores_gemma":[0.9990023,0.0003540958,0.00008197531,0.0002412381,0.0002110287,0.0001093593],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00007257393,0.00002949258,0.02072582,0.000008759819,0.00004393538,0.00008556883,0.0004285712,0.8113446,0.0001482308,0.04046469,0.03203055,0.09461719],"study_design_scores_gemma":[0.000218882,0.000112531,0.001647112,0.0000773771,0.00001387126,0.00002620346,0.0006992795,0.4968185,0.0009188312,0.001673984,0.4975235,0.0002698225],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03571983,0.0004495655,0.8778352,0.001422486,0.0003602993,0.0001495656,0.0000058337,0.0002537946,0.08380346],"genre_scores_gemma":[0.9665365,0.0002829203,0.001790423,0.0001464923,0.0001379186,0.000007461029,0.00008731562,0.00001198196,0.03099902],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9308167,"threshold_uncertainty_score":0.9994351,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07769634120264451,"score_gpt":0.3399878389204604,"score_spread":0.2622914977178159,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}