{"id":"W4403702578","doi":"10.48550/arxiv.2409.10096","title":"Robust Reinforcement Learning with Dynamic Distortion Risk Measures","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Reinforcement; Distortion (music); Computer science; Artificial intelligence; Psychology; Social psychology; Telecommunications","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003461464,0.001367917,0.001619482,0.000455672,0.0003311214,0.001344687,0.001451729,0.001442957,0.001516526],"category_scores_gemma":[0.01024992,0.0006207657,0.0007277327,0.0004319425,0.001580106,0.001845384,0.001704537,0.002167977,0.0002866914],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001545355,"about_ca_system_score_gemma":0.001330946,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003945538,"about_ca_topic_score_gemma":0.001776836,"domain_scores_codex":[0.9986132,0.0006152391,0.00006626325,0.0003067964,0.0002606652,0.0001377451],"domain_scores_gemma":[0.9962823,0.002548228,0.000409332,0.0002603913,0.0003449206,0.0001548402],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003879662,0.00002372383,0.000308453,0.00002566771,0.00002840598,0.00003759397,0.00002914769,0.9707925,0.0004182748,0.01749733,0.0002781469,0.01052202],"study_design_scores_gemma":[0.00000718837,0.00001426883,0.00003116589,0.00000302934,0.000003062423,0.000006040044,0.000002017868,0.9911986,0.0001486876,0.008487278,0.00009535898,0.000003258334],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01041379,0.0001347057,0.9881287,0.0001571235,0.00001457009,0.00002551858,0.00001575907,0.0001582918,0.0009515381],"genre_scores_gemma":[0.8455141,0.0001941727,0.150891,0.0001491043,0.00004037179,0.0001408062,0.0000745246,0.000106853,0.002889102],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003945538,"threshold_uncertainty_score":0.0183062,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05243449871161009,"score_gpt":0.1774675049997577,"score_spread":0.1250330062881476,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}