{"id":"W4403702578","doi":"10.48550/arxiv.2409.10096","title":"Robust Reinforcement Learning with Dynamic Distortion Risk Measures","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Reinforcement; Distortion (music); Computer science; Artificial intelligence; Psychology; Social psychology; Telecommunications","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0004652517,0.0004759171,0.0003610263,0.000414541,0.0003313415,0.000384282,0.001555177,0.0002759774,0.00002270236],"category_scores_gemma":[0.00008279295,0.0004781296,0.0002148086,0.0006974178,0.0001286675,0.0003288947,0.00269046,0.002026927,0.0002612682],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000848069,"about_ca_system_score_gemma":0.0002913785,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001982544,"about_ca_topic_score_gemma":0.00006177794,"domain_scores_codex":[0.9973621,0.0002026365,0.0003092225,0.001313893,0.0003193634,0.0004928149],"domain_scores_gemma":[0.9977206,0.0001035631,0.0005268034,0.001270118,0.0002006158,0.0001782847],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002726189,0.00001407875,0.001608621,0.0001101043,0.0002098292,0.0002084944,0.0003565115,0.9819856,0.000004578091,0.01459169,0.00006555503,0.0008176956],"study_design_scores_gemma":[0.0003015263,0.0001897268,0.0004062285,0.0002838966,0.0002219973,0.000004791067,0.00007804015,0.9949169,0.00001877966,0.002089095,0.00091282,0.0005761905],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02121739,0.0001047618,0.9720947,0.00005738919,0.0007285826,0.0003728927,0.00000192762,0.0008035357,0.00461889],"genre_scores_gemma":[0.9836692,0.0004867123,0.002748762,0.00002151679,0.00004871909,0.000002376735,0.00003665828,0.00004435813,0.01294167],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9693459,"threshold_uncertainty_score":0.9997671,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05243449871161009,"score_gpt":0.1774675049997577,"score_spread":0.1250330062881476,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}