{"id":"W2998116579","doi":"10.1609/aaai.v34i04.6086","title":"Reinforcement Learning with Perturbed Rewards","year":2020,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":101,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"Defense Advanced Research Projects Agency; National Science Foundation","keywords":"Reinforcement learning; Computer science; Noise (video); Confusion matrix; Artificial intelligence; Convergence (economics); Confusion; Set (abstract data type); Gaussian; Matrix (chemical analysis); Mathematical optimization; Machine learning; Algorithm; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003072414,0.001433771,0.001550443,0.0004944922,0.000455455,0.00119369,0.001478011,0.00134324,0.001897832],"category_scores_gemma":[0.01545961,0.0006311451,0.0005415133,0.0003646695,0.002103872,0.00158656,0.001693644,0.002321545,0.0004041076],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001590761,"about_ca_system_score_gemma":0.001562113,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003592426,"about_ca_topic_score_gemma":0.002476167,"domain_scores_codex":[0.9975473,0.001041015,0.0001274863,0.0005533872,0.0004727286,0.0002581217],"domain_scores_gemma":[0.9928646,0.004687838,0.0008604575,0.0005263369,0.0007574988,0.0003032303],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001179932,0.00004357854,0.0007508854,0.0000538021,0.00003744979,0.00006317751,0.00005628242,0.9641032,0.000748848,0.01617385,0.0006002072,0.01725067],"study_design_scores_gemma":[0.00001146515,0.00002296482,0.00006585637,0.000005770238,0.000004465222,0.000009374304,0.000003625878,0.9918436,0.0002745295,0.007569454,0.0001844675,0.00000443109],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02487299,0.0003071302,0.9718664,0.0003653483,0.00005015177,0.00006807598,0.00005753352,0.0005145414,0.00189785],"genre_scores_gemma":[0.9246601,0.0001734558,0.07204021,0.0002539973,0.00005685918,0.0001682798,0.000123574,0.00008477795,0.00243864],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003592426,"threshold_uncertainty_score":0.01624864,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0602824730324913,"score_gpt":0.2752996143925658,"score_spread":0.2150171413600745,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}