{"id":"W2895478303","doi":"10.48550/arxiv.1810.01032","title":"Reinforcement Learning with Perturbed Rewards","year":2018,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Noise (video); Confusion matrix; Artificial intelligence; Convergence (economics); Confusion; Set (abstract data type); Gaussian; Machine learning; Matrix (chemical analysis); Mathematical optimization; Algorithm; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002795853,0.001427395,0.001592421,0.000480973,0.0004424687,0.00118514,0.001442122,0.001300751,0.001923606],"category_scores_gemma":[0.01468758,0.0006194672,0.00053645,0.0003703299,0.001953406,0.001490895,0.001600783,0.002249337,0.0004149539],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001549922,"about_ca_system_score_gemma":0.001645589,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004046855,"about_ca_topic_score_gemma":0.002907058,"domain_scores_codex":[0.9977103,0.0009557036,0.000124276,0.0005363463,0.0004314668,0.0002417845],"domain_scores_gemma":[0.9935306,0.004187455,0.000771853,0.0004802303,0.0007417015,0.0002881908],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001405349,0.00005542979,0.0009591464,0.00007013173,0.00004395815,0.0000690605,0.00006553119,0.9587994,0.0008603268,0.01545677,0.0007183386,0.02276136],"study_design_scores_gemma":[0.00001317035,0.00002478117,0.00007962972,0.00000654248,0.000005109887,0.000009718729,0.000004091681,0.9917758,0.0002916467,0.007579865,0.0002048921,0.000004630415],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02747287,0.0003425587,0.9691117,0.0003817783,0.00005474201,0.00007188218,0.00006345378,0.0005808074,0.00192015],"genre_scores_gemma":[0.9221777,0.0001745748,0.07459135,0.0002526303,0.00005636132,0.000174462,0.0001348205,0.00008646631,0.002351552],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004046855,"threshold_uncertainty_score":0.01478612,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05333297638058677,"score_gpt":0.185023905452045,"score_spread":0.1316909290714582,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}