{"id":"W3131546278","doi":"10.1609/aaai.v35i13.17378","title":"How RL Agents Behave When Their Actions Are Modified","year":2021,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Action (physics); Supervisor; Markov decision process; Computer science; Process (computing); Q-learning; Reinforcement; Control (management); Risk analysis (engineering); Intervention (counseling); Artificial intelligence; Markov process; Psychology; Business; Social psychology; Mathematics; Political science; Law","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00175774,0.000362039,0.0003567986,0.0002586009,0.000302238,0.001444364,0.0007712796,0.001176853,0.001434864],"category_scores_gemma":[0.0111394,0.0003294575,0.0003033098,0.0001576073,0.00148596,0.002329903,0.0006589775,0.00105484,0.0005703611],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008522718,"about_ca_system_score_gemma":0.0007948545,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004133081,"about_ca_topic_score_gemma":0.002447321,"domain_scores_codex":[0.9988137,0.0005999251,0.00004860388,0.0001961671,0.0001990899,0.0001425521],"domain_scores_gemma":[0.9972362,0.001343038,0.0004138453,0.0005139972,0.0003177841,0.0001751848],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002005405,0.00009209545,0.009564803,0.0001211701,0.0001438001,0.0003189487,0.0006763741,0.8431384,0.01105277,0.08572289,0.002465991,0.04650228],"study_design_scores_gemma":[0.00003405642,0.00004473866,0.001325191,0.00002452955,0.00002140256,0.00007143507,0.000120041,0.9293588,0.002218972,0.06513131,0.001621435,0.00002814118],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3597941,0.0005344427,0.6131535,0.00386993,0.0001732841,0.0001132894,0.0001637218,0.001441332,0.02075638],"genre_scores_gemma":[0.9710667,0.0001999222,0.02562527,0.0002031408,0.00001799562,0.00004727331,0.00005068659,0.0001150372,0.00267411],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004133081,"threshold_uncertainty_score":0.00929594,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2113528704672706,"score_gpt":0.3166812681019577,"score_spread":0.1053283976346872,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}