{"id":"W3131546278","doi":"10.1609/aaai.v35i13.17378","title":"How RL Agents Behave When Their Actions Are Modified","year":2021,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Action (physics); Supervisor; Markov decision process; Computer science; Process (computing); Q-learning; Reinforcement; Control (management); Risk analysis (engineering); Intervention (counseling); Artificial intelligence; Markov process; Psychology; Business; Social psychology; Mathematics; Political science; Law","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication","open_science"],"consensus_categories":[],"category_scores_codex":[0.0004903844,0.0006067873,0.0006643537,0.0002553429,0.0003677997,0.002516711,0.005548385,0.0004218243,0.0000692743],"category_scores_gemma":[0.0007517709,0.000486385,0.0004346037,0.0005008172,0.0003354616,0.0006392227,0.004000958,0.001639181,0.00004560231],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001825561,"about_ca_system_score_gemma":0.0003411173,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005976595,"about_ca_topic_score_gemma":0.00001553006,"domain_scores_codex":[0.9963326,0.00005284521,0.0008017552,0.00116965,0.001046528,0.0005966594],"domain_scores_gemma":[0.9952906,0.0001215317,0.001533841,0.001168627,0.001717315,0.0001680615],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006652023,0.0005364755,0.0006050752,0.0006506441,0.0003485674,0.000007825822,0.01448909,0.1442671,0.01478334,0.786023,0.001545583,0.03667677],"study_design_scores_gemma":[0.00003880668,0.0001540239,0.0003316634,0.001617698,0.00006816451,0.000007401172,0.003357694,0.7408228,0.1817849,0.07070894,0.0003338579,0.0007739866],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05571116,0.0000633569,0.9059023,0.01561553,0.003652923,0.001421562,0.00002403783,0.00037411,0.01723496],"genre_scores_gemma":[0.9893325,0.0001282937,0.007306213,0.0002547817,0.000153774,0.00009369967,0.000005999677,0.0000369614,0.002687793],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9336213,"threshold_uncertainty_score":0.9998321,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2113528704672706,"score_gpt":0.3166812681019577,"score_spread":0.1053283976346872,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}