{"id":"W7124317441","doi":"10.65109/ltgs2519","title":"Be Considerate: Avoiding Negative Side Effects in Reinforcement Learning","year":2022,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute","funders":"","keywords":"Reinforcement learning; Agency (philosophy); Reinforcement; Control (management); Action (physics); Discretion","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004565902,0.0009883732,0.0007378528,0.0002915151,0.0007974113,0.001247731,0.001177464,0.001286758,0.003145964],"category_scores_gemma":[0.0170246,0.0003476128,0.0003780909,0.0002479153,0.002532694,0.002217274,0.002197807,0.002109678,0.000509372],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006987636,"about_ca_system_score_gemma":0.00126233,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001186657,"about_ca_topic_score_gemma":0.001551037,"domain_scores_codex":[0.9975773,0.001324273,0.00009484133,0.0003742825,0.0004453621,0.000184001],"domain_scores_gemma":[0.9913582,0.005237684,0.001034973,0.001262625,0.0006206943,0.0004858599],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00156202,0.0008216709,0.01315762,0.000462122,0.0002335052,0.0006074827,0.001492325,0.5372607,0.02300439,0.1856939,0.003143401,0.2325608],"study_design_scores_gemma":[0.0002149143,0.0007954231,0.001347044,0.00006476412,0.00007395491,0.0001933669,0.000122324,0.8145801,0.005769691,0.1730527,0.003746141,0.00003960197],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1580084,0.0004044909,0.8213705,0.001482845,0.00008279024,0.0002112771,0.0000483306,0.001139018,0.01725233],"genre_scores_gemma":[0.926269,0.00008718768,0.07077555,0.0002840252,0.00001985138,0.0001583586,0.00002817559,0.00009136737,0.002286463],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004565902,"threshold_uncertainty_score":0.02414703,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02380526822181108,"score_gpt":0.2594788744108522,"score_spread":0.2356736061890411,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}