{"id":"W4405055855","doi":"10.1109/lra.2024.3512374","title":"Safety Filtering While Training: Improving the Performance and Sample Efficiency of Reinforcement Learning Agents","year":2024,"lang":"en","type":"article","venue":"IEEE Robotics and Automation Letters","topic":"Occupational Health and Safety Research","field":"Health Professions","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Training (meteorology); Sample (material); Reinforcement learning; Reinforcement; Computer science; Artificial intelligence; Psychology; Chromatography; Social psychology; Geography; Chemistry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002685007,0.0009263447,0.0006224657,0.000338394,0.0003535308,0.000670501,0.001008634,0.0008962512,0.001264405],"category_scores_gemma":[0.01684157,0.0003593634,0.000259009,0.0001303531,0.000808404,0.0009844932,0.0008500877,0.001278298,0.0002598862],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006528075,"about_ca_system_score_gemma":0.001061688,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005523541,"about_ca_topic_score_gemma":0.003602384,"domain_scores_codex":[0.9990686,0.000350664,0.00005653372,0.000162285,0.0002211684,0.0001407278],"domain_scores_gemma":[0.993275,0.004746535,0.0005366176,0.0005995695,0.0006695804,0.000172687],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003503149,0.0003213715,0.004500759,0.0001063231,0.00005402674,0.00007294107,0.0001772855,0.8775348,0.008771134,0.002551701,0.0006200619,0.1049392],"study_design_scores_gemma":[0.00002762775,0.0001577615,0.0003974196,0.00001024785,0.000008927626,0.0000120949,0.00001275279,0.9959513,0.002405326,0.0008378763,0.0001736304,0.000005036932],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3335225,0.0003809984,0.6590812,0.0004922151,0.00006705424,0.0001655524,0.00004021714,0.00180602,0.004444224],"genre_scores_gemma":[0.9616289,0.00004468373,0.03754488,0.00009613261,0.00001267873,0.00005304867,0.00002406705,0.00004774401,0.0005478615],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005523541,"threshold_uncertainty_score":0.01419985,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07189234968498567,"score_gpt":0.3722205461360454,"score_spread":0.3003281964510597,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}