{"id":"W4404102001","doi":"10.1109/tse.2024.3491496","title":"SMARLA: A Safety Monitoring Approach for Deep Reinforcement Learning Agents","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"École de Technologie Supérieure; University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Science Foundation Ireland","keywords":"Computer science; Reinforcement learning; Artificial intelligence; Machine learning; Software engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002152692,0.001146469,0.0009763165,0.0006847046,0.0005558784,0.001027524,0.002863397,0.001420843,0.003932673],"category_scores_gemma":[0.004901146,0.0006909085,0.0008282402,0.0002454313,0.0008446661,0.001466611,0.002022125,0.002452915,0.0009440963],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001421127,"about_ca_system_score_gemma":0.00233155,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005575088,"about_ca_topic_score_gemma":0.006611646,"domain_scores_codex":[0.9989038,0.0003363695,0.00007464801,0.000226772,0.0003350437,0.0001233411],"domain_scores_gemma":[0.998457,0.0006168024,0.0002520014,0.0002055115,0.0003382419,0.0001305149],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003194126,0.0002054946,0.001619219,0.0001333757,0.0001092322,0.0001411832,0.0001455837,0.7462417,0.004991782,0.01638633,0.007330824,0.2223759],"study_design_scores_gemma":[0.00001864817,0.00003031545,0.00005501049,0.000006475904,0.000006211652,0.000008911282,0.000003528548,0.9941394,0.001130045,0.003482244,0.00111401,0.00000521594],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.00785913,0.0001318317,0.9821203,0.000281647,0.00005892673,0.0001056833,0.0001104839,0.007172874,0.002159027],"genre_scores_gemma":[0.4665076,0.0001488327,0.5252829,0.0004827265,0.00006837415,0.0004868653,0.0003825749,0.0005166833,0.006123519],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005575088,"threshold_uncertainty_score":0.01315612,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01605624727510781,"score_gpt":0.2502795647694263,"score_spread":0.2342233174943184,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}