{"id":"W4385680786","doi":"10.48550/arxiv.2308.02594","title":"SMARLA: A Safety Monitoring Approach for Deep Reinforcement Learning Agents","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Science Foundation Ireland","keywords":"Reinforcement learning; Reinforcement; Computer science; Safety monitoring; Artificial intelligence; Psychology; Social psychology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00221794,0.001344467,0.001105876,0.0007569571,0.0004799261,0.0009603012,0.00273478,0.001434222,0.003204033],"category_scores_gemma":[0.006588452,0.0007584137,0.0008305688,0.0002690786,0.0008849895,0.001372175,0.001926449,0.002561922,0.0006040251],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001304929,"about_ca_system_score_gemma":0.002373974,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007104923,"about_ca_topic_score_gemma":0.008240048,"domain_scores_codex":[0.9990419,0.0003016048,0.00006339356,0.000221816,0.000243105,0.0001282252],"domain_scores_gemma":[0.9975006,0.001239871,0.0004196338,0.0002225929,0.0004423379,0.0001749604],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001772666,0.0001373464,0.002548649,0.0001009423,0.00008209276,0.000112413,0.00009716937,0.9049966,0.002221921,0.004969448,0.002004323,0.08255181],"study_design_scores_gemma":[0.00001063143,0.00002553728,0.00007169446,0.00000550189,0.000005013361,0.000006755723,0.000003325377,0.9975078,0.0003887814,0.00169765,0.0002735094,0.000003715756],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02340309,0.0002603518,0.9673665,0.0004046774,0.00007174651,0.0001493007,0.0001634664,0.006195075,0.0019856],"genre_scores_gemma":[0.7345586,0.0001337478,0.260986,0.0004414359,0.00005072479,0.0003974365,0.0003066925,0.0002925384,0.00283284],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007104923,"threshold_uncertainty_score":0.01412714,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1158806875982378,"score_gpt":0.231981940174605,"score_spread":0.1161012525763672,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}