{"id":"W4391903608","doi":"10.24251/hicss.2023.799","title":"Safe Reinforcement Learning via Observation Shielding","year":2023,"lang":"en","type":"article","venue":"Proceedings of the ... Annual Hawaii International Conference on System Sciences/Proceedings of the Annual Hawaii International Conference on System Sciences","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université Laval","funders":"National Science Foundation","keywords":"Robustness (evolution); Computer science; Adversarial system; Agile software development; SAFER; Artificial intelligence; Machine learning; Reinforcement learning; Deep neural networks; Artificial neural network; Computer security","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001973558,0.001526198,0.001019331,0.0004475857,0.0005020251,0.0008128327,0.001484891,0.001085752,0.002269769],"category_scores_gemma":[0.008717036,0.0006135038,0.0006463647,0.0002059772,0.001961617,0.001523052,0.002649988,0.002767045,0.0005123207],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008298896,"about_ca_system_score_gemma":0.001823719,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002612984,"about_ca_topic_score_gemma":0.002048242,"domain_scores_codex":[0.9988417,0.0003491713,0.00005382501,0.0002842134,0.0002988589,0.0001722776],"domain_scores_gemma":[0.9955856,0.002626306,0.0005814987,0.0006072693,0.0003769273,0.0002223585],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001469982,0.00005889701,0.0008960741,0.00005319947,0.00003734828,0.0000890283,0.00008772399,0.93902,0.003139974,0.01082163,0.000958177,0.04469093],"study_design_scores_gemma":[0.00001456604,0.00004052072,0.00004695105,0.000005871924,0.000005260012,0.00001383618,0.000004885078,0.9934139,0.0007916031,0.005414862,0.0002437986,0.000004070091],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0183108,0.0001374589,0.9782508,0.0001984434,0.00003936795,0.00004295079,0.00002084775,0.001166207,0.001833138],"genre_scores_gemma":[0.9154555,0.000119142,0.08134421,0.000260477,0.00005575025,0.0001241965,0.0000774565,0.0002040364,0.002359245],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.002612984,"threshold_uncertainty_score":0.01043725,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05429878908399725,"score_gpt":0.3038269335608544,"score_spread":0.2495281444768571,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}