{"id":"W4378498877","doi":"10.48550/arxiv.2305.15602","title":"Control invariant set enhanced safe reinforcement learning: improved sampling efficiency, guaranteed stability and robustness","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Viral Infectious Diseases and Gene Expression in Insects","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Robustness (evolution); Computer science; Supervisor; Stability (learning theory); Invariant (physics); Training set; Control theory (sociology); Sampling (signal processing); Artificial intelligence; Mathematical optimization; Machine learning; Control (management); Mathematics; Detector","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001743504,0.0008136213,0.0007986554,0.0003296344,0.0002375975,0.000681767,0.001054338,0.0006406161,0.001367922],"category_scores_gemma":[0.006009056,0.0003142986,0.0004170227,0.0001905588,0.001086888,0.0007051734,0.001145456,0.001382534,0.0002188751],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005969157,"about_ca_system_score_gemma":0.001154775,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002786206,"about_ca_topic_score_gemma":0.001685142,"domain_scores_codex":[0.9992362,0.0002169185,0.0000365163,0.0001606615,0.0002409869,0.0001086947],"domain_scores_gemma":[0.9973553,0.001535589,0.0003177921,0.0002861841,0.0003902998,0.0001147154],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000164424,0.00008319208,0.001001753,0.0000855986,0.00003465919,0.000108128,0.00008755467,0.929026,0.008243851,0.009801016,0.0004194436,0.05094444],"study_design_scores_gemma":[0.00001057485,0.00004607727,0.00006819925,0.0000038988,0.000003250989,0.00001098577,0.000002370372,0.9973159,0.00117144,0.001241505,0.0001225894,0.000003208087],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03528506,0.0001604859,0.9619405,0.0001160131,0.00002649777,0.00003820329,0.00001704509,0.0005598318,0.001856413],"genre_scores_gemma":[0.9339802,0.00008179576,0.06440421,0.00008163568,0.00002253697,0.00007942703,0.00004284898,0.0000702873,0.001236923],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002786206,"threshold_uncertainty_score":0.00922066,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05811226930098883,"score_gpt":0.2214898710407897,"score_spread":0.1633776017398008,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}