{"id":"W4391756312","doi":"10.21203/rs.3.rs-3918353/v1","title":"Specialized Deep Residual Policy Safe Reinforcement Learning-Based Controller for Complex and Continuous State-Action Spaces","year":2024,"lang":"en","type":"preprint","venue":"Research Square","topic":"Smart Grid Security and Resilience","field":"Engineering","cited_by":5,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"Science Foundation Ireland; European Commission; Canadian Institute of Steel Construction","keywords":"Reinforcement learning; Residual; Action (physics); Computer science; State (computer science); Controller (irrigation); Artificial intelligence; Control theory (sociology); Reinforcement; Control engineering; Control (management); Engineering; Psychology; Algorithm; Physics; Social psychology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001126501,0.0009355396,0.001067375,0.0002675986,0.0002917289,0.00081827,0.001133136,0.00104274,0.003700897],"category_scores_gemma":[0.002886572,0.0004338816,0.0004288426,0.0002095939,0.0009755605,0.000669169,0.001506311,0.002032668,0.0005473942],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009189161,"about_ca_system_score_gemma":0.001981987,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007567128,"about_ca_topic_score_gemma":0.007963547,"domain_scores_codex":[0.9995537,0.00008148398,0.00002110391,0.0001204167,0.0001290303,0.00009421646],"domain_scores_gemma":[0.998817,0.0006415369,0.000115844,0.000137082,0.0002128532,0.00007572323],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001534784,0.00005627967,0.0002693372,0.00006901193,0.00002604383,0.00006347294,0.00004457722,0.9500127,0.002618379,0.01040663,0.0009344099,0.03534558],"study_design_scores_gemma":[0.000005316018,0.00001390741,0.00002058859,0.000002304073,0.00000172999,0.000003530744,0.00000117101,0.9984152,0.0002206238,0.001240335,0.00007388233,0.000001441472],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01394936,0.000136002,0.9831724,0.0001003537,0.00004678477,0.00003124757,0.00003930407,0.000526417,0.001998174],"genre_scores_gemma":[0.924897,0.0001123017,0.06971817,0.0001336164,0.00003767451,0.0001236544,0.0001300544,0.0001145767,0.004732965],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007567128,"threshold_uncertainty_score":0.01504618,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05785780633926838,"score_gpt":0.3738595371655954,"score_spread":0.3160017308263271,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}