{"id":"W4388904294","doi":"10.1016/j.ifacol.2023.10.923","title":"A modular framework for stabilizing deep reinforcement learning control","year":2023,"lang":"en","type":"article","venue":"IFAC-PapersOnLine","topic":"Model Reduction and Neural Networks","field":"Physics and Astronomy","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Honeywell (Canada); University of British Columbia","funders":"","keywords":"Reinforcement learning; Modular design; Computer science; Realization (probability); Parameterized complexity; Construct (python library); Stability (learning theory); Set (abstract data type); Artificial intelligence; Artificial neural network; Domain (mathematical analysis); Deep learning; Nonlinear system; Control (management); Control engineering; Machine learning; Engineering; Algorithm; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001077628,0.0009601955,0.0007626027,0.0003783828,0.0003887385,0.0009056078,0.001792219,0.001023552,0.003706532],"category_scores_gemma":[0.001335315,0.0004115335,0.0007922088,0.0002613562,0.001405746,0.0008587231,0.002002617,0.001923175,0.0008496803],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008613056,"about_ca_system_score_gemma":0.001200502,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001643587,"about_ca_topic_score_gemma":0.001888774,"domain_scores_codex":[0.9995812,0.0000815268,0.00002449016,0.00009592289,0.0001630647,0.0000538961],"domain_scores_gemma":[0.9996722,0.00009599148,0.00004195869,0.00007055923,0.00008057848,0.0000387352],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004155206,0.00003852753,0.0001478371,0.00007491428,0.00003353666,0.00007970531,0.00006512478,0.7431147,0.007807297,0.2152195,0.001096623,0.03228078],"study_design_scores_gemma":[0.000011921,0.00003583316,0.00002066701,0.000008267889,0.000005277121,0.00001151271,0.000003480594,0.9638726,0.0008920404,0.03369098,0.001441091,0.00000629358],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.001491521,0.00004742804,0.9962267,0.00006344097,0.0000192987,0.00001728511,0.00001544725,0.0002422873,0.001876731],"genre_scores_gemma":[0.6201555,0.0003125082,0.3718367,0.0002305209,0.0001110312,0.0005134015,0.0001110624,0.000182696,0.006546567],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003706532,"threshold_uncertainty_score":0.01239961,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02128174126494181,"score_gpt":0.2848373929165303,"score_spread":0.2635556516515885,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}