{"id":"W7123338223","doi":"10.1109/cdc57313.2025.11312866","title":"Stochastic Reinforcement Learning with Stability Guarantees for Control of Unknown Nonlinear Systems","year":2025,"lang":"","type":"article","venue":"","topic":"Adaptive Dynamic Programming Control","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Convergence (economics); Stability (learning theory); Nonlinear system; Representation (politics); Control theory (sociology); Controller (irrigation); Control (management); Artificial neural network","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002152039,0.001148036,0.0009801161,0.0004930609,0.0005266455,0.0008980514,0.000837311,0.0009860467,0.002133064],"category_scores_gemma":[0.008736051,0.0004367618,0.0004052043,0.0003554302,0.001776768,0.0008260235,0.001353034,0.001577171,0.0003077219],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001424915,"about_ca_system_score_gemma":0.001921779,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006266935,"about_ca_topic_score_gemma":0.00325812,"domain_scores_codex":[0.9990645,0.0003357688,0.00004165038,0.0001404333,0.000311543,0.0001060604],"domain_scores_gemma":[0.9959075,0.002800209,0.0004801808,0.000142619,0.0005482842,0.0001211413],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004137947,0.00002202565,0.0002081415,0.00005619664,0.00001681479,0.00004085239,0.00003157852,0.9762297,0.000764774,0.01563523,0.0003425743,0.006610673],"study_design_scores_gemma":[0.000007834078,0.00001141568,0.00002301861,0.000003402632,0.000001262413,0.000003122446,0.000001522677,0.9959183,0.0001183027,0.003828838,0.00008111272,0.000001785628],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01761451,0.0003255998,0.9781728,0.0004224348,0.00003967568,0.00003923023,0.00002204711,0.0003159942,0.003047638],"genre_scores_gemma":[0.9672955,0.0002643756,0.02991538,0.0001402174,0.00005526883,0.00013929,0.00004191016,0.00006174899,0.002086475],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006266935,"threshold_uncertainty_score":0.01246089,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0109348647064978,"score_gpt":0.2436693270617748,"score_spread":0.232734462355277,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}