{"id":"W2618318883","doi":"10.48550/arxiv.1705.08551","title":"Safe Model-based Reinforcement Learning with Stability Guarantees","year":2017,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":337,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung","keywords":"Reinforcement learning; Inverted pendulum; Stability (learning theory); Computer science; State space; Artificial neural network; Lyapunov function; Gaussian process; Process (computing); State (computer science); Artificial intelligence; Control (management); Control theory (sociology); Machine learning; Gaussian; Algorithm; Mathematics; Nonlinear system","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002560797,0.001260261,0.001053607,0.0005568001,0.0005661238,0.0008758652,0.001274856,0.001137524,0.001682073],"category_scores_gemma":[0.01141073,0.0005065722,0.0006158674,0.0003034458,0.002176057,0.001184562,0.002313674,0.002146201,0.0003155238],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001192759,"about_ca_system_score_gemma":0.002123005,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003187144,"about_ca_topic_score_gemma":0.00219792,"domain_scores_codex":[0.9987628,0.0003927869,0.0000648069,0.0002748753,0.000334259,0.0001705319],"domain_scores_gemma":[0.9938971,0.004151972,0.000646176,0.0005899097,0.0004901285,0.000224634],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006941459,0.00003420972,0.0005437215,0.0000326302,0.0000180384,0.00005506758,0.00005188907,0.9703953,0.001316029,0.01766301,0.0002613351,0.009559456],"study_design_scores_gemma":[0.00000851464,0.00001871437,0.00002666555,0.000003055284,0.000002277169,0.000005449042,0.000002201621,0.9907759,0.0003847613,0.008706854,0.00006332902,0.000002277221],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03403479,0.00007352766,0.9634,0.0002648414,0.00001819642,0.00004874494,0.00003627564,0.0004801044,0.001643484],"genre_scores_gemma":[0.9492623,0.0000611408,0.04909005,0.0001093351,0.00001930192,0.0001295813,0.00006298273,0.00007015083,0.001195071],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003187144,"threshold_uncertainty_score":0.01354295,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0607400630856931,"score_gpt":0.1995221572087716,"score_spread":0.1387820941230785,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}