{"id":"W4388741154","doi":"10.1016/j.engappai.2023.107518","title":"Reinforcement learning to achieve real-time control of triple inverted pendulum","year":2023,"lang":"en","type":"article","venue":"Engineering Applications of Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Reinforcement learning; Inverted pendulum; Markov decision process; Process (computing); Convergence (economics); Sample (material); Artificial intelligence; Machine learning; Markov process","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008345326,0.000525987,0.0006320782,0.0002575115,0.0004483421,0.0004742176,0.0006378268,0.0006402639,0.001634025],"category_scores_gemma":[0.001485246,0.0002655919,0.0002864845,0.0001782977,0.0006889307,0.0002850134,0.0007310254,0.0006728204,0.0001687596],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005346378,"about_ca_system_score_gemma":0.0007171186,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006738767,"about_ca_topic_score_gemma":0.003638842,"domain_scores_codex":[0.9998094,0.00005068353,0.00001013587,0.00002948117,0.00005297807,0.00004740674],"domain_scores_gemma":[0.9995547,0.0001928353,0.00005591518,0.00002464035,0.0001221913,0.00004984375],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001201193,0.00007299136,0.0003884369,0.00006010896,0.00002776545,0.0001202695,0.00008235272,0.9503579,0.00627796,0.009027278,0.0006476552,0.03281713],"study_design_scores_gemma":[0.00001016104,0.00004903406,0.00005687069,0.000002363947,0.000002505852,0.000008133452,0.000002937432,0.9985965,0.0003793714,0.0007863311,0.0001032289,0.000002514143],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1577089,0.0002830498,0.8333761,0.0002880664,0.000155838,0.00007482098,0.00001747884,0.0004635975,0.007632162],"genre_scores_gemma":[0.9872322,0.00003160262,0.01160279,0.00003026964,0.000008897689,0.00003933365,0.000008235485,0.00001034446,0.001036283],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006738767,"threshold_uncertainty_score":0.01339906,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01839544774908922,"score_gpt":0.2608942312425003,"score_spread":0.2424987834934111,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}