{"id":"W4400459471","doi":"10.1007/s10846-024-02118-y","title":"Deep Model-Based Reinforcement Learning for Predictive Control of Robotic Systems with Dense and Sparse Rewards","year":2024,"lang":"en","type":"article","venue":"Journal of Intelligent & Robotic Systems","topic":"Advanced Control Systems Optimization","field":"Engineering","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Artificial intelligence; Computer science; Task (project management); Machine learning; Model predictive control; Robotics; Sample (material); Field (mathematics); Bellman equation; Robot; Binary classification; Control (management); Mathematical optimization; Engineering; Support vector machine; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001673159,0.0007561363,0.001124474,0.000348477,0.0003251508,0.0007215603,0.001026983,0.000773995,0.001442997],"category_scores_gemma":[0.004896559,0.0004723587,0.0003461389,0.0002896539,0.001097512,0.0007953202,0.0009699853,0.001588596,0.0001538323],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00133834,"about_ca_system_score_gemma":0.001610857,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01015218,"about_ca_topic_score_gemma":0.00842621,"domain_scores_codex":[0.9995323,0.0001666358,0.00001954289,0.00007191674,0.000124534,0.00008503168],"domain_scores_gemma":[0.997497,0.001731509,0.0002287246,0.0001270987,0.0003100901,0.0001055761],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000282592,0.00001814933,0.0001982849,0.000014909,0.00000846984,0.00001274323,0.00001101817,0.9915795,0.0002163416,0.002050037,0.0001829605,0.005679369],"study_design_scores_gemma":[0.000002758102,0.000005572423,0.00001318801,9.890827e-7,7.078935e-7,9.09445e-7,5.763103e-7,0.9992074,0.00004464365,0.000702757,0.00001986121,6.192602e-7],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05909557,0.0003025371,0.9374113,0.0004224822,0.00004680085,0.00004393117,0.00004484487,0.0005606916,0.002071775],"genre_scores_gemma":[0.9733703,0.00005781278,0.02546124,0.00007620299,0.00001414753,0.00006158201,0.00004222946,0.00003106405,0.0008854411],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01015218,"threshold_uncertainty_score":0.02018619,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01113539873541915,"score_gpt":0.2244118147702045,"score_spread":0.2132764160347854,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}