{"id":"W4293741188","doi":"10.3389/frobt.2022.854212","title":"Model-free reinforcement learning for robust locomotion using demonstrations from trajectory optimization","year":2022,"lang":"en","type":"article","venue":"Frontiers in Robotics and AI","topic":"Robotic Locomotion and Control","field":"Engineering","cited_by":25,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"European Commission; York University; National Science Foundation","keywords":"Computer science; Reinforcement learning; Bounding overwatch; Robustness (evolution); Robot; Trajectory optimization; Trajectory; Task (project management); Artificial intelligence","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000875645,0.0007280202,0.0008068483,0.0002738485,0.0002648513,0.0004899552,0.001092477,0.0007517844,0.002233916],"category_scores_gemma":[0.003145079,0.0005099535,0.000444371,0.0001928609,0.0008847627,0.0007128895,0.00132703,0.001429333,0.0003926049],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005811775,"about_ca_system_score_gemma":0.000924231,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002499427,"about_ca_topic_score_gemma":0.002391249,"domain_scores_codex":[0.9997001,0.00008758791,0.00001569868,0.00005548515,0.0001027781,0.00003840515],"domain_scores_gemma":[0.9989015,0.0006553138,0.0001344644,0.0001207788,0.000123765,0.00006423053],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003206453,0.00002370745,0.0001588023,0.00003643981,0.00001346969,0.0000367636,0.00002075482,0.9780955,0.002137551,0.004493516,0.0002447006,0.01470665],"study_design_scores_gemma":[0.000004790698,0.00001535353,0.00001694038,0.000001992479,0.00000114867,0.000004475376,9.173672e-7,0.9983632,0.0003263541,0.001172006,0.00009129676,0.000001626369],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01018797,0.00006890828,0.9881027,0.00006096144,0.00001263872,0.00003020356,0.00001627108,0.0005588359,0.0009615052],"genre_scores_gemma":[0.8282596,0.00009183135,0.169619,0.00006284778,0.00002078434,0.0002133598,0.00009104201,0.0001646496,0.001476841],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002499427,"threshold_uncertainty_score":0.00747323,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01804940780921152,"score_gpt":0.207816876286318,"score_spread":0.1897674684771065,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}