{"id":"W3092277419","doi":"10.1145/3424636.3426907","title":"Learning to Locomote: Understanding How Environment Design Matters for Deep Reinforcement Learning","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Human Motion and Animation","field":"Engineering","cited_by":41,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Animation; Action (physics); Embodied cognition; Embodied agent; Control (management); Machine learning; Human–computer interaction","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002367277,0.0005174769,0.0006132439,0.0002638969,0.0004050074,0.001760126,0.0007811126,0.001316925,0.002932884],"category_scores_gemma":[0.01502434,0.0004584556,0.0003293671,0.0002401711,0.001815887,0.004411372,0.001432453,0.002355031,0.0003650662],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000872597,"about_ca_system_score_gemma":0.0007916779,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001703763,"about_ca_topic_score_gemma":0.002179453,"domain_scores_codex":[0.9991721,0.0004664649,0.00002721682,0.0001519971,0.00008867305,0.00009363356],"domain_scores_gemma":[0.9964244,0.002632401,0.0002693195,0.0002550094,0.0001853183,0.0002336454],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003434862,0.0001450539,0.005603348,0.0002562722,0.00007959714,0.0001461878,0.0004446084,0.7207943,0.007472361,0.1927591,0.001945494,0.0700102],"study_design_scores_gemma":[0.00003969091,0.0000769683,0.0009064941,0.0000492677,0.00001885181,0.00002489072,0.00007243838,0.8490352,0.001520325,0.1466186,0.001620402,0.000016912],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1823501,0.001117804,0.8012506,0.003478355,0.0001088144,0.00004822911,0.0001353666,0.0004014712,0.01110936],"genre_scores_gemma":[0.9307934,0.0005360325,0.06579071,0.0003275592,0.00004528803,0.00008205225,0.00007645413,0.0001917796,0.002156751],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002932884,"threshold_uncertainty_score":0.01251948,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06800566205986544,"score_gpt":0.2325204118775476,"score_spread":0.1645147498176821,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}