{"id":"W3206746211","doi":"10.1109/icra48506.2021.9561017","title":"Distilling a Hierarchical Policy for Planning and Control via Representation and Reinforcement Learning","year":2021,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Reinforcement learning; Computer science; Hierarchy; Task (project management); Set (abstract data type); Artificial intelligence; Representation (politics); Control (management); Latent variable; Imitation; Machine learning; Sequence (biology); State space; Human–computer interaction; Engineering; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002311861,0.0001074473,0.0001518672,0.00008909976,0.0002466957,0.0002844769,0.0001205763,0.0000442957,0.000004839034],"category_scores_gemma":[0.0004958879,0.0001028699,0.0000318143,0.0001752354,0.00004148649,0.0002797799,0.0001881046,0.0001472585,0.000001528239],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002672647,"about_ca_system_score_gemma":0.00005438857,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002400001,"about_ca_topic_score_gemma":7.421995e-7,"domain_scores_codex":[0.9989325,0.00005696313,0.0002369422,0.0003347494,0.0001849514,0.0002539132],"domain_scores_gemma":[0.9991106,0.00042903,0.00008488623,0.0001922296,0.00007987003,0.0001033486],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001741402,0.000004831519,0.006083015,0.00003705374,0.00003268762,0.00001305162,0.001155682,0.9227993,0.0016937,0.05708624,0.00002658296,0.01105048],"study_design_scores_gemma":[0.0008637529,0.0001104704,0.001978134,0.00002578817,0.000009179085,0.00003874248,0.00008230972,0.9938407,0.000806858,0.001100164,0.00101864,0.0001252554],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002816148,0.0000911785,0.9939978,0.001478117,0.00006784526,0.0001885254,1.70133e-7,0.00009724073,0.001263009],"genre_scores_gemma":[0.9384687,0.00001914244,0.05933425,0.0004628458,0.00008468239,0.00001899767,0.000008800356,0.000008347676,0.001594232],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9356526,"threshold_uncertainty_score":0.4194913,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02105963226643383,"score_gpt":0.2996378555369776,"score_spread":0.2785782232705438,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}