{"id":"W3101152497","doi":"","title":"On Efficiency in Hierarchical Reinforcement Learning","year":2020,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":false,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007894389,0.001277611,0.002315088,0.001086685,0.0007262702,0.002179761,0.002690543,0.001685998,0.006612667],"category_scores_gemma":[0.03881,0.0008834267,0.0009631118,0.001513894,0.003700076,0.006300599,0.003283378,0.003684916,0.0006387731],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003089639,"about_ca_system_score_gemma":0.001498281,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00374565,"about_ca_topic_score_gemma":0.002270818,"domain_scores_codex":[0.995819,0.002375241,0.0001969194,0.0004120285,0.0008037924,0.0003929434],"domain_scores_gemma":[0.9597682,0.03549071,0.0008159085,0.002008767,0.0014211,0.0004953783],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001952543,0.000110378,0.0009029418,0.0001781453,0.00007878063,0.00004769162,0.0001908548,0.3277506,0.001085979,0.6242138,0.002122822,0.04312288],"study_design_scores_gemma":[0.00004107354,0.00004266058,0.0002647526,0.00002391171,0.00002328696,0.00001709059,0.00002045431,0.5754522,0.0005404972,0.4228132,0.0007499947,0.00001086709],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03802394,0.001735178,0.9415618,0.001726835,0.00009915572,0.00007081374,0.00008797563,0.000220911,0.01647336],"genre_scores_gemma":[0.8739032,0.001526042,0.1127439,0.0005554815,0.0002551205,0.0002648946,0.0001548123,0.0003635653,0.01023307],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007894389,"threshold_uncertainty_score":0.04174995,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02283141343365988,"score_gpt":0.2504900927677116,"score_spread":0.2276586793340517,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}