{"id":"W4390489156","doi":"10.1109/ssci52147.2023.10371909","title":"Hierarchical Reinforcement Learning for Non-Stationary Environments","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University; Carleton University","funders":"","keywords":"Reinforcement learning; Computer science; Process (computing); Temporal difference learning; Action (physics); Differential game; Term (time); Artificial intelligence; Mathematical optimization; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0003369124,0.0001200099,0.0001074845,0.0001514294,0.0002423651,0.0000924842,0.0005282558,0.0000482027,0.0000546542],"category_scores_gemma":[0.00008943519,0.0001144298,0.00006782429,0.0003036234,0.00003150682,0.0003135327,0.0003451304,0.0001512748,0.0008292332],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000523792,"about_ca_system_score_gemma":0.00003411085,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000002228945,"about_ca_topic_score_gemma":1.12835e-7,"domain_scores_codex":[0.9986373,0.00002607011,0.0002565327,0.0003002285,0.000395605,0.0003842658],"domain_scores_gemma":[0.9992494,0.0002579344,0.00007375539,0.0003115477,0.00001689502,0.00009049701],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000004025218,0.00000545968,0.0003525137,0.00001013983,0.0000145298,0.000003355051,0.0002846523,0.970359,0.0004092193,0.02128685,0.003350306,0.003919946],"study_design_scores_gemma":[0.0003655864,0.0001765897,0.002215235,0.000007083129,0.000002651998,0.000001265615,0.0000325648,0.9449158,0.0004151643,0.0005364487,0.0511919,0.0001397267],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0007232288,0.000001633358,0.9895923,0.0008149978,0.0002279239,0.0003198936,1.560502e-7,0.0003090241,0.008010803],"genre_scores_gemma":[0.7816303,0.0000400873,0.09370934,0.0006912585,0.0001035291,0.0001640143,0.0001170357,0.00002860696,0.1235158],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.895883,"threshold_uncertainty_score":0.9999487,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02224100655619201,"score_gpt":0.2667573711172034,"score_spread":0.2445163645610114,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}