{"id":"W3090658167","doi":"10.1109/ijcnn48605.2020.9207473","title":"Automatic Policy Decomposition through Abstract State Space Dynamic Specialization","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Royal Military College of Canada","funders":"","keywords":"Reinforcement learning; Computer science; Bottleneck; State space; Artificial intelligence; State (computer science); Q-learning; Space (punctuation); Bellman equation; Decomposition; Function (biology); Macro; Action (physics); Machine learning; Mathematical optimization; Algorithm; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00007227125,0.0001272526,0.0001229648,0.00005884446,0.00009732258,0.0002731488,0.0004692765,0.00003564821,0.0001779405],"category_scores_gemma":[0.00006769437,0.0001236312,0.00004404911,0.0005371169,0.00002242655,0.0008614155,0.0001458002,0.0001017555,0.0005528627],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001090624,"about_ca_system_score_gemma":0.00009288621,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005531569,"about_ca_topic_score_gemma":0.000008421244,"domain_scores_codex":[0.9989144,0.00003146248,0.0002614141,0.0002545401,0.0003053337,0.0002328486],"domain_scores_gemma":[0.9994045,0.00004652914,0.000138498,0.0002718718,0.00004947866,0.00008916189],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000002396542,0.00001640198,0.00009190635,0.00004590845,0.00001729409,0.000009754748,0.003263791,0.9253076,0.00104567,0.06058619,0.001185598,0.00842746],"study_design_scores_gemma":[0.0001745303,0.00007005287,0.003154871,0.00001333734,0.000002964117,0.000003492134,0.00001639981,0.9923275,0.0006763621,0.00137469,0.002040322,0.0001455069],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.005718234,0.000006683702,0.9700404,0.009433285,0.0001327846,0.0001661042,7.013809e-7,0.0005418412,0.01395994],"genre_scores_gemma":[0.871002,0.00002381809,0.1261467,0.002235119,0.00009457426,0.000002959232,0.0000142419,0.00001411688,0.0004664578],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8652838,"threshold_uncertainty_score":0.7106116,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01841084822761371,"score_gpt":0.3002906061636799,"score_spread":0.2818797579360662,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}