{"id":"W3090658167","doi":"10.1109/ijcnn48605.2020.9207473","title":"Automatic Policy Decomposition through Abstract State Space Dynamic Specialization","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Royal Military College of Canada","funders":"","keywords":"Reinforcement learning; Computer science; Bottleneck; State space; Artificial intelligence; State (computer science); Q-learning; Space (punctuation); Bellman equation; Decomposition; Function (biology); Macro; Action (physics); Machine learning; Mathematical optimization; Algorithm; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009430054,0.0006986746,0.001037241,0.0005758228,0.0003187411,0.0009189371,0.0008778036,0.0007802782,0.003895774],"category_scores_gemma":[0.00272931,0.0005495491,0.0008185538,0.000525217,0.001004101,0.002122116,0.002130684,0.001662498,0.0005780515],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001168733,"about_ca_system_score_gemma":0.001173742,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003327851,"about_ca_topic_score_gemma":0.003946904,"domain_scores_codex":[0.9994795,0.000142033,0.00003453462,0.0001608246,0.00009729514,0.00008574957],"domain_scores_gemma":[0.9991041,0.0004618664,0.00009319317,0.0001725273,0.00009200042,0.0000764496],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001364793,0.00006770481,0.0008786584,0.00008950051,0.00004241726,0.0001233011,0.0001581758,0.8238918,0.004634095,0.05179463,0.002117691,0.1160655],"study_design_scores_gemma":[0.000004316615,0.000007160739,0.00004431534,0.000004171233,0.00000246018,0.000006597311,0.000004894389,0.9778714,0.0004459636,0.02126445,0.0003420061,0.000002280613],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01945191,0.0001290534,0.9774265,0.0001792611,0.0000192105,0.00003441366,0.0000882664,0.001080305,0.001591123],"genre_scores_gemma":[0.8027123,0.0002312899,0.1925323,0.0002030384,0.00003603238,0.0001795605,0.000462173,0.0002457838,0.003397521],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003895774,"threshold_uncertainty_score":0.01303262,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01841084822761371,"score_gpt":0.3002906061636799,"score_spread":0.2818797579360662,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}