{"id":"W1178556121","doi":"10.1609/aaai.v29i1.9613","title":"Policy Tree: Adaptive Representation for Policy Gradient","year":2015,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Softmax function; Representation (politics); Computer science; Tree (set theory); Decision tree; Reinforcement learning; Tree structure; Mathematical optimization; Focus (optics); Base (topology); Artificial intelligence; Machine learning; Mathematics; Algorithm; Political science; Binary tree; Law; Artificial neural network","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002663243,0.001095409,0.001288222,0.0009412245,0.0005456338,0.001449297,0.001957123,0.002004304,0.00641426],"category_scores_gemma":[0.01404937,0.0006796563,0.0007805303,0.0008970893,0.001194588,0.002800903,0.001743556,0.003060036,0.001450454],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001213226,"about_ca_system_score_gemma":0.002173105,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003042133,"about_ca_topic_score_gemma":0.00256769,"domain_scores_codex":[0.9988837,0.0004882204,0.00006932088,0.0002067082,0.0002508586,0.0001012334],"domain_scores_gemma":[0.9968689,0.00201954,0.0002363769,0.0002987019,0.0004355722,0.000140903],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002334484,0.0001320578,0.0009803402,0.0001489745,0.00004800091,0.00009896752,0.0001583523,0.6915336,0.00210766,0.07980519,0.007705629,0.2170477],"study_design_scores_gemma":[0.00001593607,0.00002494218,0.00003497278,0.00001362145,0.000004782787,0.00001556682,0.000005519641,0.9783838,0.0004445439,0.0199853,0.001064705,0.000006321327],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.004235291,0.000176851,0.9929471,0.0002159726,0.00005540429,0.00006858345,0.00007753186,0.0009760953,0.001247221],"genre_scores_gemma":[0.3939346,0.0005309758,0.5991588,0.0005081113,0.0001226321,0.0008112881,0.0005536909,0.0006819793,0.003697934],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00641426,"threshold_uncertainty_score":0.02145785,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1818638819922812,"score_gpt":0.3610695971998846,"score_spread":0.1792057152076035,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}