{"id":"W2996347495","doi":"","title":"Exploring Model-based Planning with Policy Networks","year":2020,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Benchmarking; Artificial neural network; Code (set theory); Artificial intelligence; Mathematical optimization; Sample (material); Optimization problem; Control (management); Action (physics); State space; Machine learning; Algorithm; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009269963,0.001062986,0.0009235229,0.000488079,0.0003706635,0.000851089,0.0009581726,0.001009437,0.002099963],"category_scores_gemma":[0.003446482,0.0007407936,0.0006096532,0.0004319307,0.00120282,0.001427264,0.001196824,0.001362926,0.0002786094],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001151736,"about_ca_system_score_gemma":0.001630904,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007857161,"about_ca_topic_score_gemma":0.007104903,"domain_scores_codex":[0.9995833,0.0001744283,0.00001900826,0.00009209206,0.00008247737,0.00004863034],"domain_scores_gemma":[0.9987085,0.001004539,0.00008075763,0.00009050672,0.00007716742,0.00003856243],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002531414,0.00002186346,0.0001826444,0.0000247633,0.00001203715,0.00002003535,0.00001917754,0.9814368,0.0003644202,0.005751365,0.0002318562,0.01190967],"study_design_scores_gemma":[0.000004908865,0.000007035508,0.00001237407,0.000002048954,0.000001613053,0.000002173011,0.000002108329,0.9964239,0.0001102284,0.003313296,0.0001191598,0.000001176906],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03994743,0.0002895816,0.9557342,0.0003145888,0.00003168009,0.00004644298,0.00005955014,0.0008286273,0.002747982],"genre_scores_gemma":[0.8175042,0.0002921406,0.1794038,0.000186685,0.00003113269,0.0002637197,0.000181228,0.0001478838,0.001989306],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007857161,"threshold_uncertainty_score":0.01562285,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2271002738639286,"score_gpt":0.1933980817790889,"score_spread":0.03370219208483966,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}