{"id":"W2966234803","doi":"10.24963/ijcai.2019/445","title":"Hill Climbing on Value Estimates for Search-control in Dyna","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; University of Toronto; Huawei Technologies (Canada); University of Alberta","funders":"","keywords":"Hill climbing; Climbing; Reinforcement learning; Trajectory; Computer science; Bellman equation; Langevin dynamics; Value (mathematics); Sampling (signal processing); Artificial intelligence; Mathematical optimization; Mathematics; Machine learning; Statistics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001805249,0.0007023618,0.001072268,0.0006219122,0.0006701231,0.001055198,0.001113118,0.000889909,0.002251586],"category_scores_gemma":[0.01120483,0.0006455444,0.0005302739,0.0004210113,0.00175897,0.001363218,0.001263374,0.001659367,0.0003228349],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001460524,"about_ca_system_score_gemma":0.001546247,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005060378,"about_ca_topic_score_gemma":0.005857378,"domain_scores_codex":[0.9992698,0.0003426265,0.00003839367,0.0001208279,0.0001657176,0.00006270139],"domain_scores_gemma":[0.9962432,0.002602849,0.0003360355,0.0003184754,0.0003381096,0.0001613313],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006137623,0.00003095502,0.0009306394,0.00005052539,0.00002545793,0.00004949298,0.00009193477,0.9411657,0.0008326317,0.04154067,0.0005862577,0.0146343],"study_design_scores_gemma":[0.000007347784,0.00001603767,0.00006941099,0.000006841688,0.00000244457,0.000007334685,0.000005398616,0.9892898,0.0002487316,0.01011181,0.000230621,0.000004279631],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03308062,0.0001564655,0.9639745,0.0002289102,0.00002811324,0.00005383474,0.0000413024,0.0003692836,0.002067033],"genre_scores_gemma":[0.7687293,0.0001567029,0.2276989,0.0001540691,0.00002632004,0.0002756561,0.0001316243,0.0001990715,0.002628277],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005060378,"threshold_uncertainty_score":0.01059687,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03048606102777367,"score_gpt":0.2988601795223709,"score_spread":0.2683741184945972,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}