{"id":"W3043761458","doi":"","title":"Beyond Prioritized Replay: Sampling States in Model-Based RL via Simulated Priorities","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Advanced Bandit Algorithms Research","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Sampling (signal processing); Key (lock); Benchmark (surveying); Equivalence (formal languages); Sample (material); Ideal (ethics); Mathematical optimization; Sampling bias; Sample size determination; Mathematics; Statistics; Telecommunications","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004339144,0.001408397,0.001722197,0.0007255864,0.0006329272,0.001632095,0.002565771,0.001618756,0.002691914],"category_scores_gemma":[0.02191809,0.0007289224,0.0006689079,0.0006140631,0.001423736,0.003377267,0.002631898,0.002856135,0.0004933242],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001131882,"about_ca_system_score_gemma":0.001820532,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004247817,"about_ca_topic_score_gemma":0.003759657,"domain_scores_codex":[0.9981503,0.001014595,0.00007561383,0.0003191036,0.0002778497,0.0001625491],"domain_scores_gemma":[0.9910114,0.006703315,0.0005154109,0.0007758166,0.0005738183,0.0004202888],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005637576,0.0001604012,0.001560402,0.0001785754,0.0001068753,0.0001114157,0.0003189588,0.9012084,0.001868006,0.0289162,0.00168447,0.06332258],"study_design_scores_gemma":[0.0000343692,0.00006958631,0.0000720758,0.00001232908,0.00000951843,0.00001793428,0.00001808514,0.9860845,0.0004601388,0.01291626,0.0002972684,0.000007863755],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02442723,0.0003764829,0.9725879,0.0003945667,0.00004696538,0.00008430815,0.0000477515,0.0006146122,0.001420015],"genre_scores_gemma":[0.8498125,0.0002862851,0.1465647,0.0004419399,0.00009164088,0.0002478245,0.0001417444,0.0001737503,0.002239598],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004339144,"threshold_uncertainty_score":0.02294785,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2194841626866739,"score_gpt":0.3263061400790038,"score_spread":0.1068219773923299,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}