{"id":"W2817967199","doi":"10.1613/jair.1.12463","title":"The Bottleneck Simulator: A Model-Based Deep Reinforcement Learning Approach","year":2020,"lang":"en","type":"preprint","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; Université de Montréal; Polytechnique Montréal; Mila - Quebec Artificial Intelligence Institute","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Compute Canada; Amazon Web Services; Nuance Foundation; Canadian Institute for Advanced Research; Nvidia","keywords":"Bottleneck; Reinforcement learning; Computer science; Variance (accounting); Task (project management); Obstacle; Artificial intelligence; State space; Engineering; Statistics; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002220534,0.000825707,0.001172533,0.0005155936,0.0003128567,0.0006754214,0.002461623,0.001194487,0.002029599],"category_scores_gemma":[0.005421053,0.0006536679,0.0006076131,0.0003295942,0.0009568813,0.001380081,0.001606604,0.002274984,0.0003501729],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001168651,"about_ca_system_score_gemma":0.002060269,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005216222,"about_ca_topic_score_gemma":0.004409615,"domain_scores_codex":[0.9994335,0.000267086,0.00002289778,0.00008923807,0.0001234791,0.0000639436],"domain_scores_gemma":[0.9977538,0.001416313,0.000184848,0.0002127202,0.0002685204,0.0001638017],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000732745,0.00004623044,0.0003962856,0.00002542351,0.00003113765,0.0000247717,0.0000214138,0.9791047,0.000841835,0.004663364,0.0003779765,0.01439365],"study_design_scores_gemma":[0.000003890369,0.00001050277,0.00001055345,9.832464e-7,0.000001181834,0.000001624485,6.248345e-7,0.9987715,0.0001159911,0.001039762,0.00004208276,0.000001222942],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01927142,0.00009373725,0.9785452,0.0001679106,0.0000290121,0.00005272475,0.00004500249,0.0009110093,0.0008838736],"genre_scores_gemma":[0.7849011,0.0001253184,0.2118689,0.0002036616,0.00003769122,0.0002822304,0.0001892908,0.000219639,0.002172042],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005216222,"threshold_uncertainty_score":0.01174343,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2009218619017905,"score_gpt":0.4031124373316161,"score_spread":0.2021905754298257,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}