{"id":"W4221161302","doi":"10.48550/arxiv.2201.01666","title":"Sample Efficient Deep Reinforcement Learning via Uncertainty Estimation","year":2022,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Reinforcement learning; Weighting; Computer science; Heteroscedasticity; Variance (accounting); Sample (material); Noise (video); Bayesian probability; Probabilistic logic; Process (computing); Artificial intelligence; Machine learning; Mathematical optimization; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003019582,0.001263293,0.001814437,0.0003925395,0.0004149161,0.0009819597,0.00138648,0.001146605,0.001545436],"category_scores_gemma":[0.01299356,0.0007711484,0.0005543542,0.0004297174,0.001611782,0.002175624,0.002147041,0.002256473,0.0002869991],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00128131,"about_ca_system_score_gemma":0.001530507,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003589947,"about_ca_topic_score_gemma":0.003301438,"domain_scores_codex":[0.9985994,0.0005837482,0.00006702743,0.0002560698,0.0003742985,0.0001195196],"domain_scores_gemma":[0.9940414,0.004305342,0.0004699261,0.0005268567,0.0004879613,0.0001684623],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001278261,0.00006086508,0.0005875338,0.00006360358,0.00004581994,0.0000355697,0.00005901655,0.9445383,0.001099661,0.01395673,0.0006142278,0.03881077],"study_design_scores_gemma":[0.000006794012,0.00001711432,0.00003809949,0.000003806406,0.000003307719,0.000004832677,0.000002131335,0.9933082,0.0003413295,0.006192018,0.00007962303,0.000002752752],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0152217,0.0001688124,0.9832985,0.0001616226,0.00001826905,0.00003089485,0.00002364403,0.0003615512,0.0007149266],"genre_scores_gemma":[0.8550052,0.000156521,0.1428254,0.000171727,0.00005123916,0.0001730046,0.0001168929,0.0001496867,0.001350387],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003589947,"threshold_uncertainty_score":0.01596928,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04515281141028903,"score_gpt":0.2008328326548355,"score_spread":0.1556800212445464,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}