{"id":"W3196801835","doi":"10.48550/arxiv.2109.00157","title":"A Survey of Exploration Methods in Reinforcement Learning","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Reinforcement; Computer science; Artificial intelligence; Machine learning; Error-driven learning; Component (thermodynamics); Psychology; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001864352,0.001280181,0.001628216,0.001082512,0.0004785061,0.001573801,0.001244592,0.001341335,0.003954363],"category_scores_gemma":[0.004363618,0.0006349074,0.001223288,0.002101877,0.001166999,0.002361835,0.001618142,0.002253126,0.001460768],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001220099,"about_ca_system_score_gemma":0.001432634,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002042986,"about_ca_topic_score_gemma":0.001403551,"domain_scores_codex":[0.9984136,0.0005233974,0.000139741,0.0002439968,0.0005949505,0.00008433952],"domain_scores_gemma":[0.9982951,0.001190972,0.00009767312,0.0001231407,0.000223809,0.00006916292],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001154868,0.0002098512,0.001777347,0.00272181,0.0001525713,0.00009731013,0.0002730135,0.1107439,0.001685615,0.2296983,0.01201281,0.6405119],"study_design_scores_gemma":[0.00007794388,0.0003448711,0.001315796,0.001177657,0.00009374037,0.0003562059,0.0001170005,0.4824062,0.001932041,0.3671286,0.1449622,0.00008781101],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"review","genre_scores_codex":[0.003393578,0.1073589,0.8713456,0.001581541,0.0004129966,0.00009126398,0.0001269598,0.0004203668,0.01526891],"genre_scores_gemma":[0.2483155,0.2041927,0.5249192,0.001467862,0.002522926,0.0009066078,0.0005881563,0.0005723597,0.01651464],"genre_candidate":"review","genre_consensus":null,"teacher_disagreement_score":0.003954363,"threshold_uncertainty_score":0.0132286,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1871962960421404,"score_gpt":0.2701565752149955,"score_spread":0.08296027917285512,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}