{"id":"W3204811657","doi":"","title":"LOCO: Adaptive exploration in reinforcement learning via local estimation of contraction coefficients","year":2021,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Reinforcement learning; Contraction (grammar); Computer science; Reinforcement; Artificial intelligence; Control theory (sociology); Mathematics; Mathematical optimization; Engineering; Structural engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001226708,0.0007911137,0.001156918,0.0004670208,0.0003591716,0.0008674663,0.001754437,0.001363646,0.005199634],"category_scores_gemma":[0.003947115,0.0004943167,0.0003994802,0.0003945683,0.0009939972,0.001500304,0.002480543,0.001624546,0.0009483136],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004874688,"about_ca_system_score_gemma":0.0008720825,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002124987,"about_ca_topic_score_gemma":0.002599864,"domain_scores_codex":[0.9996768,0.0001006004,0.00001425751,0.00007497714,0.00009081679,0.00004247815],"domain_scores_gemma":[0.9990753,0.0004978036,0.0000757314,0.0001338968,0.0001196177,0.00009769336],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006252609,0.0002812261,0.001362297,0.0002771499,0.00009247206,0.000195245,0.0001846002,0.5905268,0.01129335,0.03340147,0.008827946,0.3529322],"study_design_scores_gemma":[0.00002491328,0.00003820654,0.00005111713,0.000005715662,0.000003531612,0.00001362734,0.000004604035,0.9956803,0.0008137893,0.002935369,0.0004236074,0.000005177058],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01231952,0.0001745733,0.9832496,0.0001263406,0.00007483478,0.00005543118,0.0000572983,0.002443936,0.001498313],"genre_scores_gemma":[0.6783382,0.000155811,0.3140426,0.0002424896,0.0000816809,0.0004872496,0.0002230761,0.0008891205,0.005539875],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005199634,"threshold_uncertainty_score":0.01739454,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05327394860873119,"score_gpt":0.3325311095023211,"score_spread":0.2792571608935899,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}