{"id":"W3185313117","doi":"","title":"Exploration-Driven Representation Learning in Reinforcement Learning","year":2021,"lang":"en","type":"article","venue":"International Conference on Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Representation (politics); Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002876054,0.0005939637,0.001413371,0.0004703724,0.0003965852,0.001063882,0.00169917,0.001617069,0.003295768],"category_scores_gemma":[0.009238759,0.0006711719,0.0005677511,0.0006179644,0.001563174,0.002080952,0.002137799,0.002305597,0.0003331587],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001106864,"about_ca_system_score_gemma":0.001242948,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003642827,"about_ca_topic_score_gemma":0.002481311,"domain_scores_codex":[0.9990627,0.0005491183,0.00004100285,0.0001275792,0.0001330508,0.00008659947],"domain_scores_gemma":[0.9951717,0.003901067,0.0001741146,0.0002197694,0.0003891447,0.0001441453],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001328401,0.00008668648,0.0005919084,0.000105463,0.00005464225,0.00004582067,0.0000826308,0.8887298,0.0008470824,0.05299797,0.001283301,0.05504181],"study_design_scores_gemma":[0.000009212011,0.00001848942,0.00002240656,0.000004238905,0.000002887238,0.0000042263,0.000002590736,0.9889577,0.0001150787,0.01072654,0.0001341423,0.000002505182],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01690266,0.000500291,0.9800181,0.0004054369,0.00006573708,0.00003354884,0.00003128483,0.0001842333,0.001858674],"genre_scores_gemma":[0.8632681,0.0004132143,0.1304999,0.0002475424,0.00008139807,0.0002408502,0.0001140722,0.0001217306,0.005013236],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003642827,"threshold_uncertainty_score":0.01521021,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06580708750621926,"score_gpt":0.328197841659694,"score_spread":0.2623907541534747,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}