{"id":"W3165914412","doi":"","title":"Pretraining Reward-Free Representations for Data-Efficient Reinforcement Learning","year":2021,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"HEC Montréal; Université de Montréal","funders":"","keywords":"Reinforcement learning; Computer science; Reinforcement; Artificial intelligence; Cognitive psychology; Machine learning; Human–computer interaction; Psychology; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001177944,0.001013942,0.001158348,0.000513427,0.0003801138,0.0009448378,0.001928044,0.001655894,0.00520735],"category_scores_gemma":[0.008243588,0.0007683203,0.0005442366,0.0005792119,0.0008327863,0.002135003,0.001875446,0.00404458,0.001163753],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009174573,"about_ca_system_score_gemma":0.001829553,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003888599,"about_ca_topic_score_gemma":0.0058513,"domain_scores_codex":[0.9994009,0.0001534029,0.00003904166,0.0001412357,0.0001559061,0.0001095303],"domain_scores_gemma":[0.9970729,0.001726179,0.0001699792,0.0004751346,0.0004399218,0.0001158658],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003761877,0.0003536829,0.001110588,0.0001526209,0.00006206656,0.00009968751,0.00008512425,0.6467198,0.007822943,0.01610433,0.005601181,0.3215117],"study_design_scores_gemma":[0.00001802251,0.00003168353,0.00005399448,0.000006783714,0.000004329706,0.00001141771,0.000004878778,0.9916238,0.001628482,0.006329594,0.0002826983,0.000004198195],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02430886,0.0001803319,0.9714717,0.0002846629,0.00007441799,0.00007238128,0.0001493345,0.002066538,0.001391705],"genre_scores_gemma":[0.7672386,0.0001332092,0.2283617,0.0002683761,0.00004852336,0.0003248257,0.0005621524,0.0002491511,0.002813398],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00520735,"threshold_uncertainty_score":0.01742035,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1276518897166019,"score_gpt":0.378776401532506,"score_spread":0.2511245118159041,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}