{"id":"W3213430327","doi":"","title":"Pretraining Representations for Data-Efficient Reinforcement Learning","year":2021,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"HEC Montréal; Université de Montréal","funders":"","keywords":"Computer science; Reinforcement learning; Task (project management); Encoder; Artificial intelligence; Representation (politics); Key (lock); Code (set theory); Machine learning; Feature learning; External Data Representation","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002054347,0.001399281,0.001325858,0.0004260606,0.0004605569,0.001211508,0.002474958,0.001386109,0.004495142],"category_scores_gemma":[0.01309662,0.0008418622,0.0006643087,0.0005152151,0.001519179,0.002683342,0.002454173,0.005069511,0.001768143],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001202301,"about_ca_system_score_gemma":0.00200453,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004254175,"about_ca_topic_score_gemma":0.006645767,"domain_scores_codex":[0.9991135,0.000310243,0.0000467396,0.0002497915,0.0001646744,0.0001150811],"domain_scores_gemma":[0.9953754,0.002585031,0.0002867475,0.001122476,0.0004579466,0.0001723522],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002097738,0.0003056328,0.001677928,0.0001961546,0.00007621189,0.00009185643,0.0001379898,0.8023813,0.006226528,0.01607812,0.006084051,0.1665344],"study_design_scores_gemma":[0.00001958576,0.00004125723,0.0001134824,0.00001529724,0.000005483166,0.00001550946,0.00001442339,0.9830787,0.00209212,0.01376304,0.000833991,0.00000712641],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0206071,0.0001881,0.974338,0.0003722833,0.0000580147,0.0001076892,0.0002175196,0.002542119,0.001569167],"genre_scores_gemma":[0.6481318,0.0001808145,0.3461249,0.0004476508,0.00005950802,0.0005772026,0.001105581,0.0004475543,0.002924913],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004495142,"threshold_uncertainty_score":0.01503778,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1512692487779895,"score_gpt":0.2391395437057067,"score_spread":0.08787029492771722,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}