{"id":"W3213430327","doi":"","title":"Pretraining Representations for Data-Efficient Reinforcement Learning","year":2021,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"HEC Montréal; Université de Montréal","funders":"","keywords":"Computer science; Reinforcement learning; Task (project management); Encoder; Artificial intelligence; Representation (politics); Key (lock); Code (set theory); Machine learning; Feature learning; External Data Representation","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003766523,0.0001400318,0.0001486245,0.000109734,0.0004004776,0.0001605239,0.001237576,0.00005933742,0.00004997982],"category_scores_gemma":[0.0004279142,0.0001718378,0.00008129219,0.0007529719,0.00005120546,0.0005462465,0.001035699,0.0002018565,0.00005145543],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009541949,"about_ca_system_score_gemma":0.000180329,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001301154,"about_ca_topic_score_gemma":0.000003284797,"domain_scores_codex":[0.998423,0.00008913387,0.0002106321,0.0007845548,0.0001282481,0.0003643758],"domain_scores_gemma":[0.9978207,0.0002910554,0.0001624174,0.001382103,0.0002276823,0.0001160383],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000005366356,0.0000165101,0.0006604356,0.00001304031,0.00003682429,0.00004959484,0.0002940977,0.859218,0.00008015295,0.1389321,0.0002774709,0.0004164378],"study_design_scores_gemma":[0.0005467739,0.00005971561,0.0001488663,0.00002089271,0.00003055027,0.000006391334,0.0003457834,0.9926105,0.000276533,0.0002826951,0.005476141,0.0001951161],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.004262079,0.00001691087,0.9870611,0.0001400531,0.00027823,0.0002065907,0.000001715967,0.0002086036,0.007824678],"genre_scores_gemma":[0.9667357,0.00002239227,0.0222268,0.0001041721,0.00004233778,0.000001188389,0.00009943209,0.00001235832,0.01075559],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9648343,"threshold_uncertainty_score":0.7007341,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1512692487779895,"score_gpt":0.2391395437057067,"score_spread":0.08787029492771722,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}