{"id":"W3202097587","doi":"","title":"Learning One Representation to Optimize All Rewards","year":2021,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Representation (politics); Markov decision process; Artificial intelligence; Reinforcement learning; A priori and a posteriori; Machine learning; Markov process; Temporal difference learning; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007935601,0.00115237,0.00108602,0.0004840825,0.0003357542,0.001209875,0.001493048,0.001535766,0.00418278],"category_scores_gemma":[0.003573019,0.0004529767,0.0006503341,0.0004938149,0.0009367267,0.002236883,0.001521267,0.002207001,0.0009996616],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001028848,"about_ca_system_score_gemma":0.001748765,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002155302,"about_ca_topic_score_gemma":0.002631182,"domain_scores_codex":[0.9994866,0.000145674,0.00002371974,0.0001620523,0.00009086708,0.0000910508],"domain_scores_gemma":[0.9992404,0.0003300628,0.00008532734,0.000169946,0.0001050089,0.00006928456],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001161471,0.0001156755,0.0006039432,0.00008852842,0.00004630242,0.000065874,0.00006557727,0.8505257,0.002053009,0.05762158,0.002984689,0.08571283],"study_design_scores_gemma":[0.00001300437,0.0000364458,0.000049435,0.000009157401,0.000007690132,0.00001557658,0.00000495031,0.974234,0.0007040763,0.02430378,0.0006144056,0.000007433046],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01270208,0.0001544819,0.983266,0.0004043854,0.00004695354,0.00003501672,0.000148979,0.0006605607,0.002581574],"genre_scores_gemma":[0.6829805,0.000381755,0.3074086,0.0002911112,0.0001033279,0.0003181518,0.0004996384,0.0003033586,0.007713533],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00418278,"threshold_uncertainty_score":0.01399279,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04197470961659992,"score_gpt":0.2918375720424113,"score_spread":0.2498628624258114,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}