{"id":"W4352994281","doi":"10.1007/978-3-031-25549-6_3","title":"Efficient Deep Reinforcement Learning via Policy-Extended Successor Feature Approximator","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Successor cardinal; Reinforcement learning; Computer science; Artificial intelligence; Generalization; Generalizability theory; Representation (politics); Decoupling (probability); Feature learning; Machine learning; Task (project management); Policy learning; Feature (linguistics); Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008468799,0.0006441115,0.001236528,0.0002778211,0.0003027889,0.0007618179,0.001292194,0.001119965,0.00412077],"category_scores_gemma":[0.002058628,0.0005064363,0.0003585567,0.0003480426,0.0006330594,0.0009186539,0.001298258,0.001752856,0.0006876234],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008457651,"about_ca_system_score_gemma":0.001435683,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004918677,"about_ca_topic_score_gemma":0.005914256,"domain_scores_codex":[0.9996886,0.00006434457,0.00001751548,0.00007345906,0.00009215211,0.00006398274],"domain_scores_gemma":[0.9992582,0.0004299051,0.00005614982,0.00008871935,0.0001251562,0.00004181683],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001833817,0.0001267359,0.0003663892,0.00007869209,0.00003443525,0.00007237656,0.0000358141,0.7883933,0.003542497,0.01521912,0.002882184,0.1890651],"study_design_scores_gemma":[0.000005590382,0.00001038187,0.00001690988,0.000001817398,0.000001781113,0.000005240022,9.71207e-7,0.9979637,0.0002319319,0.001660547,0.00009987099,0.000001353177],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01618578,0.0003135743,0.9792314,0.0001694143,0.00008473618,0.00003313396,0.00004479789,0.0009460963,0.002991047],"genre_scores_gemma":[0.8295059,0.000157931,0.1638134,0.0001354467,0.00005221927,0.0001263927,0.0001278224,0.0001211668,0.005959799],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004918677,"threshold_uncertainty_score":0.01378536,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0133481828389032,"score_gpt":0.2506453274915137,"score_spread":0.2372971446526105,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}