{"id":"W4352994281","doi":"10.1007/978-3-031-25549-6_3","title":"Efficient Deep Reinforcement Learning via Policy-Extended Successor Feature Approximator","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Successor cardinal; Reinforcement learning; Computer science; Artificial intelligence; Generalization; Generalizability theory; Representation (politics); Decoupling (probability); Feature learning; Machine learning; Task (project management); Policy learning; Feature (linguistics); Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication","open_science","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.00162027,0.001090109,0.0009389219,0.00234952,0.0007789225,0.001211025,0.005779724,0.0006739221,0.00002823784],"category_scores_gemma":[0.000499665,0.001021557,0.0003114085,0.002192985,0.0007979212,0.000407919,0.003896101,0.002578834,0.0003945629],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001065569,"about_ca_system_score_gemma":0.000986351,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000047062,"about_ca_topic_score_gemma":0.00002325286,"domain_scores_codex":[0.9920106,0.00008773438,0.0009655756,0.002390832,0.002811792,0.001733483],"domain_scores_gemma":[0.9953309,0.0006840029,0.0008313222,0.002276765,0.0004514326,0.000425616],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000007393078,0.00001404294,0.00002171841,0.0000819732,0.00002706171,0.0001034539,0.0008159119,0.9000931,0.00005866151,0.02206216,0.00001976202,0.07669473],"study_design_scores_gemma":[0.0004198753,0.0002974547,0.00006966686,0.0004105356,0.00001511853,0.00007088158,5.830062e-7,0.9862104,0.0003245995,0.009909145,0.00121244,0.001059312],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00001328718,0.0001369459,0.9880436,0.001555906,0.00243227,0.0008844127,9.017525e-7,0.0009258796,0.006006792],"genre_scores_gemma":[0.3632488,0.0001658179,0.5993195,0.00377364,0.002788804,0.0001108889,0.00007816938,0.0004462118,0.0300682],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.3887241,"threshold_uncertainty_score":0.9998258,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0133481828389032,"score_gpt":0.2506453274915137,"score_spread":0.2372971446526105,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}