{"id":"W7124326325","doi":"10.65109/sjgw9760","title":"Empowering Generalization for Deep Reinforcement Learning via Symbolic Planning","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Generalization; Pearl; Planner; Limiting; Automated planning and scheduling; Plan (archaeology); Deep learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0009658503,0.0005972619,0.0005693552,0.0006915769,0.001303137,0.001115758,0.00134294,0.0003018034,0.0001716269],"category_scores_gemma":[0.0004001484,0.0006692622,0.0002767624,0.001300597,0.00008930647,0.0009378238,0.0009864665,0.000533171,0.00006963349],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000404392,"about_ca_system_score_gemma":0.0002750051,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004324607,"about_ca_topic_score_gemma":0.000002379883,"domain_scores_codex":[0.9955925,0.0001481325,0.001360845,0.001029589,0.0006445522,0.001224334],"domain_scores_gemma":[0.9975845,0.0003269825,0.0005472719,0.0009034724,0.0004432628,0.000194523],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003043476,0.00001959558,0.0017538,0.0002937861,0.0001754238,0.00000374907,0.002552337,0.9311686,0.0007833435,0.04914995,0.0004368257,0.01363214],"study_design_scores_gemma":[0.001136158,0.0003731364,0.0002437251,0.0004341434,0.00009042551,0.000004343658,0.0001843014,0.972623,0.002787423,0.0003093999,0.02118115,0.0006328049],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0004467044,0.0005841643,0.9662664,0.0005108673,0.002976453,0.001250487,1.344311e-7,0.0004397577,0.02752506],"genre_scores_gemma":[0.8922126,0.0001216895,0.05026512,0.001269056,0.0002629803,0.0001213248,0.00004812659,0.00005604805,0.05564304],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9160013,"threshold_uncertainty_score":0.999997,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01938705309028489,"score_gpt":0.3083807653414555,"score_spread":0.2889937122511707,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}