{"id":"W4393159809","doi":"10.1609/aaai.v38i10.28970","title":"Contextual Pre-planning on Reward Machine Abstractions for Enhanced Transfer in Deep Reinforcement Learning","year":2024,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Reinforcement; Artificial intelligence; Transfer of learning; Cognitive psychology; Machine learning; Psychology; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006964034,0.0002783141,0.0002901996,0.0002957643,0.0002228323,0.0004260055,0.001201656,0.0001143967,0.00006362957],"category_scores_gemma":[0.000452468,0.0002227545,0.0001596846,0.000540514,0.0001282326,0.0005045348,0.0001291912,0.0007423408,0.0000712853],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001246522,"about_ca_system_score_gemma":0.00009956447,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002665283,"about_ca_topic_score_gemma":0.000008169053,"domain_scores_codex":[0.9976549,0.00001962025,0.0007653238,0.0005735266,0.0005305817,0.0004560348],"domain_scores_gemma":[0.9988731,0.0003668294,0.0001589049,0.0002333999,0.000291394,0.00007635111],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001222057,0.0000406045,0.00003064684,0.0001002044,0.00002224201,8.264607e-7,0.004776701,0.5293257,0.01788853,0.4275047,0.00002298042,0.02016463],"study_design_scores_gemma":[0.00005151614,0.0005657937,0.00004953535,0.000660761,0.00001159936,0.000001997695,0.0005071908,0.7577788,0.2340603,0.005842123,0.0002635839,0.0002068372],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02454128,0.00003481639,0.9624026,0.001188188,0.0006325775,0.0007867698,0.000001215345,0.0001801269,0.01023241],"genre_scores_gemma":[0.9971423,0.00003334369,0.001401358,0.000129672,0.00006486572,0.0001098107,0.000002133105,0.00002243909,0.001094049],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9726011,"threshold_uncertainty_score":0.9083665,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06539929613530107,"score_gpt":0.3236612521517399,"score_spread":0.2582619560164389,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}