{"id":"W4409361576","doi":"10.1609/aaai.v39i25.34864","title":"ModelDiff: Symbolic Dynamic Programming for Model-Aware Policy Transfer in Deep Q-Learning","year":2025,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Transfer of learning; Transfer (computing); Dynamic programming; Artificial intelligence; Programming language; Parallel computing; Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005478131,0.000286519,0.000358549,0.0004957999,0.0002505225,0.0003184646,0.002193267,0.0001474492,0.000004319854],"category_scores_gemma":[0.0004891427,0.0002447572,0.0001673495,0.001347516,0.0001740016,0.000422241,0.0003210364,0.0005217534,0.000007671167],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001694198,"about_ca_system_score_gemma":0.000249244,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003934681,"about_ca_topic_score_gemma":0.00002094223,"domain_scores_codex":[0.9976828,0.00002032966,0.0007186862,0.0005784034,0.000405475,0.0005943528],"domain_scores_gemma":[0.998798,0.0001315994,0.0001791594,0.0003000364,0.0005232928,0.00006793076],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003149064,0.00004931485,0.0001026213,0.00008920532,0.00001029116,1.039044e-7,0.001330053,0.3641958,0.001934314,0.577323,0.000002644036,0.05493113],"study_design_scores_gemma":[0.00006132804,0.0001188357,0.0000412948,0.0003255842,0.00001115128,7.400587e-7,0.0003805055,0.9085708,0.02340014,0.06683898,0.00004569832,0.0002049831],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01838145,0.00001859291,0.9748995,0.002664498,0.0001504628,0.0008803745,0.000001245238,0.0001289379,0.002874913],"genre_scores_gemma":[0.9881387,0.00004410338,0.0106593,0.0001987488,0.00002032162,0.0001406273,0.000001087317,0.00001826005,0.0007788024],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9697573,"threshold_uncertainty_score":0.9980911,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04088646043850981,"score_gpt":0.3149990820066595,"score_spread":0.2741126215681498,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}