{"id":"W4409361576","doi":"10.1609/aaai.v39i25.34864","title":"ModelDiff: Symbolic Dynamic Programming for Model-Aware Policy Transfer in Deep Q-Learning","year":2025,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Transfer of learning; Transfer (computing); Dynamic programming; Artificial intelligence; Programming language; Parallel computing; Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002518711,0.001056483,0.001179917,0.0004699079,0.000431209,0.00114268,0.002161447,0.001600827,0.005674388],"category_scores_gemma":[0.008564932,0.0007339686,0.0006888112,0.0005600405,0.001508517,0.001607212,0.002842233,0.003419574,0.0007267094],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001528575,"about_ca_system_score_gemma":0.003420617,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005552784,"about_ca_topic_score_gemma":0.006738093,"domain_scores_codex":[0.9990834,0.0004035978,0.00003791987,0.0001707225,0.000207884,0.00009657807],"domain_scores_gemma":[0.9971999,0.002097462,0.0001545725,0.0002172885,0.0001925756,0.0001382761],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006206595,0.00006071868,0.0003656398,0.00008771971,0.00002555498,0.0000470875,0.00006400926,0.9057125,0.0004859059,0.04026559,0.002258234,0.0505649],"study_design_scores_gemma":[0.000007958289,0.00000891213,0.000008847668,0.000004136159,0.000001365789,0.000002905911,0.000002119327,0.9878902,0.0001000117,0.01168279,0.0002892609,0.000001549884],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0039194,0.0001670061,0.9931344,0.0002643684,0.00003757954,0.0000430166,0.00004901385,0.0006564404,0.001728712],"genre_scores_gemma":[0.5557053,0.0003467247,0.4374454,0.0006343871,0.00008505116,0.0005395773,0.0003288749,0.0004738699,0.004440785],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005674388,"threshold_uncertainty_score":0.01898271,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04088646043850981,"score_gpt":0.3149990820066595,"score_spread":0.2741126215681498,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}