{"id":"W3177455005","doi":"10.65109/mpbl3656","title":"TDprop: Does Adaptive Optimization With Jacobi Preconditioning Help Temporal Difference Learning?","year":2021,"lang":"en","type":"article","venue":"","topic":"Domain Adaptation and Few-Shot Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; Polytechnique Montréal; McGill University","funders":"","keywords":"Computer science; Temporal difference learning; Mathematical optimization; Mathematics; Artificial intelligence; Reinforcement learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002608096,0.0008795731,0.0009562087,0.000266656,0.0003332359,0.0009809827,0.001330108,0.001520536,0.005212695],"category_scores_gemma":[0.01380928,0.0003417041,0.000449648,0.0002856257,0.001267508,0.002285116,0.001247073,0.002231171,0.001082472],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003966083,"about_ca_system_score_gemma":0.001308279,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002098101,"about_ca_topic_score_gemma":0.002484476,"domain_scores_codex":[0.9994552,0.0002249829,0.00003483575,0.0001356987,0.00008677471,0.00006268976],"domain_scores_gemma":[0.9965329,0.002297137,0.0001901163,0.000503483,0.0003219155,0.0001544267],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007745592,0.0003696827,0.004928076,0.0005589488,0.0001783562,0.0002904414,0.0002832508,0.6316842,0.01456859,0.07326853,0.00964439,0.263451],"study_design_scores_gemma":[0.00003571557,0.0000957349,0.0001808518,0.00002403856,0.00001047803,0.00003983683,0.00001603384,0.9858143,0.002516008,0.01029777,0.0009606484,0.000008599427],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.06443274,0.0007284122,0.9260791,0.001326991,0.0002306522,0.00007451834,0.00007775175,0.001598793,0.00545113],"genre_scores_gemma":[0.6648267,0.0003581499,0.3299814,0.0006922299,0.0001075591,0.0001495974,0.0001632622,0.0004271337,0.003294011],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005212695,"threshold_uncertainty_score":0.01743817,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01143651249107494,"score_gpt":0.2124903183998896,"score_spread":0.2010538059088147,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}