{"id":"W3177455005","doi":"10.65109/mpbl3656","title":"TDprop: Does Adaptive Optimization With Jacobi Preconditioning Help Temporal Difference Learning?","year":2021,"lang":"en","type":"article","venue":"","topic":"Domain Adaptation and Few-Shot Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; Polytechnique Montréal; McGill University","funders":"","keywords":"Computer science; Temporal difference learning; Mathematical optimization; Mathematics; Artificial intelligence; Reinforcement learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000157789,0.0001589485,0.0001599927,0.00008294557,0.0003262922,0.000403977,0.0002706579,0.00005577761,0.0005355474],"category_scores_gemma":[0.0000798231,0.0001140143,0.00004135644,0.0004876139,0.0000522825,0.000645806,0.0001279125,0.0002507911,0.00004722625],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004148672,"about_ca_system_score_gemma":0.0001679799,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002374858,"about_ca_topic_score_gemma":0.00004405744,"domain_scores_codex":[0.9985622,0.0001779509,0.000201136,0.0004891283,0.0003126978,0.00025691],"domain_scores_gemma":[0.9991133,0.0001081265,0.0001373123,0.0002614247,0.0002696036,0.0001102662],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004839889,0.0001841773,0.02485853,0.00002883755,0.00009264617,0.000107318,0.003715679,0.837781,0.0007834124,0.09118731,0.0002128894,0.04099979],"study_design_scores_gemma":[0.0007042952,0.0002469233,0.01213983,0.00007437608,0.000009840239,0.00005236507,0.001860698,0.979778,0.002644236,0.0006227049,0.001466548,0.0004001827],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.008203928,0.00003410676,0.9742685,0.0004919032,0.0001427647,0.0001069667,7.964469e-7,0.0003595982,0.01639145],"genre_scores_gemma":[0.6468836,0.000008743511,0.3389044,0.0002577869,0.00003655047,0.00001745988,0.00002298602,0.00001152662,0.01385691],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.6386797,"threshold_uncertainty_score":0.5863869,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01143651249107494,"score_gpt":0.2124903183998896,"score_spread":0.2010538059088147,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}