{"id":"W2162664081","doi":"","title":"Dual Temporal Difference Learning","year":2009,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Dual (grammatical number); Convergence (economics); Temporal difference learning; Reinforcement learning; Computer science; Artificial intelligence; Dynamic programming; Basis (linear algebra); Machine learning; Algorithm; Mathematical optimization; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000136605,0.00009773458,0.00009444063,0.0000580934,0.0001138247,0.0001879417,0.0004691283,0.0000373361,0.00004696804],"category_scores_gemma":[0.00005880922,0.00008263969,0.00003478614,0.0002047797,0.00001647256,0.0002549699,0.0001103145,0.0002042291,0.0001998483],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001913664,"about_ca_system_score_gemma":0.00002718712,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000007338937,"about_ca_topic_score_gemma":4.577239e-7,"domain_scores_codex":[0.9990814,0.00003682174,0.0001523434,0.0002170137,0.0002628483,0.0002495688],"domain_scores_gemma":[0.9994984,0.00004171703,0.00005638745,0.0002969013,0.00003454046,0.00007204224],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000004824521,0.00005419633,0.02324273,0.000006064424,0.00001322859,0.00005489493,0.0009723413,0.4279094,0.002227339,0.4007896,0.001372998,0.1433524],"study_design_scores_gemma":[0.0002258755,0.0003877364,0.0491012,0.00001038836,0.000001945253,0.00001703208,0.00002245829,0.9418923,0.0006243736,0.001375886,0.006099777,0.0002410244],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.00809576,0.000009566525,0.9513201,0.0009417676,0.0001218982,0.00004815696,1.590444e-8,0.0004497983,0.03901294],"genre_scores_gemma":[0.9310511,0.000003156445,0.05089851,0.0005352856,0.00004102203,7.803336e-7,0.000001218847,0.000002963338,0.01746598],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9229553,"threshold_uncertainty_score":0.3369949,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01651114802915592,"score_gpt":0.2444750998830643,"score_spread":0.2279639518539084,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}