{"id":"W2269274350","doi":"10.48550/arxiv.1512.04087","title":"True Online Temporal-Difference Learning","year":2015,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":56,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Temporal difference learning; Computer science; Artificial intelligence; Equivalence (formal languages); Lambda; Reinforcement learning; Domain (mathematical analysis); Online learning; Machine learning; Binary number; Online algorithm; Simple (philosophy); Algorithm; Theoretical computer science; Mathematics; Discrete mathematics; Arithmetic; Mathematical analysis","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003887258,0.001167707,0.001448091,0.0005504508,0.00042385,0.001335588,0.003230354,0.002011353,0.006732861],"category_scores_gemma":[0.01378052,0.0004688712,0.0009079475,0.0004866713,0.001307498,0.003344719,0.002351512,0.002951929,0.001532895],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001131021,"about_ca_system_score_gemma":0.002153141,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002834871,"about_ca_topic_score_gemma":0.003175037,"domain_scores_codex":[0.9980962,0.0006262008,0.0001264132,0.0004974932,0.0004713626,0.0001824226],"domain_scores_gemma":[0.9923006,0.004441848,0.0004126986,0.001482735,0.001017002,0.0003450948],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001030013,0.0005355137,0.002934888,0.0004171749,0.0001543993,0.0001239427,0.0001354297,0.5088012,0.00391915,0.03398545,0.009626715,0.4383361],"study_design_scores_gemma":[0.00004174204,0.00009924239,0.0001423929,0.00001075209,0.000009351058,0.00003375515,0.0000105182,0.9891303,0.001222711,0.008626433,0.000663515,0.00000929726],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0371146,0.0007083499,0.9532283,0.0005286254,0.0002887548,0.0001110064,0.000260054,0.00286384,0.004896434],"genre_scores_gemma":[0.649852,0.0002438827,0.3415572,0.0006751075,0.0001350021,0.0002608075,0.0006469206,0.0003691971,0.006259927],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006732861,"threshold_uncertainty_score":0.02252364,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1219179917530586,"score_gpt":0.2099222055253362,"score_spread":0.08800421377227757,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}