{"id":"W3035547404","doi":"10.48550/arxiv.2007.00611","title":"Gradient Temporal-Difference Learning with Regularized Corrections","year":2020,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Neural Networks and Applications","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Temporal difference learning; Divergence (linguistics); Computer science; Artificial neural network; Soundness; Range (aeronautics); Stability (learning theory); Artificial intelligence; Algorithm; Reinforcement learning; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003162106,0.001019401,0.001316304,0.0005456723,0.0003979433,0.001059358,0.002100758,0.001674194,0.003808013],"category_scores_gemma":[0.01782162,0.000502021,0.0007118447,0.0005943474,0.00127772,0.001985349,0.001625065,0.002574092,0.00105576],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008943178,"about_ca_system_score_gemma":0.001953436,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003900392,"about_ca_topic_score_gemma":0.002974604,"domain_scores_codex":[0.9988905,0.000461765,0.00007002006,0.0002362422,0.000259084,0.00008245189],"domain_scores_gemma":[0.9945915,0.003501543,0.0003266783,0.0005793645,0.0008199551,0.0001809544],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004452176,0.0001897435,0.001581687,0.0002588298,0.000104219,0.0001317286,0.0001738977,0.7022942,0.004293078,0.05693123,0.006833164,0.226763],"study_design_scores_gemma":[0.00001963547,0.00003534085,0.0000417169,0.00000709846,0.000004172954,0.00001522391,0.000003843055,0.9912402,0.0005502998,0.007476687,0.000600667,0.000005163211],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.006851744,0.0002332528,0.9906347,0.0002371894,0.000111898,0.00004314023,0.00002912574,0.0007283436,0.001130414],"genre_scores_gemma":[0.4289167,0.0002349201,0.5629007,0.0005316422,0.0001755884,0.0002549378,0.0002467735,0.0005404054,0.006198367],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003900392,"threshold_uncertainty_score":0.01672298,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05664012624384622,"score_gpt":0.1580861226124704,"score_spread":0.1014459963686242,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}