{"id":"W3035547404","doi":"10.48550/arxiv.2007.00611","title":"Gradient Temporal-Difference Learning with Regularized Corrections","year":2020,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Neural Networks and Applications","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Temporal difference learning; Divergence (linguistics); Computer science; Artificial neural network; Soundness; Range (aeronautics); Stability (learning theory); Artificial intelligence; Algorithm; Reinforcement learning; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00003313498,0.0001021513,0.0001044719,0.00003418899,0.0002715565,0.00005786693,0.0004842704,0.00003013674,0.0000123582],"category_scores_gemma":[0.000007756636,0.00009505499,0.00004398713,0.0009766528,0.0000549178,0.0002017641,0.0001357639,0.0001927999,0.00004906083],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002352349,"about_ca_system_score_gemma":0.00002999391,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003352256,"about_ca_topic_score_gemma":0.00001809518,"domain_scores_codex":[0.9992184,0.00004462857,0.00007020048,0.0004429394,0.00004633274,0.0001775001],"domain_scores_gemma":[0.9993996,0.00004258426,0.00006980148,0.0002671741,0.00005223152,0.0001686046],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00007417904,0.0001399436,0.05388063,0.00001501522,0.00005860651,0.0002179312,0.0007385866,0.2476888,0.001822732,0.6898805,0.0008846008,0.004598529],"study_design_scores_gemma":[0.000446373,0.0001808554,0.004004054,0.00001185205,0.00001632312,0.000005760691,0.00008919332,0.9878646,0.0001959908,0.002081557,0.004889928,0.0002135325],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2936426,0.000004728943,0.7044258,0.0007701237,0.00003794107,0.00009542093,5.517032e-7,0.0002386649,0.0007841148],"genre_scores_gemma":[0.9959508,0.00001186565,0.002098559,0.000234229,0.0000310882,9.228883e-7,0.00000275908,0.000005910469,0.001663831],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7401758,"threshold_uncertainty_score":0.387623,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05664012624384622,"score_gpt":0.1580861226124704,"score_spread":0.1014459963686242,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}