{"id":"W3158681970","doi":"10.48550/arxiv.2104.13844","title":"A Generalized Projected Bellman Error for Off-policy Value Estimation in Reinforcement Learning","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Bellman equation; Temporal difference learning; Function approximation; Hyperparameter; Nonlinear system; Mean squared error; Artificial neural network; Mathematical optimization; Function (biology); Computer science; Approximation error; Linear approximation; Value (mathematics); Mathematics; Applied mathematics; Algorithm; Artificial intelligence; Machine learning; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006833239,0.001717434,0.001616959,0.0008523189,0.0005568887,0.001811417,0.002148996,0.002333392,0.003183255],"category_scores_gemma":[0.02812581,0.0006957107,0.00091637,0.0008781934,0.002995707,0.003607173,0.003246524,0.004142316,0.0005816563],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002124152,"about_ca_system_score_gemma":0.003256649,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003002718,"about_ca_topic_score_gemma":0.002302143,"domain_scores_codex":[0.9963097,0.001640526,0.0002166135,0.0006845507,0.0009374554,0.0002111131],"domain_scores_gemma":[0.9901482,0.006802515,0.0006648773,0.0008706296,0.001214063,0.0002996822],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001120415,0.000056263,0.0007847711,0.0001732573,0.00007184216,0.00005530126,0.0001005851,0.8121724,0.001838457,0.1254773,0.001586748,0.05757104],"study_design_scores_gemma":[0.000009228474,0.00003664613,0.00008907732,0.00002391589,0.000006428827,0.00001520158,0.000004754073,0.9718871,0.0006453589,0.02675757,0.0005138142,0.00001092927],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.002513115,0.0001566298,0.9962668,0.0001427007,0.00004171014,0.00002697293,0.00002012732,0.0001128708,0.0007191888],"genre_scores_gemma":[0.3432379,0.0005551323,0.6506271,0.0004334027,0.000181097,0.0004803901,0.0002392494,0.000366986,0.003878567],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006833239,"threshold_uncertainty_score":0.03613806,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07410652224032353,"score_gpt":0.2396721545053905,"score_spread":0.165565632265067,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}