{"id":"W2952014622","doi":"10.48550/arxiv.1503.04269","title":"An Emphatic Approach to the Problem of Off-policy Temporal-Difference Learning","year":2015,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Temporal difference learning; Discounting; Lambda; Bootstrapping (finance); Computer science; Computation; Parametric statistics; Function (biology); Algorithm; State (computer science); Applied mathematics; Artificial intelligence; Mathematics; Reinforcement learning; Statistics; Econometrics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005564186,0.0008741574,0.001134607,0.0004000817,0.0006063221,0.001583305,0.003338916,0.002586208,0.003882853],"category_scores_gemma":[0.02485414,0.0007724694,0.0005578867,0.0005578796,0.003247886,0.00385627,0.002865991,0.006619625,0.0005176471],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001282478,"about_ca_system_score_gemma":0.001545996,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001616052,"about_ca_topic_score_gemma":0.00106888,"domain_scores_codex":[0.9978962,0.0009707258,0.00009491129,0.0004771211,0.0004640681,0.00009694102],"domain_scores_gemma":[0.9926998,0.005169953,0.000383419,0.001025695,0.0005246325,0.0001965937],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0003754057,0.0001616177,0.000929482,0.0003038282,0.0001035755,0.0001500092,0.0002742917,0.3759036,0.004327424,0.4715724,0.004360349,0.1415381],"study_design_scores_gemma":[0.00002983414,0.00008490399,0.00009261812,0.00002399805,0.00001174981,0.00006982867,0.00001420358,0.8765548,0.001466584,0.1186778,0.002958217,0.00001539275],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.002592172,0.0001685284,0.994605,0.000822049,0.00008605247,0.00002099838,0.00001668541,0.0001104008,0.00157803],"genre_scores_gemma":[0.5215029,0.0005337363,0.468243,0.001358613,0.0004709783,0.0002253387,0.00008718183,0.0002297397,0.007348503],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005564186,"threshold_uncertainty_score":0.02942657,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09377041807705082,"score_gpt":0.2191130987965938,"score_spread":0.125342680719543,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}