{"id":"W2919205453","doi":"10.48550/arxiv.1903.00194","title":"Should All Temporal Difference Learning Use Emphasis?","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Counterexample; Convergence (economics); Temporal difference learning; Computer science; Simple (philosophy); Class (philosophy); Artificial intelligence; Mathematics; Economics; Reinforcement learning; Epistemology; Discrete mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0003121516,0.0005355574,0.0005212705,0.0003620234,0.0001858131,0.0006143313,0.003243783,0.0005079582,0.00004667459],"category_scores_gemma":[0.0001506208,0.0006147572,0.000304976,0.0004810654,0.0001155004,0.0007724855,0.004624381,0.001822905,0.0004505735],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003136576,"about_ca_system_score_gemma":0.0002658573,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002686908,"about_ca_topic_score_gemma":0.00001309213,"domain_scores_codex":[0.9969279,0.0002945088,0.0003314401,0.001525118,0.0002635121,0.0006575892],"domain_scores_gemma":[0.996745,0.0003029727,0.0005371541,0.001978464,0.0001976394,0.0002387365],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001299025,0.00003085688,0.08017468,0.00005992669,0.00009845562,0.0001706763,0.0002365078,0.8996584,0.00001956815,0.01913247,0.0001569785,0.0002485012],"study_design_scores_gemma":[0.0004234254,0.0001105722,0.008725437,0.0001278541,0.00006479557,0.000004681741,0.00003539109,0.9842499,0.00003223225,0.0009804505,0.004538842,0.0007064362],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1620576,0.00002188672,0.8342632,0.00008744911,0.000920164,0.0003328431,0.00000263976,0.0005343828,0.001779855],"genre_scores_gemma":[0.9675691,0.0001503423,0.003776469,0.0002178133,0.00006279454,8.279506e-7,0.00004521078,0.00003864078,0.02813887],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8304868,"threshold_uncertainty_score":0.9996304,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1825729635131934,"score_gpt":0.2228682985839207,"score_spread":0.0402953350707273,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}