{"id":"W3025133396","doi":"10.65109/uazl2918","title":"META-Learning State-based Eligibility Traces for More Sample-Efficient Policy Evaluation","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Machine learning; Robustness (evolution); Artificial intelligence; Temporal difference learning; Sample (material); Q-learning; Bellman equation; Mathematical optimization; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003460923,0.001191414,0.001843841,0.0008053615,0.0003894392,0.001545669,0.00230315,0.001596804,0.003335312],"category_scores_gemma":[0.0192164,0.0007182902,0.0006225888,0.0005710649,0.001263913,0.002603257,0.001902181,0.003144081,0.0006136827],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001308436,"about_ca_system_score_gemma":0.002784216,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002756353,"about_ca_topic_score_gemma":0.002884257,"domain_scores_codex":[0.9989736,0.0003579488,0.00007713618,0.0002174509,0.0002627182,0.0001111457],"domain_scores_gemma":[0.9923741,0.00538676,0.0005475877,0.0007456652,0.000650155,0.000295628],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002331468,0.0002585044,0.001342223,0.000129212,0.00005684451,0.00006470696,0.0001250356,0.8550237,0.003316554,0.018782,0.001314606,0.1193535],"study_design_scores_gemma":[0.00001408505,0.00002840831,0.00004806429,0.000008861103,0.000004284553,0.000008353875,0.000004042738,0.9942358,0.0007361074,0.004728129,0.0001798401,0.000003892285],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01601114,0.0001404292,0.9818901,0.0001980133,0.00003385989,0.00006458376,0.00003263368,0.0007923952,0.0008368939],"genre_scores_gemma":[0.7607452,0.0001233999,0.2365841,0.0002472543,0.00005642911,0.0002988529,0.0001486265,0.000253758,0.001542346],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003460923,"threshold_uncertainty_score":0.01830333,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1683899976315094,"score_gpt":0.3973292214127686,"score_spread":0.2289392237812592,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}