{"id":"W3037828233","doi":"10.48550/arxiv.1904.11439","title":"META-Learning State-based Eligibility Traces for More Sample-Efficient\\n Policy Evaluation","year":2019,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Robustness (evolution); Machine learning; Temporal difference learning; Artificial intelligence; Q-learning; Sample (material); TRACE (psycholinguistics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003179064,0.001523064,0.002274198,0.001159849,0.0005292966,0.001878216,0.003209003,0.002065467,0.005788213],"category_scores_gemma":[0.01783272,0.0009075089,0.000827544,0.0008020511,0.001189213,0.003363486,0.002150273,0.003620896,0.001022719],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001668147,"about_ca_system_score_gemma":0.003383079,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005145632,"about_ca_topic_score_gemma":0.006960625,"domain_scores_codex":[0.998948,0.0003250362,0.00007851329,0.0002760575,0.0002263764,0.0001459861],"domain_scores_gemma":[0.993468,0.004583108,0.000392959,0.0006311656,0.000622171,0.0003027088],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003268439,0.0003130403,0.002181379,0.000161565,0.00007700306,0.00008233662,0.000129925,0.814072,0.002659116,0.01493612,0.002590417,0.1624703],"study_design_scores_gemma":[0.00001173625,0.00002171894,0.00005475373,0.000009698631,0.000004832868,0.000006154668,0.000005265411,0.9953312,0.0005918614,0.003766715,0.0001916694,0.000004360677],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02544776,0.0002403344,0.9704714,0.0003738131,0.00007159505,0.00009414996,0.0001096699,0.001809487,0.001381871],"genre_scores_gemma":[0.700618,0.0001994782,0.2943609,0.0004473236,0.000103055,0.0004049046,0.0005419917,0.000543242,0.002781037],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005788213,"threshold_uncertainty_score":0.01936346,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2079463326522307,"score_gpt":0.2850733927490826,"score_spread":0.07712706009685186,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}