{"id":"W2998494185","doi":"10.1609/aaai.v34i04.6027","title":"Gamma-Nets: Generalizing Value Estimation over Timescale","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates; Alberta Machine Intelligence Institute; Compute Canada","keywords":"Reinforcement learning; Computer science; Estimator; Bellman equation; Abstraction; Artificial intelligence; Scalability; Function (biology); Set (abstract data type); Temporal difference learning; Machine learning; Value (mathematics); Representation (politics); Key (lock); Markov decision process; Mathematical optimization; Markov process; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002149136,0.001582956,0.001111295,0.0007964516,0.0004615631,0.001266025,0.00246972,0.001593516,0.00231283],"category_scores_gemma":[0.007778883,0.0008605263,0.001012829,0.0006057021,0.001452598,0.003473045,0.002050091,0.003389277,0.000483571],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002160993,"about_ca_system_score_gemma":0.001630158,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01355181,"about_ca_topic_score_gemma":0.01211795,"domain_scores_codex":[0.9993997,0.0001574919,0.00003905346,0.0001865618,0.0001350877,0.00008211402],"domain_scores_gemma":[0.9977511,0.001372508,0.0002668506,0.0002253468,0.0002533545,0.0001308166],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00008299572,0.00003008702,0.001264412,0.00004460598,0.00003588576,0.00005779808,0.00006400648,0.9384231,0.0009370685,0.01291735,0.001148983,0.04499379],"study_design_scores_gemma":[0.000005053399,0.00001219726,0.00005937641,0.00000694202,0.000004220977,0.000006215806,0.000004429533,0.9863127,0.0002945358,0.01302275,0.0002677021,0.000004013421],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02268905,0.0003960851,0.9733612,0.000401876,0.00007214749,0.00004923847,0.0001756679,0.0012044,0.001650322],"genre_scores_gemma":[0.749095,0.0006920793,0.2440497,0.0005496042,0.0001278774,0.000249509,0.000620691,0.0003590758,0.004256541],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01355181,"threshold_uncertainty_score":0.02694583,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02043068963763373,"score_gpt":0.2437083033784362,"score_spread":0.2232776137408025,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}