{"id":"W4283815468","doi":"10.1609/aaai.v36i6.20639","title":"A Generalized Bootstrap Target for Value-Learning, Efficiently Combining Value and Feature Predictions","year":2022,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian Institute for Advanced Research; McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"Fonds de recherche du Québec – Nature et technologies; Compute Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Bootstrapping (finance); Computer science; Reinforcement learning; Successor cardinal; Value (mathematics); Artificial intelligence; Bellman equation; Machine learning; Function (biology); Generality; Feature (linguistics); Mathematics; Econometrics; Mathematical optimization","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009146431,0.0002521251,0.0002905852,0.0001830579,0.001179595,0.0003441616,0.001710601,0.00007672748,0.00004595267],"category_scores_gemma":[0.0005252184,0.0002198887,0.0001473528,0.0006775961,0.0002247543,0.0002635276,0.000870619,0.0006968147,0.000005674863],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008237784,"about_ca_system_score_gemma":0.0001300527,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001879438,"about_ca_topic_score_gemma":4.242049e-7,"domain_scores_codex":[0.997723,0.00005865489,0.0004774758,0.0005831948,0.0007113742,0.0004462952],"domain_scores_gemma":[0.9985419,0.0001864028,0.0005180336,0.0002720677,0.0003758757,0.0001057055],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000525227,0.00006404671,0.000200475,0.00003135133,0.00002364291,2.350604e-7,0.001552576,0.3044171,0.004970009,0.6866579,0.0004505952,0.001579521],"study_design_scores_gemma":[0.0001054369,0.0009194465,0.00008183095,0.00006581956,0.00002518539,0.00001100626,0.0008084377,0.9401678,0.02449054,0.03046285,0.002623162,0.0002384145],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07951201,0.00009163772,0.9028898,0.007405485,0.001503629,0.001700116,0.00002014988,0.0003675852,0.006509594],"genre_scores_gemma":[0.9789051,0.00001954805,0.01902091,0.0002589639,0.0000571233,0.0001505719,0.000003538459,0.00002220862,0.001562001],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8993931,"threshold_uncertainty_score":0.9072612,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05540860206408431,"score_gpt":0.2917252413087566,"score_spread":0.2363166392446723,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}