{"id":"W4283815468","doi":"10.1609/aaai.v36i6.20639","title":"A Generalized Bootstrap Target for Value-Learning, Efficiently Combining Value and Feature Predictions","year":2022,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian Institute for Advanced Research; McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"Fonds de recherche du Québec – Nature et technologies; Compute Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Bootstrapping (finance); Computer science; Reinforcement learning; Successor cardinal; Value (mathematics); Artificial intelligence; Bellman equation; Machine learning; Function (biology); Generality; Feature (linguistics); Mathematics; Econometrics; Mathematical optimization","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005188135,0.001327744,0.001748307,0.0008848289,0.0005041694,0.001897405,0.003460839,0.00183394,0.003488737],"category_scores_gemma":[0.02663818,0.000851788,0.0009731524,0.0009636231,0.001751434,0.004546142,0.003799612,0.003208555,0.001096404],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001340304,"about_ca_system_score_gemma":0.001623673,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002145907,"about_ca_topic_score_gemma":0.002051418,"domain_scores_codex":[0.9977884,0.000834048,0.0001036278,0.0004561993,0.0006664688,0.0001512652],"domain_scores_gemma":[0.9923027,0.004566452,0.0005706104,0.001340942,0.000944791,0.0002745218],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003639812,0.0002431622,0.002891969,0.0001657148,0.000150352,0.0001349529,0.0002255839,0.682734,0.006324376,0.08194112,0.002970049,0.2218547],"study_design_scores_gemma":[0.000007857966,0.00004073249,0.0001017878,0.0000108577,0.000008485288,0.00002121778,0.000005581631,0.981124,0.001351357,0.01680569,0.0005132562,0.000009210896],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.005518645,0.00005486175,0.9932458,0.00008843128,0.00001538292,0.00003067444,0.00002513913,0.000438055,0.000583039],"genre_scores_gemma":[0.4816167,0.0001284377,0.5148178,0.0002126672,0.00008660222,0.0003494787,0.0002596717,0.0003609407,0.002167726],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005188135,"threshold_uncertainty_score":0.02743781,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05540860206408431,"score_gpt":0.2917252413087566,"score_spread":0.2363166392446723,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}