{"id":"W2956872166","doi":"10.48550/arxiv.1907.04651","title":"Incrementally Learning Functions of the Return","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Taylor series; Value (mathematics); Computer science; Temporal difference learning; Mathematics; Mathematical optimization; Applied mathematics; Artificial intelligence; Machine learning; Mathematical analysis","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002809405,0.000210285,0.0002336223,0.0001464058,0.000186356,0.00007541815,0.00240605,0.0001966241,0.0000477025],"category_scores_gemma":[0.00007720217,0.0001957238,0.0002454421,0.0005295879,0.0001090388,0.0002277512,0.004175777,0.001010509,0.0001048826],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001884157,"about_ca_system_score_gemma":0.0002366138,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000839241,"about_ca_topic_score_gemma":0.000007615534,"domain_scores_codex":[0.9985834,0.0002053071,0.0002180717,0.0005819135,0.0001667691,0.0002445487],"domain_scores_gemma":[0.9977218,0.0001070326,0.0005104981,0.001450766,0.0001534413,0.00005647926],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000005165272,0.00001361189,0.02963483,0.00004694832,0.00006387148,0.000005487539,0.00014945,0.9538195,0.0000806788,0.01590765,0.0002241843,0.00004867213],"study_design_scores_gemma":[0.0002495816,0.00005825894,0.004702851,0.0001242959,0.00005700034,0.00000182962,0.000104573,0.9918023,0.0001608217,0.0007638798,0.001741275,0.0002333565],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.08603965,0.00001511402,0.898011,0.0000811845,0.001151744,0.0002747789,0.000001927288,0.0001252163,0.01429937],"genre_scores_gemma":[0.9780043,0.00003210871,0.0006285842,0.0000526845,0.00003127049,3.115136e-7,0.000006830002,0.00001332737,0.02123057],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8973824,"threshold_uncertainty_score":0.7981385,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05402498703515711,"score_gpt":0.1752706064785677,"score_spread":0.1212456194434106,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}