{"id":"W2919328485","doi":"10.1609/aaai.v33i01.33017949","title":"Rethinking the Discount Factor in Reinforcement Learning: A Decision Theoretic Approach","year":2019,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Decision-Making and Behavioral Economics","field":"Decision Sciences","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Discounting; Reinforcement learning; Axiom; Markov decision process; Bellman equation; Generalization; Mathematical economics; Expected utility hypothesis; Preference; Computer science; Revealed preference; Decision theory; Markov process; Mathematics; Econometrics; Economics; Artificial intelligence; Microeconomics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication","open_science","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.006432599,0.0005972717,0.000975492,0.000523577,0.0003518675,0.001714592,0.00672311,0.0004889012,0.0004295397],"category_scores_gemma":[0.007977975,0.0003154167,0.0004581963,0.0009147881,0.0007209711,0.0003111156,0.003370916,0.00259172,0.0003759669],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002696112,"about_ca_system_score_gemma":0.0003740827,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001209796,"about_ca_topic_score_gemma":0.00003330731,"domain_scores_codex":[0.9926531,0.0001355228,0.002499877,0.001495007,0.002634376,0.0005821175],"domain_scores_gemma":[0.9931783,0.001911465,0.00231861,0.001294006,0.001191022,0.0001065953],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0006774561,0.0003692552,0.003850769,0.00008876982,0.00003455109,0.00000147039,0.01800542,0.113129,0.00117561,0.3426131,0.0003478355,0.5197067],"study_design_scores_gemma":[0.00003814777,0.0001566297,0.0004345086,0.0008904648,0.00002422462,0.000003049146,0.003557683,0.1728532,0.004819327,0.8166537,0.000198118,0.0003708919],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9666739,0.00004813995,0.009638513,0.001550547,0.00175939,0.001475216,0.00001465457,0.00004449437,0.01879512],"genre_scores_gemma":[0.9981618,0.0001187091,0.0006675507,0.0001477221,0.0001043817,0.00006996493,0.000002487148,0.0000378118,0.0006895766],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5193358,"threshold_uncertainty_score":0.9999298,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2499531624091111,"score_gpt":0.3937038071337248,"score_spread":0.1437506447246137,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}