{"id":"W2919328485","doi":"10.1609/aaai.v33i01.33017949","title":"Rethinking the Discount Factor in Reinforcement Learning: A Decision Theoretic Approach","year":2019,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Decision-Making and Behavioral Economics","field":"Decision Sciences","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Discounting; Reinforcement learning; Axiom; Markov decision process; Bellman equation; Generalization; Mathematical economics; Expected utility hypothesis; Preference; Computer science; Revealed preference; Decision theory; Markov process; Mathematics; Econometrics; Economics; Artificial intelligence; Microeconomics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01115697,0.001956027,0.003186549,0.001069675,0.0007921926,0.003498433,0.004348249,0.003772481,0.002464084],"category_scores_gemma":[0.0376957,0.001584779,0.001613217,0.001211057,0.005604636,0.009714984,0.003060407,0.009833383,0.0002797801],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003053649,"about_ca_system_score_gemma":0.00201869,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003468674,"about_ca_topic_score_gemma":0.002256025,"domain_scores_codex":[0.9951093,0.002751441,0.0002538131,0.0008799735,0.0007334095,0.0002719653],"domain_scores_gemma":[0.9738939,0.02239178,0.001128437,0.001014955,0.0009094693,0.0006613805],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001481283,0.000124453,0.0008345025,0.0002106537,0.0001340749,0.0001035989,0.0002043789,0.4729594,0.0009981132,0.4969577,0.0008132386,0.02651171],"study_design_scores_gemma":[0.00003155358,0.00003822271,0.00007470744,0.00003295978,0.00002813983,0.00001708524,0.00001263123,0.7198831,0.0002044234,0.2790986,0.0005517537,0.00002677771],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.008403671,0.001072011,0.9872093,0.001153113,0.0001199582,0.00002778087,0.00002535755,0.00005543522,0.001933334],"genre_scores_gemma":[0.7577699,0.00208716,0.2361263,0.0005846406,0.0005686282,0.0001596028,0.00004996234,0.0001660682,0.00248781],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01115697,"threshold_uncertainty_score":0.05900443,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2499531624091111,"score_gpt":0.3937038071337248,"score_spread":0.1437506447246137,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}