{"id":"W2966362327","doi":"10.1609/aaai.v33i01.33017949","title":"Rethinking the Discount Factor in Reinforcement Learning: A Decision Theoretic Approach","year":2019,"lang":"en","type":"article","venue":"","topic":"Supply Chain and Inventory Management","field":"Business, Management and Accounting","cited_by":23,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Factor (programming language); Reinforcement; Discounting; Economics; Artificial intelligence; Computer science; Econometrics; Psychology; Social psychology; Finance; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008797501,0.001598715,0.002538679,0.0008283771,0.0006574328,0.002620126,0.003582007,0.002991539,0.002353393],"category_scores_gemma":[0.02749622,0.00130031,0.001181401,0.0008909456,0.004300084,0.006797078,0.002624865,0.007421847,0.0002320986],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002598875,"about_ca_system_score_gemma":0.001967836,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004496836,"about_ca_topic_score_gemma":0.002881313,"domain_scores_codex":[0.9963394,0.002109503,0.0001887752,0.000590644,0.0005468887,0.0002248842],"domain_scores_gemma":[0.979007,0.01809642,0.0009004779,0.0007067703,0.0008039965,0.0004852575],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001143016,0.0001076094,0.000654691,0.000148202,0.00009673126,0.0000765876,0.0001493732,0.771646,0.0007808144,0.1990732,0.0006269098,0.02652552],"study_design_scores_gemma":[0.00002041622,0.00002954773,0.0000448559,0.00001873333,0.00001661382,0.000009356527,0.000007812512,0.9209348,0.0001478753,0.07844756,0.000307456,0.00001494499],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.007800086,0.000638123,0.9890164,0.0007809899,0.00007779129,0.00002585426,0.00001805332,0.00006250045,0.001580235],"genre_scores_gemma":[0.7992656,0.001171569,0.1964334,0.0004275239,0.0003235469,0.000123267,0.00003607935,0.0001278995,0.002091229],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008797501,"threshold_uncertainty_score":0.04652619,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02158486183632454,"score_gpt":0.222876826992513,"score_spread":0.2012919651561885,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}