{"id":"W3012922890","doi":"10.1007/s10614-021-10119-4","title":"Reinforcement Learning in Economics and Finance","year":2021,"lang":"en","type":"preprint","venue":"Computational Economics","topic":"Advanced Bandit Algorithms Research","field":"Decision Sciences","cited_by":30,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université du Québec à Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; Centre National de la Recherche Scientifique; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada; AXA Research Fund","keywords":"Reinforcement learning; Action (physics); Computer science; Set (abstract data type); Artificial intelligence; Time horizon; Q-learning; Behavioral economics; Order (exchange); Reinforcement; Process (computing); Term (time); Temporal difference learning; Economics; Microeconomics; Psychology; Finance; Social psychology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002257501,0.0009146516,0.001415114,0.001108617,0.0005877785,0.003401731,0.0009371436,0.002649575,0.006425094],"category_scores_gemma":[0.0150124,0.0005567176,0.0004838814,0.002041675,0.002331882,0.004218115,0.001205385,0.00388472,0.0006554861],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002353121,"about_ca_system_score_gemma":0.001461619,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004296629,"about_ca_topic_score_gemma":0.002037437,"domain_scores_codex":[0.9989754,0.000674414,0.00002974766,0.00009446757,0.0001746251,0.00005138469],"domain_scores_gemma":[0.9926292,0.006344863,0.0002292497,0.0002298429,0.0003840298,0.0001829567],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003007976,0.00005038781,0.0004041971,0.0001628656,0.00003879617,0.00002692106,0.00005564528,0.02925777,0.0001527088,0.9350365,0.005855975,0.0289281],"study_design_scores_gemma":[0.0000144937,0.000005002183,0.0001548376,0.00002445683,0.000005568632,0.000008293582,0.00001455749,0.1022112,0.00006551688,0.8942791,0.003212111,0.000004923856],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0373807,0.04771404,0.8397238,0.03443277,0.001700306,0.0000632796,0.0002719231,0.0003319548,0.03838134],"genre_scores_gemma":[0.7888808,0.02758426,0.140045,0.001739656,0.004592985,0.0002552595,0.0003296433,0.0002075534,0.03636479],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006425094,"threshold_uncertainty_score":0.02149403,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08548769842633776,"score_gpt":0.3792800501201351,"score_spread":0.2937923516937974,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}