{"id":"W3184471030","doi":"","title":"Deep Reinforcement Learning for Optimal Stopping with Application in Financial Engineering","year":2021,"lang":"en","type":"article","venue":"Les Cahiers du GERAD","topic":"Advanced Bandit Algorithms Research","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"HEC Montréal","funders":"","keywords":"Optimal stopping; Reinforcement learning; Computer science; Benchmark (surveying); Categorical variable; Stochastic control; Deep learning; Artificial intelligence; Mathematical optimization; Machine learning; Optimal control; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002791302,0.001218587,0.001644015,0.0008261293,0.0004271421,0.001326459,0.001142997,0.001702018,0.003221501],"category_scores_gemma":[0.01457691,0.0005364098,0.0006409684,0.001073084,0.001435846,0.001343816,0.001388786,0.003296359,0.0004279334],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001645212,"about_ca_system_score_gemma":0.001700824,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006102272,"about_ca_topic_score_gemma":0.004198622,"domain_scores_codex":[0.9992223,0.0003436784,0.00005299546,0.0001425038,0.0001697615,0.00006877101],"domain_scores_gemma":[0.9936326,0.005217037,0.0002971801,0.0001808272,0.0004924289,0.0001799413],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006981685,0.00007966509,0.001315588,0.0002191454,0.00006820371,0.0000593289,0.00006559618,0.886597,0.0006262388,0.05738948,0.001354128,0.05215596],"study_design_scores_gemma":[0.000008325746,0.00001636262,0.00006669595,0.00001597713,0.000004672156,0.000005178525,0.000003332994,0.9806224,0.0001486264,0.01868362,0.0004204818,0.000004220136],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0158812,0.004945555,0.9736785,0.001283521,0.0001452911,0.00003872781,0.00007593391,0.0004042446,0.003547144],"genre_scores_gemma":[0.7911559,0.004288428,0.1980464,0.0005097839,0.0002621724,0.0002497173,0.0002571805,0.0001604631,0.005069954],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006102272,"threshold_uncertainty_score":0.01476198,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02369073774361516,"score_gpt":0.3158420411063364,"score_spread":0.2921513033627213,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}