{"id":"W2904489222","doi":"10.48550/arxiv.1812.06600","title":"Double Deep Q-Learning for Optimal Execution","year":2018,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Stock Market Forecasting Methods","field":"Decision Sciences","cited_by":22,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Benchmark (surveying); Computer science; Limit (mathematics); Function (biology); Bellman equation; Artificial neural network; Value (mathematics); Control (management); Action (physics); State (computer science); Q-learning; Optimal control; Mathematical optimization; Artificial intelligence; Machine learning; Algorithm; Reinforcement learning; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003055418,0.001092783,0.00159755,0.0006097377,0.0003932716,0.001092408,0.0017176,0.001544957,0.003921266],"category_scores_gemma":[0.008105365,0.0006004869,0.0005086198,0.0007668351,0.001359896,0.002041676,0.001483613,0.002792788,0.0004830838],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001866878,"about_ca_system_score_gemma":0.002437507,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006823341,"about_ca_topic_score_gemma":0.005798646,"domain_scores_codex":[0.9991863,0.000257202,0.00005359107,0.0002120654,0.0001604725,0.000130348],"domain_scores_gemma":[0.9969427,0.002076071,0.0002325498,0.000239778,0.0003568047,0.0001520811],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001088679,0.00008467034,0.001167678,0.00008414509,0.00004197834,0.00004931153,0.0000481645,0.9113088,0.0005922863,0.02976665,0.001943718,0.0548037],"study_design_scores_gemma":[0.000004655376,0.00001115366,0.0000492539,0.0000046205,0.000002153896,0.000003460466,0.000002186759,0.988177,0.0001293981,0.01147609,0.0001378037,0.000002185802],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02189285,0.000714551,0.9739973,0.0007105618,0.00005974652,0.00004000075,0.0001060143,0.0005442896,0.001934773],"genre_scores_gemma":[0.8492601,0.0005314576,0.1436816,0.0004321956,0.0001083467,0.0001781179,0.0004391441,0.0001616424,0.005207393],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006823341,"threshold_uncertainty_score":0.01615882,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2709709277858849,"score_gpt":0.3182128170699396,"score_spread":0.04724188928405476,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}