{"id":"W4220814865","doi":"10.1080/14697688.2022.2049356","title":"The reinforcement learning Kelly strategy","year":2022,"lang":"en","type":"article","venue":"Quantitative Finance","topic":"Advanced Bandit Algorithms Research","field":"Decision Sciences","cited_by":14,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Portfolio; Reinforcement; Computer science; Artificial intelligence; Face (sociological concept); Mathematical optimization; Operations research; Economics; Mathematics; Psychology; Sociology; Finance; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002164967,0.0008325272,0.0009389302,0.0004524377,0.0003962085,0.001349073,0.001564338,0.001232813,0.005243911],"category_scores_gemma":[0.009184971,0.0002663108,0.0003857478,0.0003701381,0.001154047,0.00155385,0.001095913,0.001111106,0.001066714],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009313258,"about_ca_system_score_gemma":0.001478456,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002996549,"about_ca_topic_score_gemma":0.001612642,"domain_scores_codex":[0.9987471,0.0005877966,0.00005730007,0.0001915166,0.000292545,0.0001237373],"domain_scores_gemma":[0.9978034,0.001235553,0.0002512562,0.000228726,0.0003571799,0.0001238739],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002724645,0.000133684,0.001333928,0.0001803773,0.0001153338,0.0002155825,0.0001478994,0.496873,0.002940608,0.2700049,0.007182494,0.2205999],"study_design_scores_gemma":[0.00004585692,0.0001117838,0.0002292921,0.00003140599,0.00002233196,0.00009449264,0.00002149238,0.9303976,0.0009427312,0.06393496,0.004139716,0.00002834717],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01765895,0.000385918,0.966455,0.0006345033,0.0000868409,0.00009116067,0.00006518489,0.0003918729,0.01423056],"genre_scores_gemma":[0.8582797,0.000600048,0.1222735,0.0005244899,0.0001152504,0.0002321611,0.00007769558,0.00006752614,0.01782973],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005243911,"threshold_uncertainty_score":0.0175426,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1802992795779892,"score_gpt":0.4634093563427167,"score_spread":0.2831100767647275,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}