{"id":"W4400868993","doi":"10.31219/osf.io/s98ex","title":"Reinforcement Learning: Tutorial and Survey","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Data Stream Mining Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Markov decision process; Temporal difference learning; Q-learning; Bellman equation; Reinforcement; Computer science; Markov process; Markov chain; Artificial intelligence; Process (computing); Machine learning; Mathematical optimization; Mathematics; Engineering; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001055405,0.001437356,0.001573824,0.001395175,0.0003178513,0.001498201,0.001268119,0.001395039,0.01058594],"category_scores_gemma":[0.002848583,0.0007231468,0.0009788279,0.003282407,0.000684246,0.002487365,0.001131497,0.002525633,0.00531496],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001074547,"about_ca_system_score_gemma":0.001193647,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001970128,"about_ca_topic_score_gemma":0.001112607,"domain_scores_codex":[0.9992936,0.0001548155,0.00007489416,0.0001688178,0.0002576734,0.00005021557],"domain_scores_gemma":[0.9989372,0.0007218631,0.00004281821,0.00008175962,0.0001644912,0.00005183474],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00007383533,0.0002347641,0.0008389361,0.003407003,0.0001283145,0.0001278369,0.0001346524,0.02730947,0.0009258795,0.08804657,0.06861859,0.8101542],"study_design_scores_gemma":[0.00002991798,0.0002153568,0.001272413,0.001558353,0.00008769715,0.0008315017,0.00009441572,0.05863303,0.001098998,0.1759414,0.7601553,0.00008162246],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.002124696,0.6423221,0.3032735,0.002399872,0.002904085,0.0001109844,0.0005455672,0.0009487777,0.04537046],"genre_scores_gemma":[0.0633465,0.7626027,0.1302778,0.002812037,0.007210067,0.0004839242,0.001831944,0.0006289743,0.03080613],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.01058594,"threshold_uncertainty_score":0.0354135,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03534541906737268,"score_gpt":0.2944780889739285,"score_spread":0.2591326699065558,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}