{"id":"W4400868993","doi":"10.31219/osf.io/s98ex","title":"Reinforcement Learning: Tutorial and Survey","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Data Stream Mining Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Markov decision process; Temporal difference learning; Q-learning; Bellman equation; Reinforcement; Computer science; Markov process; Markov chain; Artificial intelligence; Process (computing); Machine learning; Mathematical optimization; Mathematics; Engineering; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.001003279,0.0002216849,0.0002225104,0.0001447296,0.00004287114,0.0008800307,0.001153869,0.0001931482,0.00002295684],"category_scores_gemma":[0.0001821228,0.0001957286,0.00004294162,0.0001222257,0.00003775479,0.00009767626,0.01379953,0.0007620135,0.00006967883],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005510712,"about_ca_system_score_gemma":0.0001629567,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00118354,"about_ca_topic_score_gemma":0.0000397813,"domain_scores_codex":[0.9984232,0.0001090098,0.0002500513,0.000732412,0.0002834374,0.0002018543],"domain_scores_gemma":[0.9987329,0.000127418,0.00008212638,0.0009178792,0.00006197689,0.00007772941],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002148588,0.0000479716,0.002383437,0.0005193884,0.0002452871,0.0001118324,0.001658409,0.0007220518,0.00008151548,0.4468608,0.4158984,0.1314494],"study_design_scores_gemma":[0.0005626482,0.0008874472,0.006427923,0.001037446,0.00008574326,0.0000581379,0.00002595313,0.4756183,0.005103983,0.2357236,0.2714699,0.002998817],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.005381688,0.0003661925,0.9135582,0.0007519726,0.006598036,0.0005965402,0.00003028733,0.005119695,0.06759737],"genre_scores_gemma":[0.7872434,0.0002259222,0.1970151,0.0001364173,0.000575141,0.00009222919,0.0003074408,0.00004347302,0.01436093],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.7818617,"threshold_uncertainty_score":0.9941767,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03534541906737268,"score_gpt":0.2944780889739285,"score_spread":0.2591326699065558,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}