{"id":"W4310613925","doi":"10.1002/9781119808602.ch3","title":"Reinforcement Learning","year":2022,"lang":"en","type":"other","venue":"","topic":"Adaptive Dynamic Programming Control","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Markov decision process; Dynamic programming; Optimal control; Mathematical proof; Computer science; Bellman equation; Mathematical optimization; Class (philosophy); Stochastic control; Control (management); Field (mathematics); Markov process; Mathematics; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006155191,0.0006921303,0.0004839273,0.00032041,0.0003672433,0.001194927,0.001059253,0.0008672663,0.02371237],"category_scores_gemma":[0.002714141,0.00017564,0.0003769046,0.0003042055,0.0007661267,0.00121226,0.0009372847,0.001385369,0.003427766],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007456292,"about_ca_system_score_gemma":0.0007524313,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001044961,"about_ca_topic_score_gemma":0.001411681,"domain_scores_codex":[0.9995546,0.0001414899,0.00001932187,0.00009961313,0.0001511244,0.00003391491],"domain_scores_gemma":[0.999433,0.000327095,0.00004024099,0.00007570926,0.00007943453,0.00004442872],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00004205948,0.00009135099,0.0003822585,0.0001749236,0.0000349688,0.00006914961,0.00008636525,0.08177592,0.0008552794,0.7049196,0.02271859,0.1888495],"study_design_scores_gemma":[0.00004345035,0.00008390607,0.0002521987,0.0001270926,0.00001746678,0.0001351852,0.00005337108,0.2719294,0.0010031,0.5785427,0.1477881,0.00002402766],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"other","genre_scores_codex":[0.003879681,0.003347162,0.8664241,0.002864419,0.0005069555,0.000124107,0.000297457,0.0007455495,0.1218105],"genre_scores_gemma":[0.4504609,0.01106334,0.3830813,0.001852763,0.0009113627,0.0008460631,0.0009195576,0.0003420966,0.1505226],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.02371237,"threshold_uncertainty_score":0.0793258,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.00743149318627673,"score_gpt":0.2207259863854765,"score_spread":0.2132944931991998,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}