{"id":"W4285255083","doi":"10.1007/978-3-031-79167-3_2","title":"Reinforcement Learning Theory","year":2022,"lang":"en","type":"book-chapter","venue":"Synthesis lectures on artificial intelligence and machine learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta; Canadian Institute for Advanced Research","funders":"","keywords":"Reinforcement learning; Reinforcement; Computer science; Plan (archaeology); Context (archaeology); Cognitive science; Artificial intelligence; Psychology; Social psychology; History","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004618171,0.0008975125,0.0007600316,0.0005182783,0.0004729725,0.001551247,0.0008542576,0.001138113,0.0248644],"category_scores_gemma":[0.001512469,0.0003039574,0.0003957243,0.000584167,0.001582602,0.001413899,0.0006766078,0.001948968,0.005349969],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001425793,"about_ca_system_score_gemma":0.0007569863,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001642988,"about_ca_topic_score_gemma":0.001439187,"domain_scores_codex":[0.9996815,0.00009758776,0.00000928932,0.00006117237,0.0001248719,0.00002557819],"domain_scores_gemma":[0.999686,0.0001807984,0.00001579276,0.00004199346,0.00005797524,0.00001743153],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000009482601,0.00002535624,0.0000700151,0.00008206214,0.00001504873,0.00002606859,0.00005077343,0.01363208,0.0002408591,0.8889941,0.03058247,0.06627173],"study_design_scores_gemma":[0.000009310575,0.00001331228,0.00009234592,0.00005366662,0.000007780041,0.00003658761,0.00001735494,0.02038667,0.0002345278,0.8993231,0.0798174,0.000008035408],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"other","genre_gemma":"methods","genre_scores_codex":[0.002973957,0.01712814,0.4308399,0.005412612,0.00159446,0.00006378479,0.0002706502,0.0004526637,0.5412639],"genre_scores_gemma":[0.324924,0.0178678,0.08920492,0.002718871,0.001819557,0.0003924222,0.000493066,0.0003835413,0.5621958],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.0248644,"threshold_uncertainty_score":0.08317971,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04072940189197726,"score_gpt":0.2609339799200801,"score_spread":0.2202045780281028,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}