{"id":"W2909958171","doi":"10.1007/978-3-030-10546-4_2","title":"Reinforcement Learning and Deep Reinforcement Learning","year":2019,"lang":"en","type":"book-chapter","venue":"Springer briefs in electrical and computer engineering","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"ca_institutions":"Carleton University","funders":"","keywords":"Reinforcement learning; Reinforcement; Artificial intelligence; Deep learning; Q-learning; Computer science; Psychology; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003208279,0.0008952185,0.0006922636,0.0005892866,0.0001792295,0.001238839,0.0005825323,0.000923981,0.01537039],"category_scores_gemma":[0.001220306,0.0003425757,0.0002753432,0.001170595,0.0009525597,0.001815763,0.000710169,0.001978663,0.004113698],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009308565,"about_ca_system_score_gemma":0.0005602767,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001212166,"about_ca_topic_score_gemma":0.00155455,"domain_scores_codex":[0.9998237,0.00003747708,0.000009783261,0.00003813222,0.00007467444,0.00001621458],"domain_scores_gemma":[0.9997161,0.0001685648,0.00002016251,0.00003367671,0.00004310699,0.0000183973],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00003408728,0.00006126606,0.0001428087,0.0004391106,0.00002564615,0.0000484203,0.0000444106,0.03707804,0.00123445,0.2832113,0.06464452,0.613036],"study_design_scores_gemma":[0.0000131501,0.00005346044,0.0004836755,0.0003281311,0.00001828898,0.0001583374,0.00003228518,0.09194309,0.00194833,0.5530242,0.3519603,0.00003666047],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.004519853,0.1759236,0.5990353,0.006806851,0.006217887,0.00004696691,0.0003958385,0.00102694,0.2060268],"genre_scores_gemma":[0.1797313,0.1517903,0.1487873,0.002004463,0.006532528,0.0002016908,0.000888589,0.0006360666,0.5094278],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01537039,"threshold_uncertainty_score":0.05141908,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.006681597574720062,"score_gpt":0.1933348836968352,"score_spread":0.1866532861221151,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}