{"id":"W4244090038","doi":"10.1007/978-1-4471-7452-3_17","title":"Reinforcement Learning","year":2019,"lang":"en","type":"book-chapter","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":false,"ca_institutions":"Concordia University","funders":"","keywords":"Reinforcement learning; Reinforcement; Computer science; Error-driven learning; Formalism (music); Artificial intelligence; Psychology; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001815363,0.0007733561,0.0003785861,0.0004187627,0.0003075539,0.001024439,0.0006618498,0.00066486,0.05584573],"category_scores_gemma":[0.0007027984,0.0002262119,0.0002460693,0.0004601758,0.0005745317,0.001042849,0.0007633425,0.001337999,0.0223146],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007610962,"about_ca_system_score_gemma":0.0005709082,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009202816,"about_ca_topic_score_gemma":0.001566487,"domain_scores_codex":[0.9998542,0.00002004695,0.000004322388,0.00003410408,0.00007666907,0.00001070109],"domain_scores_gemma":[0.9998684,0.00004849445,0.000006764978,0.00002695465,0.0000358107,0.00001355587],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00002618182,0.00007401361,0.00009061438,0.0001955629,0.00001264682,0.0000350535,0.00005613989,0.01115093,0.001703045,0.1806864,0.1434331,0.6625364],"study_design_scores_gemma":[0.00001273933,0.00005469434,0.000297166,0.0002006523,0.00001103094,0.0001899234,0.00003871274,0.03160063,0.002693062,0.2264787,0.7384009,0.00002176228],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"other","genre_gemma":"methods","genre_scores_codex":[0.001841705,0.008910156,0.2482214,0.001717344,0.00112108,0.0000836572,0.0002722786,0.0015386,0.7362937],"genre_scores_gemma":[0.05372624,0.008980601,0.05871386,0.0007702147,0.000507107,0.0001651509,0.0005593674,0.0003679387,0.8762096],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.05584573,"threshold_uncertainty_score":0.1868225,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01983590515921726,"score_gpt":0.2269463777335071,"score_spread":0.2071104725742899,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}