{"id":"W2073384958","doi":"10.1007/978-3-031-01551-9","title":"Algorithms for Reinforcement Learning","year":2010,"lang":"en","type":"book","venue":"Synthesis lectures on artificial intelligence and machine learning","topic":"Advanced Bandit Algorithms Research","field":"Decision Sciences","cited_by":750,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Machine learning; Learning classifier system; Hyper-heuristic; Instance-based learning; Active learning (machine learning); Term (time); Unsupervised learning; Core (optical fiber); Robot learning; Robot","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000523601,0.001322646,0.001110132,0.0005523246,0.0004457257,0.001485027,0.001113923,0.001164806,0.02105755],"category_scores_gemma":[0.002201933,0.0004421466,0.0005456653,0.0009108802,0.0009849702,0.001681204,0.001075604,0.002461988,0.006246857],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009265375,"about_ca_system_score_gemma":0.0005173201,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001292076,"about_ca_topic_score_gemma":0.001412686,"domain_scores_codex":[0.9996834,0.000083503,0.00001548262,0.00007871607,0.0001115687,0.00002733831],"domain_scores_gemma":[0.9995756,0.0002282543,0.00001960386,0.0000903739,0.00006854667,0.00001767255],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00004085949,0.0000548857,0.0001756495,0.0001424175,0.00004752535,0.0000453638,0.00006346054,0.147449,0.000719836,0.3975665,0.04332735,0.4103672],"study_design_scores_gemma":[0.0000294695,0.00002252364,0.0001195374,0.00005496201,0.00001612069,0.00005780324,0.00001809724,0.42947,0.0006275054,0.520232,0.04933888,0.00001313114],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.001586803,0.003964288,0.9395554,0.0006565573,0.0006556123,0.00004091419,0.0001163267,0.001006599,0.05241738],"genre_scores_gemma":[0.2061871,0.005526741,0.6102038,0.0007931858,0.001227043,0.0006069793,0.0008349484,0.0009225525,0.1736977],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02105755,"threshold_uncertainty_score":0.07044452,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1604075222470484,"score_gpt":0.4127314576545998,"score_spread":0.2523239354075514,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}