{"id":"W3209247505","doi":"10.1109/ccece53047.2021.9569056","title":"Reinforcement Learning Algorithms: An Overview and Classification","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":102,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Machine learning; Perspective (graphical); Field (mathematics); Robotics; Learning classifier system; Robot learning; Drone; Algorithm; Robot; Mobile robot","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00246105,0.001727441,0.001612244,0.003749877,0.0005806332,0.003209795,0.002114681,0.002441392,0.002778934],"category_scores_gemma":[0.005906574,0.0008884426,0.00135236,0.005328145,0.001532659,0.003731048,0.001469993,0.003682625,0.001826853],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001627053,"about_ca_system_score_gemma":0.001296984,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002577533,"about_ca_topic_score_gemma":0.0009697261,"domain_scores_codex":[0.9982217,0.0003527775,0.0002279784,0.0003799603,0.0007093558,0.0001081701],"domain_scores_gemma":[0.9967967,0.002105144,0.0002313748,0.0001979387,0.0005469323,0.0001219176],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001080875,0.0002054698,0.003947278,0.002222497,0.0001272777,0.0001420604,0.0002098553,0.06290054,0.000688921,0.1422313,0.01829607,0.7689206],"study_design_scores_gemma":[0.00006116784,0.0003741871,0.004132259,0.00215078,0.0001176896,0.001196621,0.0002206986,0.3452923,0.00197144,0.3777535,0.2665586,0.0001707661],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"review","genre_scores_codex":[0.005523545,0.3375566,0.6236008,0.003331001,0.0009922091,0.0002844983,0.0002946283,0.0007821093,0.02763465],"genre_scores_gemma":[0.1509203,0.4617899,0.3666858,0.001564079,0.004019055,0.0008376352,0.001421973,0.0003942512,0.01236698],"genre_candidate":"review","genre_consensus":null,"teacher_disagreement_score":0.003749877,"threshold_uncertainty_score":0.01301539,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0971463287431973,"score_gpt":0.3273202482426542,"score_spread":0.2301739194994569,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}