{"id":"W4243101510","doi":"10.1002/9781118445112.stat08000","title":"Reinforcement Learning","year":2017,"lang":"en","type":"other","venue":"Wiley StatsRef: Statistics Reference Online","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University","funders":"","keywords":"Reinforcement learning; Computer science; A priori and a posteriori; Sequence (biology); Artificial intelligence; Quality (philosophy); Reinforcement; Machine learning; Psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0003181897,0.000869224,0.0008907907,0.0005439047,0.0004241988,0.0006670341,0.003467757,0.0005071291,0.001447718],"category_scores_gemma":[0.0006271472,0.000850586,0.00009707216,0.000206586,0.0002751589,0.0002760036,0.001213831,0.001532712,0.0009891036],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002107453,"about_ca_system_score_gemma":0.0005520557,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004464115,"about_ca_topic_score_gemma":0.0002432949,"domain_scores_codex":[0.995591,0.0001605884,0.0007326906,0.001125431,0.001354274,0.00103606],"domain_scores_gemma":[0.9947877,0.0001980464,0.001776146,0.002593876,0.0002710316,0.0003732364],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000009794948,0.00007089866,0.00006739977,0.0002777941,0.0002106738,0.0001729863,0.0001302759,0.05183984,0.000009880401,0.08148804,0.8169449,0.04877753],"study_design_scores_gemma":[0.0005045365,0.0003226104,0.00002847635,0.0008596986,0.00005067655,0.000008992144,0.00001242068,0.1584417,0.000004432714,0.0006409884,0.8382706,0.0008548518],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"other","genre_scores_codex":[1.813994e-7,0.0003840112,0.7244181,0.00006407052,0.0008171682,0.0003662418,0.00040571,0.0007050133,0.2728395],"genre_scores_gemma":[0.0001798763,0.004122622,0.3169039,0.0001315911,0.0002960957,0.00002394834,0.002705986,0.000375931,0.6752601],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.4075142,"threshold_uncertainty_score":0.9997888,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04106047788657132,"score_gpt":0.3130910355740826,"score_spread":0.2720305576875113,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}