{"id":"W3190031629","doi":"10.48550/arxiv.2108.02827","title":"An Elementary Proof that Q-learning Converges Almost Surely","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Q-learning; Function (biology); Bellman equation; Field (mathematics); Computer science; State (computer science); Value (mathematics); Elementary proof; Mathematical economics; Artificial intelligence; Connection (principal bundle); Mathematics; Discrete mathematics; Algorithm; Machine learning; Pure mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0004660072,0.0004733673,0.0004558873,0.0002704803,0.0003223728,0.0005204602,0.002788701,0.0003419593,0.0002125287],"category_scores_gemma":[0.0000457844,0.0005860093,0.0002395564,0.0005535472,0.0001202924,0.001090416,0.002927317,0.001377848,0.00006077253],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002874459,"about_ca_system_score_gemma":0.0003727326,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000223966,"about_ca_topic_score_gemma":0.00003805272,"domain_scores_codex":[0.9968442,0.0004588234,0.0002864728,0.00153788,0.0002719188,0.0006007556],"domain_scores_gemma":[0.9971231,0.0001247324,0.000468772,0.001760009,0.0002479718,0.0002754166],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000114701,0.00006343277,0.04191835,0.00008962476,0.0001467851,0.0005632158,0.0005102326,0.948858,0.00006920415,0.007295718,0.0001456887,0.0003282969],"study_design_scores_gemma":[0.0004436305,0.0001690101,0.003449599,0.0001459444,0.00008449482,0.00000921614,0.0006360165,0.9914979,0.001017395,0.0004838559,0.0013376,0.0007253876],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1852892,0.00006591523,0.8112801,0.00009058247,0.0008633484,0.0003104169,0.000002730003,0.0003925172,0.001705282],"genre_scores_gemma":[0.9912462,0.0002080979,0.005132908,0.0001910018,0.00008424607,0.000001426248,0.0001607202,0.00003867298,0.002936687],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8061472,"threshold_uncertainty_score":0.9996591,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07020033359668047,"score_gpt":0.1984780508432303,"score_spread":0.1282777172465498,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}