{"id":"W4210257517","doi":"10.1109/cdc45484.2021.9682777","title":"Convergence and Near Optimality of Q-Learning with Finite Memory for Partially Observed Models","year":2021,"lang":"en","type":"article","venue":"2021 60th IEEE Conference on Decision and Control (CDC)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Partially observable Markov decision process; Reinforcement learning; Markov decision process; Convergence (economics); Computer science; Quantization (signal processing); Q-learning; Mathematical optimization; Limit (mathematics); State space; Optimal control; Markov process; Mathematics; Algorithm; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00703588,0.001016895,0.001965969,0.001018765,0.0008301791,0.001345634,0.001687741,0.001788868,0.001977296],"category_scores_gemma":[0.03341845,0.0007332667,0.0009257371,0.0006796508,0.003474258,0.002575897,0.002588276,0.002887541,0.0002911224],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002059812,"about_ca_system_score_gemma":0.003189475,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006570459,"about_ca_topic_score_gemma":0.002697459,"domain_scores_codex":[0.9976223,0.001166555,0.0001195068,0.0003967987,0.0004755895,0.0002191779],"domain_scores_gemma":[0.9727553,0.02366278,0.000991724,0.0007027327,0.001486249,0.0004011576],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001117929,0.0000713256,0.0008779195,0.0001108376,0.00004878598,0.00005953546,0.0001394971,0.9113363,0.000631939,0.06865873,0.0003877652,0.01756564],"study_design_scores_gemma":[0.00001100923,0.00002925157,0.00005277172,0.00001152643,0.000003397395,0.000007987745,0.000006594148,0.9792536,0.0001933329,0.02032725,0.00009835701,0.000004861988],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02058181,0.0003566901,0.976894,0.0003241425,0.0000253573,0.00004998925,0.00003058563,0.0001483353,0.001589138],"genre_scores_gemma":[0.8248751,0.0006088908,0.1708193,0.0002889574,0.00006500266,0.0003408041,0.0001512085,0.0001427281,0.0027081],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.00703588,"threshold_uncertainty_score":0.03720975,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06092869236888402,"score_gpt":0.2648753527444318,"score_spread":0.2039466603755478,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}