{"id":"W3157613611","doi":"","title":"Online Sparse Reinforcement Learning","year":2021,"lang":"en","type":"article","venue":"International Conference on Artificial Intelligence and Statistics","topic":"Advanced Bandit Algorithms Research","field":"Decision Sciences","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Regret; Markov decision process; Reinforcement learning; Dimension (graph theory); Upper and lower bounds; Time horizon; Q-learning; Oracle; Mathematics; Lasso (programming language); Mathematical optimization; Computer science; Markov process; Combinatorics; Discrete mathematics; Artificial intelligence; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002330942,0.001074051,0.001882758,0.0003616973,0.000408253,0.001015909,0.001606502,0.001602812,0.00310397],"category_scores_gemma":[0.0142799,0.0005043797,0.0004878666,0.0004411484,0.001662245,0.001972788,0.001647748,0.002282195,0.0003751948],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001278404,"about_ca_system_score_gemma":0.001409462,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00327464,"about_ca_topic_score_gemma":0.002834385,"domain_scores_codex":[0.9985801,0.0005971172,0.0000527514,0.0003317843,0.0002359721,0.000202331],"domain_scores_gemma":[0.9909383,0.0070607,0.0006908505,0.0005719895,0.0003546595,0.0003835008],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002400783,0.0001498752,0.001190134,0.0001267237,0.00006067125,0.0001050375,0.0000531428,0.9395552,0.0005469309,0.03516541,0.001726626,0.02108032],"study_design_scores_gemma":[0.00001914735,0.00002533522,0.00006955346,0.000005376287,0.000003439957,0.00000751935,0.000003292576,0.9848232,0.00009993629,0.01478128,0.0001591293,0.000002712475],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06278317,0.0006849523,0.9291012,0.001444208,0.0001172393,0.00009825698,0.0002402032,0.0005905677,0.004940187],"genre_scores_gemma":[0.9285226,0.0002703202,0.06671134,0.0003585426,0.0001418587,0.0001803185,0.0002949898,0.00006624899,0.003453739],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00327464,"threshold_uncertainty_score":0.01232737,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4004819125429087,"score_gpt":0.497299212462624,"score_spread":0.09681729991971538,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}