{"id":"W3212514152","doi":"","title":"Regime Switching Bandits","year":2021,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Advanced Bandit Algorithms Research","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Markov chain; Markov decision process; Upper and lower bounds; Stochastic matrix; Markov process; Observable; Computer science; State (computer science); Partially observable Markov decision process; Matrix (chemical analysis); Reinforcement learning; Probably approximately correct learning; Artificial intelligence; Mathematical optimization; Mathematics; Markov model; Algorithm; Machine learning; Unsupervised learning; Generalization error; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003417331,0.0009064387,0.001862498,0.000758905,0.0007913026,0.002044139,0.001385742,0.002016186,0.005208036],"category_scores_gemma":[0.01353256,0.000527798,0.0007842952,0.000941743,0.00193058,0.002388865,0.001776976,0.002248338,0.0005966788],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001606374,"about_ca_system_score_gemma":0.0007662778,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002293197,"about_ca_topic_score_gemma":0.0014388,"domain_scores_codex":[0.998336,0.0007768706,0.00006727196,0.0002894025,0.0002419643,0.0002885342],"domain_scores_gemma":[0.9897671,0.008285281,0.0009191932,0.0004167901,0.0003393816,0.0002723073],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002479728,0.0001251771,0.001217901,0.000110298,0.00008646416,0.0001759138,0.000121193,0.7754666,0.00121366,0.1957227,0.001760396,0.02375182],"study_design_scores_gemma":[0.00001510899,0.00002086318,0.00009399067,0.00001011212,0.000006530825,0.00001559757,0.000009716213,0.9565457,0.0001892496,0.0428003,0.0002862586,0.000006675557],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09535291,0.0008827967,0.8884451,0.001170942,0.0001102204,0.0001065633,0.0001713243,0.0003878433,0.01337227],"genre_scores_gemma":[0.955674,0.0003983249,0.03779323,0.0002806086,0.00008865019,0.000184976,0.00009903848,0.00005611284,0.005424981],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005208036,"threshold_uncertainty_score":0.01807278,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1121872631780489,"score_gpt":0.4130105025504202,"score_spread":0.3008232393723713,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}