{"id":"W2157016390","doi":"","title":"Online Markov Decision Processes under Bandit Feedback","year":2010,"lang":"en","type":"article","venue":"","topic":"Advanced Bandit Algorithms Research","field":"Decision Sciences","cited_by":97,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Regret; Markov decision process; Computer science; Markov process; State (computer science); Markov chain; Mathematical optimization; Online learning; Reinforcement learning; Adversary; Function (biology); Action (physics); Artificial intelligence; Mathematics; Algorithm; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005270173,0.001357324,0.002608962,0.0008460726,0.001053839,0.002496416,0.001560786,0.002503314,0.004531298],"category_scores_gemma":[0.01990811,0.00083786,0.0007849715,0.001101158,0.002874091,0.002767249,0.002007205,0.002782475,0.0006980966],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003331646,"about_ca_system_score_gemma":0.00164178,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009546356,"about_ca_topic_score_gemma":0.005395324,"domain_scores_codex":[0.9969409,0.001393981,0.0001214936,0.0005448079,0.0003733161,0.0006255726],"domain_scores_gemma":[0.9782282,0.0171166,0.002261235,0.000668951,0.0009330076,0.0007919405],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003509131,0.00009438556,0.001217469,0.00007194692,0.00004932843,0.0002103478,0.0001057972,0.8608178,0.00036792,0.1307816,0.001034717,0.004897768],"study_design_scores_gemma":[0.00002735916,0.00001819691,0.00009603863,0.000009112238,0.000006748399,0.00001152188,0.000008856048,0.969237,0.0001058857,0.03035391,0.0001178143,0.000007515199],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2090917,0.0009927668,0.7764356,0.002874156,0.0001496875,0.0001598002,0.0005195177,0.0005569358,0.009219809],"genre_scores_gemma":[0.9780146,0.0003472011,0.01567182,0.000225461,0.00008952453,0.0001968936,0.0001698783,0.00004351152,0.005241133],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009546356,"threshold_uncertainty_score":0.02787167,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08846169459235069,"score_gpt":0.4450109408182472,"score_spread":0.3565492462258965,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}