{"id":"W2154806059","doi":"","title":"Online Learning in Markov Decision Processes with Adversarially Chosen Transition Probability Distributions","year":2013,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Advanced Bandit Algorithms Research","field":"Decision Sciences","cited_by":51,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Regret; Markov decision process; Adversarial system; Mathematical optimization; Computer science; Shortest path problem; Path (computing); Mathematics; Graph; Markov process; Theoretical computer science; Artificial intelligence; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005563384,0.001846935,0.002523211,0.0007986809,0.0009791711,0.002166846,0.002406219,0.002641973,0.003624765],"category_scores_gemma":[0.02008713,0.001275614,0.001544895,0.001222978,0.003011391,0.004338257,0.002902901,0.004920834,0.0004543044],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003360484,"about_ca_system_score_gemma":0.002477696,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006291115,"about_ca_topic_score_gemma":0.004276509,"domain_scores_codex":[0.996435,0.001682246,0.0001232201,0.0009166147,0.000332439,0.0005104079],"domain_scores_gemma":[0.9691498,0.02725654,0.001602985,0.0008671125,0.0004707725,0.0006527393],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001644416,0.0000799242,0.0005573628,0.00006642406,0.00005219928,0.00009459272,0.00005433791,0.9614204,0.0002465429,0.02994444,0.0004558732,0.006863462],"study_design_scores_gemma":[0.00002331244,0.0000275584,0.0000534651,0.000005022957,0.000006567874,0.00001004379,0.000006468935,0.9662714,0.0001461462,0.03331891,0.000126213,0.000004886037],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05293016,0.000587395,0.9416018,0.001345783,0.00006902346,0.0001168315,0.0001843134,0.000466921,0.00269776],"genre_scores_gemma":[0.8832706,0.0006921391,0.109276,0.000432838,0.000150185,0.0004234495,0.000375275,0.0001527412,0.005226795],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006291115,"threshold_uncertainty_score":0.02942228,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09547229580583252,"score_gpt":0.270794819568804,"score_spread":0.1753225237629715,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}