{"id":"W3182076562","doi":"10.1111/poms.13778","title":"Sublinear regret for learning POMDPs","year":2022,"lang":"en","type":"article","venue":"Production and Operations Management","topic":"Advanced Bandit Algorithms Research","field":"Decision Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Regret; Markov decision process; Reinforcement learning; Oracle; Computer science; Sublinear function; Partially observable Markov decision process; Mathematical optimization; Upper and lower bounds; Time horizon; Artificial intelligence; Markov chain; Machine learning; Markov process; Markov model; Mathematics; Statistics; Discrete mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004232096,0.001872213,0.001948477,0.0008003219,0.0006866451,0.002195832,0.002097614,0.002015684,0.004793979],"category_scores_gemma":[0.02283557,0.001017415,0.001353007,0.0007695687,0.002428145,0.003196631,0.002329949,0.003975477,0.0005536967],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004089538,"about_ca_system_score_gemma":0.002676037,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007025135,"about_ca_topic_score_gemma":0.005150304,"domain_scores_codex":[0.997026,0.001323019,0.0001166995,0.0005154307,0.0006244422,0.0003944077],"domain_scores_gemma":[0.9812136,0.01615395,0.0009592916,0.0005090978,0.0006193661,0.0005446898],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001102411,0.00005692905,0.0004744703,0.0001269333,0.00004625727,0.00006262823,0.0000574994,0.9126671,0.0002483826,0.07572598,0.001698426,0.008725126],"study_design_scores_gemma":[0.000009198547,0.00001120487,0.00002945796,0.000006433261,0.000003115488,0.000003699289,0.000002809377,0.9744905,0.00007422409,0.02521071,0.0001557755,0.000002834148],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02330226,0.0009669056,0.9656557,0.001524256,0.0001270738,0.00007023625,0.0002868944,0.0007002454,0.007366427],"genre_scores_gemma":[0.8633843,0.0008773034,0.1241986,0.0006080425,0.0002114022,0.0004375132,0.0006863431,0.0003296013,0.009266939],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007025135,"threshold_uncertainty_score":0.02967179,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1261665713380833,"score_gpt":0.4307963280159451,"score_spread":0.3046297566778619,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}