{"id":"W1969276875","doi":"10.1016/j.tcs.2014.09.029","title":"Near-optimal PAC bounds for discounted MDPs","year":2014,"lang":"en","type":"article","venue":"Theoretical Computer Science","topic":"Advanced Bandit Algorithms Research","field":"Decision Sciences","cited_by":43,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Upper and lower bounds; Markov decision process; Logarithm; Sample complexity; Mathematics; Reinforcement learning; State space; Stochastic matrix; Markov chain; Matrix (chemical analysis); Space (punctuation); Markov process; Mathematical optimization; Applied mathematics; Combinatorics; Computer science; Statistics; Mathematical analysis","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008327335,0.003738836,0.005303547,0.003462454,0.002481854,0.009079892,0.004883167,0.004728241,0.01447466],"category_scores_gemma":[0.05770148,0.002196721,0.002211889,0.003913435,0.003878361,0.01391465,0.007141346,0.01102645,0.001756873],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.008643851,"about_ca_system_score_gemma":0.008258153,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004809622,"about_ca_topic_score_gemma":0.006839177,"domain_scores_codex":[0.9934277,0.002098758,0.0002991857,0.0009537127,0.001801049,0.001419635],"domain_scores_gemma":[0.9427331,0.0491838,0.00143941,0.002520512,0.002055771,0.002067425],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0006423899,0.0003353945,0.0007585368,0.0006907861,0.0001357891,0.0001115667,0.0002401594,0.6149774,0.000930909,0.3300956,0.01056138,0.04052016],"study_design_scores_gemma":[0.00003980904,0.0000644892,0.0001258658,0.0001324424,0.00003926275,0.00005885847,0.00005098038,0.6909906,0.0005458053,0.3062524,0.001675011,0.00002452096],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0450282,0.007226659,0.8873165,0.006335709,0.0005198327,0.0003280487,0.001620838,0.001164661,0.05045956],"genre_scores_gemma":[0.7585099,0.006883887,0.2051786,0.00250922,0.001295738,0.001191945,0.001872576,0.00102127,0.02153686],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01447466,"threshold_uncertainty_score":0.06271583,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04680964409173786,"score_gpt":0.418173190851378,"score_spread":0.3713635467596401,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}