{"id":"W1526654727","doi":"","title":"Model-based reinforcement learning with nearly tight exploration complexity bounds","year":2010,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":105,"is_retracted":false,"has_abstract":true,"ca_institutions":"Athabasca University; University of Alberta","funders":"","keywords":"Reinforcement learning; Markov decision process; Upper and lower bounds; Logarithm; Computer science; Probably approximately correct learning; Markov process; Sample complexity; Contrast (vision); Binary logarithm; Exploratory analysis; Q-learning; Algorithm; Exploratory research; Markov chain; Process (computing); Mathematics; Artificial intelligence; Combinatorics; Machine learning; Active learning (machine learning); Computational learning theory; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004391732,0.002056177,0.0028031,0.0008487902,0.0007190059,0.002407833,0.002234634,0.00220216,0.003312354],"category_scores_gemma":[0.0233492,0.001048985,0.001398805,0.0008258976,0.002198512,0.004883931,0.004446089,0.005547476,0.0007712414],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002356217,"about_ca_system_score_gemma":0.002397501,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002921685,"about_ca_topic_score_gemma":0.002973648,"domain_scores_codex":[0.9966839,0.001079726,0.0001563053,0.0005742402,0.001029545,0.0004762922],"domain_scores_gemma":[0.9811255,0.01511501,0.0009648556,0.00156702,0.0006939395,0.0005336257],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002010535,0.0001271216,0.0007837479,0.000155926,0.00006851704,0.00006392212,0.000095549,0.9112448,0.001147877,0.0554816,0.001552867,0.02907709],"study_design_scores_gemma":[0.00001692431,0.00003169016,0.00004351783,0.000007728079,0.00000669387,0.00001198296,0.000003356215,0.9734664,0.0001879812,0.02604886,0.0001696849,0.000005176683],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02477024,0.00101554,0.96709,0.000818133,0.00005734742,0.00006538086,0.00007087171,0.0008329722,0.005279442],"genre_scores_gemma":[0.8122701,0.0008706289,0.180129,0.0005417372,0.0001724761,0.0004139123,0.0003030687,0.0003884427,0.004910736],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004391732,"threshold_uncertainty_score":0.02322596,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04441804103927028,"score_gpt":0.2579625849142766,"score_spread":0.2135445438750063,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}