{"id":"W2149586740","doi":"10.1613/jair.4117","title":"Scalable and Efficient Bayes-Adaptive Reinforcement Learning Based on Monte-Carlo Tree Search","year":2013,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":79,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Gatsby Charitable Foundation; Royal Society","keywords":"Computer science; Monte Carlo tree search; Machine learning; Reinforcement learning; Tree (set theory); Scalability; Bayesian probability; Artificial intelligence; Benchmark (surveying); Monte Carlo method; Mathematical optimization; Mathematics; Statistics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001926857,0.0007508426,0.001797223,0.0005669168,0.0004505221,0.0008539415,0.001940788,0.001182488,0.002998225],"category_scores_gemma":[0.00971664,0.0005829147,0.0005686133,0.0006918503,0.001239188,0.001478544,0.001574844,0.002057521,0.0005389464],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001165106,"about_ca_system_score_gemma":0.002634323,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006894886,"about_ca_topic_score_gemma":0.008601874,"domain_scores_codex":[0.9988276,0.0004542182,0.00004834862,0.0001508272,0.0004041819,0.000114712],"domain_scores_gemma":[0.9960204,0.003024276,0.000253006,0.0002649516,0.0002793802,0.0001580548],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000116672,0.00008672538,0.0007542658,0.00006337798,0.00004015286,0.00005462065,0.00007465323,0.9003484,0.0007350576,0.02501696,0.001443137,0.07126589],"study_design_scores_gemma":[0.00001105728,0.000009978257,0.0000296098,0.000003251593,0.00000262589,0.000005858009,0.000002460069,0.9931479,0.00008580121,0.006567268,0.0001318499,0.000002289593],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01081555,0.0001821295,0.9861163,0.000172377,0.00002482189,0.00006026662,0.00003358808,0.0005985299,0.001996305],"genre_scores_gemma":[0.6410341,0.0002020992,0.3562351,0.0001902058,0.00004659309,0.0003066898,0.0001483506,0.0001484593,0.001688297],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006894886,"threshold_uncertainty_score":0.01370949,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1097453708321637,"score_gpt":0.354419875162297,"score_spread":0.2446745043301333,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}