{"id":"W2295665692","doi":"10.5555/2034396.2034406","title":"Efficient planning in R-max","year":2011,"lang":"en","type":"article","venue":"Adaptive Agents and Multi-Agents Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Mathematical optimization; Markov decision process; Value (mathematics); Artificial intelligence; Algorithm; Machine learning; Mathematics; Markov process","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00231834,0.0009939711,0.001227201,0.000551354,0.0006931092,0.001295482,0.001602339,0.00115221,0.006485067],"category_scores_gemma":[0.006433127,0.0006776581,0.0009537865,0.0008113621,0.001883533,0.002297632,0.001998667,0.001608168,0.001214496],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001307407,"about_ca_system_score_gemma":0.002588694,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003253119,"about_ca_topic_score_gemma":0.004750637,"domain_scores_codex":[0.9984754,0.0006714217,0.00007578097,0.000387853,0.0002177898,0.0001716507],"domain_scores_gemma":[0.9972608,0.001939329,0.0002096124,0.0003195007,0.0001778912,0.00009295864],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000194701,0.00007243508,0.0004244225,0.0002175092,0.00003947432,0.0001133079,0.0001376161,0.8238164,0.001334257,0.1013167,0.003045019,0.06928816],"study_design_scores_gemma":[0.00003285994,0.00004773166,0.00008069953,0.00001918501,0.00001103802,0.00003192512,0.00002818847,0.9066634,0.001549079,0.08909764,0.002427042,0.00001124905],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01227435,0.0002920233,0.9777425,0.0002677164,0.00002701994,0.00009720672,0.0001248149,0.0009348306,0.008239472],"genre_scores_gemma":[0.3941182,0.000400326,0.597923,0.0002227628,0.00003672107,0.0004050702,0.0003705119,0.0003916145,0.006131679],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006485067,"threshold_uncertainty_score":0.02169472,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1242030640907431,"score_gpt":0.2922210493630056,"score_spread":0.1680179852722625,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}