{"id":"W7124138561","doi":"10.65109/kjtg4247","title":"Efficient planning in R-max","year":2011,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Value (mathematics); Automated planning and scheduling; Markov decision process; Order (exchange); Active learning (machine learning)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002299341,0.0009746728,0.001236938,0.0005290918,0.0006676781,0.001241416,0.001544026,0.001127931,0.006237436],"category_scores_gemma":[0.006315023,0.0006730783,0.0009382023,0.0007825274,0.00185374,0.002234397,0.001989224,0.001597519,0.00120269],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001283628,"about_ca_system_score_gemma":0.002527818,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003074239,"about_ca_topic_score_gemma":0.004604728,"domain_scores_codex":[0.9984614,0.000683764,0.00007532354,0.0003806108,0.0002244463,0.0001744141],"domain_scores_gemma":[0.9974075,0.00180773,0.0001994722,0.0003221433,0.0001711326,0.00009206949],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002016159,0.00007208779,0.0004080936,0.0002160645,0.00003778281,0.0001127335,0.0001294178,0.8304871,0.00139537,0.09467988,0.003146382,0.06911347],"study_design_scores_gemma":[0.00003321796,0.0000489464,0.00008063259,0.00001848681,0.00001070229,0.00003155402,0.00002671654,0.9075646,0.001607101,0.088226,0.002341099,0.00001094042],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01362514,0.0003241243,0.9760439,0.0002863837,0.00002879841,0.00009590146,0.0001340079,0.001043716,0.008418133],"genre_scores_gemma":[0.4104761,0.0004105554,0.581831,0.0002251061,0.00003671301,0.0003776459,0.0003840177,0.000401791,0.005857085],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006237436,"threshold_uncertainty_score":0.02086633,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06178843939825616,"score_gpt":0.2670726314811488,"score_spread":0.2052841920828927,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}