{"id":"W2963800416","doi":"","title":"Policy Error Bounds for Model-Based Reinforcement Learning with Factored Linear Models","year":2016,"lang":"en","type":"article","venue":"Conference on Learning Theory","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Mathematical proof; Markov decision process; Contraction (grammar); Property (philosophy); Computer science; Mathematics; Applied mathematics; Mathematical optimization; Measure (data warehouse); Class (philosophy); Markov process; Linear model; Artificial intelligence; Machine learning; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01071265,0.002455215,0.002485827,0.001451945,0.0007072944,0.002737503,0.002495757,0.002037426,0.003513793],"category_scores_gemma":[0.05358884,0.001047204,0.001562213,0.0009388541,0.003517204,0.005631635,0.004145496,0.005475139,0.0005184544],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004202228,"about_ca_system_score_gemma":0.003320546,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003676258,"about_ca_topic_score_gemma":0.002052895,"domain_scores_codex":[0.994065,0.00239491,0.0002551861,0.0008922321,0.001869509,0.0005233228],"domain_scores_gemma":[0.9556433,0.03736877,0.002347472,0.001690443,0.002174469,0.0007756434],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00008267196,0.00004476897,0.0003499541,0.0001185088,0.00004612098,0.0000312561,0.00006980357,0.9275808,0.0005475566,0.06156372,0.0003396934,0.00922502],"study_design_scores_gemma":[0.000003741375,0.00002639396,0.00002461703,0.00001769848,0.000005568214,0.000007766376,0.000004903412,0.9742284,0.000243049,0.0253048,0.0001273805,0.000005658668],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.006360601,0.0004550281,0.9913431,0.0002550637,0.00003344291,0.00002645032,0.00003235727,0.0001884765,0.001305478],"genre_scores_gemma":[0.7997343,0.001261194,0.1945091,0.000401108,0.0001837455,0.0003439819,0.0002864185,0.0004472861,0.002832884],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01071265,"threshold_uncertainty_score":0.05665463,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0519210604033747,"score_gpt":0.2934808180844667,"score_spread":0.2415597576810921,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}