{"id":"W7096680464","doi":"","title":"Reinforcement Learning for Factored Markov Decision Processes","year":2002,"lang":"en","type":"article","venue":"TSpace","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Markov decision process; Inference; Partially observable Markov decision process; Representation (politics); Core (optical fiber); Action (physics); State (computer science); Markov process; Q-learning","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002591402,0.001136217,0.001608247,0.0006158372,0.0005625222,0.001115148,0.001404176,0.001332805,0.005463582],"category_scores_gemma":[0.01255358,0.000613101,0.001001449,0.0006771423,0.001915729,0.001788666,0.001311421,0.002666617,0.0005055949],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002707144,"about_ca_system_score_gemma":0.001879714,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01125246,"about_ca_topic_score_gemma":0.008777557,"domain_scores_codex":[0.9986995,0.0006113651,0.00006262337,0.0002759704,0.0002140944,0.0001364183],"domain_scores_gemma":[0.9932384,0.005686319,0.0003930158,0.0001792655,0.0003174167,0.0001855148],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006493871,0.00003078151,0.0004153666,0.00007384158,0.00003221503,0.00004820376,0.00005663672,0.9026009,0.0001878433,0.08191296,0.0007550878,0.01382118],"study_design_scores_gemma":[0.00001579828,0.00001068834,0.00003130778,0.000005598539,0.000003370266,0.000004673563,0.000003035587,0.9540313,0.00004012607,0.04561192,0.000238654,0.00000343686],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01096671,0.0005913621,0.9846295,0.0004564609,0.00005475805,0.00005118147,0.0001118087,0.0002493568,0.002888856],"genre_scores_gemma":[0.7997829,0.001202093,0.1911723,0.0002179748,0.0001410883,0.0004819486,0.0004103458,0.0001151541,0.006476288],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01125246,"threshold_uncertainty_score":0.02237391,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03809663181555732,"score_gpt":0.3046626435827796,"score_spread":0.2665660117672223,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}