{"id":"W2128619633","doi":"10.1007/978-3-540-30115-8_53","title":"Batch Reinforcement Learning with State Importance","year":2004,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; University of Alberta","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Classifier (UML); Machine learning; Process (computing); State (computer science); Learning classifier system; Q-learning; Quality (philosophy); Function (biology); Bellman equation; Mathematical optimization; Algorithm; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001366766,0.0008322301,0.00112288,0.000365501,0.0002622318,0.0005931002,0.001592734,0.0008964766,0.007298924],"category_scores_gemma":[0.003479426,0.0005295507,0.0004121542,0.0004411987,0.0007633112,0.001245698,0.001299173,0.001862694,0.001061122],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007020369,"about_ca_system_score_gemma":0.0008109084,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002108066,"about_ca_topic_score_gemma":0.002240103,"domain_scores_codex":[0.9995319,0.0001144778,0.00002541062,0.0001155989,0.0001570793,0.00005564382],"domain_scores_gemma":[0.9985626,0.0009588745,0.00006873401,0.0001732416,0.0001725164,0.00006405162],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004194579,0.0002265306,0.0004205245,0.0001739139,0.00008218733,0.00008095882,0.00005004515,0.619171,0.005814259,0.0421744,0.007141308,0.3242455],"study_design_scores_gemma":[0.00002533975,0.00004351665,0.00006057428,0.000004577035,0.000009102373,0.00001449425,0.000001556987,0.986206,0.001049893,0.01197498,0.0006046141,0.000005250054],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.006421003,0.000288606,0.9875476,0.0001254925,0.0001220384,0.00004824861,0.00003740244,0.000722857,0.004686582],"genre_scores_gemma":[0.6634272,0.0004067112,0.3116656,0.0002666588,0.0002181672,0.0003451341,0.0002503468,0.0002910057,0.02312903],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007298924,"threshold_uncertainty_score":0.02441728,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01192789426754613,"score_gpt":0.227359115857927,"score_spread":0.2154312215903808,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}