{"id":"W2135097260","doi":"10.1007/3-540-44886-1_26","title":"Model-Based Least-Squares Policy Evaluation","year":2003,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Estimator; Computation; Least-squares function approximation; Basis (linear algebra); Algorithm; Set (abstract data type); Markov decision process; Mathematical optimization; Markov process; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00447492,0.001691479,0.003368322,0.001286046,0.0009250512,0.002042838,0.002237631,0.003436327,0.008252558],"category_scores_gemma":[0.01665684,0.001396728,0.00101849,0.0008539769,0.001066229,0.001897545,0.00187127,0.002393455,0.001796336],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001706761,"about_ca_system_score_gemma":0.003649991,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01230198,"about_ca_topic_score_gemma":0.00945727,"domain_scores_codex":[0.997829,0.001071784,0.0001241838,0.0003072908,0.0004478599,0.000219822],"domain_scores_gemma":[0.9909437,0.006485486,0.0003165701,0.000435438,0.001583503,0.0002352289],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002642635,0.00008439445,0.0003602353,0.00008455896,0.00005561459,0.00003218612,0.00002168156,0.9292284,0.0005290694,0.002878066,0.001748794,0.0647127],"study_design_scores_gemma":[0.00001216895,0.00001309019,0.0000202146,0.000003425568,0.000003933445,0.00000366173,0.000002343027,0.9988635,0.0002174491,0.0007959384,0.00006168731,0.000002638291],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01180314,0.0002590386,0.9831547,0.0002591769,0.0001011652,0.00009034839,0.00007094388,0.001728303,0.002533271],"genre_scores_gemma":[0.6062883,0.0001863423,0.3857311,0.0003383873,0.0001038313,0.0003695801,0.0004622249,0.0006072606,0.005912937],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01230198,"threshold_uncertainty_score":0.02760756,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03695932473739669,"score_gpt":0.2924345026014347,"score_spread":0.255475177864038,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}