{"id":"W2128812357","doi":"","title":"Error Propagation for Approximate Policy and Value Iteration","year":2010,"lang":"en","type":"preprint","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":135,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Value (mathematics); Computer science; Mathematical optimization; Algorithm; Mathematics; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007904045,0.001498975,0.001764087,0.001214075,0.0007377123,0.002273334,0.002082408,0.002689722,0.004207565],"category_scores_gemma":[0.04348683,0.000785767,0.0007874119,0.001217607,0.003749675,0.003928559,0.003782235,0.004313414,0.0007254895],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003285028,"about_ca_system_score_gemma":0.00252594,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00515839,"about_ca_topic_score_gemma":0.00335247,"domain_scores_codex":[0.9957072,0.001702837,0.0001874487,0.0005658721,0.001428864,0.0004077978],"domain_scores_gemma":[0.9775557,0.01726518,0.001505102,0.001344682,0.001930745,0.000398596],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000216248,0.00005781836,0.0006262692,0.0001431301,0.00005135576,0.0000601471,0.0001413998,0.7781641,0.001280987,0.1763267,0.001052212,0.04187969],"study_design_scores_gemma":[0.000007244047,0.00002177453,0.00003587197,0.00001198061,0.000003902634,0.000007196233,0.000004281988,0.9692652,0.0004684213,0.02993342,0.0002350444,0.00000567531],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.007718309,0.0002617565,0.9890178,0.0003311021,0.00004373981,0.00003156481,0.00002149304,0.0001838577,0.002390293],"genre_scores_gemma":[0.6406959,0.0005838335,0.3483793,0.0004550149,0.000150429,0.0003497949,0.0001540962,0.0003082992,0.008923279],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007904045,"threshold_uncertainty_score":0.04180104,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01625520364976859,"score_gpt":0.2649227238120397,"score_spread":0.2486675201622711,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}