{"id":"W2137486584","doi":"10.5555/1402383.1402439","title":"Sigma point policy iteration","year":2008,"lang":"en","type":"article","venue":"","topic":"Model Reduction and Neural Networks","field":"Physics and Astronomy","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Iterated function; Sigma; Bellman equation; Fixed point; Mathematical optimization; Temporal difference learning; Artificial intelligence; Algorithm; Applied mathematics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001194018,0.0008590647,0.001374436,0.0004891945,0.0004651426,0.0009817306,0.0009824487,0.001282366,0.006002059],"category_scores_gemma":[0.004661995,0.000538807,0.0006361551,0.0004860494,0.001078954,0.0008774687,0.001467803,0.001391125,0.0012347],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008590631,"about_ca_system_score_gemma":0.001626882,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003490619,"about_ca_topic_score_gemma":0.002835895,"domain_scores_codex":[0.999495,0.0001640251,0.00003309285,0.00009183179,0.0001510551,0.00006487532],"domain_scores_gemma":[0.9984095,0.001028427,0.00009686637,0.0001200151,0.0002859314,0.000059351],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001651452,0.00009490603,0.0009652105,0.0001700315,0.0000634208,0.00008208343,0.0001657723,0.7687779,0.002407738,0.0553897,0.002673349,0.1690447],"study_design_scores_gemma":[0.00001361089,0.00003046133,0.00003249395,0.00001132831,0.00000426668,0.00001301906,0.000009197598,0.9887767,0.0006815499,0.009562554,0.0008597032,0.00000501729],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.006622158,0.000144934,0.9897703,0.00009782986,0.00005065743,0.00006066776,0.00002121026,0.0003712981,0.00286089],"genre_scores_gemma":[0.4455262,0.0002932521,0.5427109,0.0003268466,0.00006515002,0.0006401631,0.0002188124,0.000302582,0.009916065],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006002059,"threshold_uncertainty_score":0.0200789,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0175388229952415,"score_gpt":0.2482758482641924,"score_spread":0.2307370252689509,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}