{"id":"W4415021619","doi":"10.1016/j.arcontrol.2025.101027","title":"A view on learning robust goal-conditioned value functions: Interplay between RL and MPC","year":2025,"lang":"en","type":"article","venue":"Annual Reviews in Control","topic":"Advanced Control Systems Optimization","field":"Engineering","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"Honeywell (Canada); University of British Columbia","funders":"Fusion Energy Sciences; Office of Science; U.S. Department of Energy","keywords":"Robustness (evolution); Reinforcement learning; Markov decision process; Bellman equation; Model predictive control; Function (biology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00299838,0.001911061,0.002189136,0.00119886,0.0005297286,0.005427852,0.003490028,0.004972012,0.003877573],"category_scores_gemma":[0.005902758,0.0009457339,0.001646198,0.001445162,0.005978332,0.008246652,0.00235674,0.007965856,0.001232898],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001494871,"about_ca_system_score_gemma":0.0009859646,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002066969,"about_ca_topic_score_gemma":0.0008601189,"domain_scores_codex":[0.9980283,0.000731806,0.00009717398,0.0004636504,0.0005677254,0.0001112037],"domain_scores_gemma":[0.9961419,0.002912603,0.0001851639,0.0003752196,0.0002978457,0.00008733344],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00003686241,0.00003083034,0.00009777486,0.0003835936,0.00007172271,0.00004439832,0.00008120437,0.07353161,0.0009723424,0.8911551,0.002456801,0.03113778],"study_design_scores_gemma":[0.00003380382,0.0001035066,0.0001568629,0.0001876587,0.0000397713,0.00005431678,0.00003927704,0.2129316,0.0009044262,0.7667942,0.01869987,0.00005474644],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"review","genre_scores_codex":[0.00155067,0.02012944,0.9537946,0.005883823,0.0005904247,0.00001796451,0.00007827456,0.0002151603,0.01773972],"genre_scores_gemma":[0.5250383,0.07840133,0.35314,0.008022963,0.01319795,0.0004241705,0.0004064607,0.0006058528,0.02076303],"genre_candidate":"review","genre_consensus":null,"teacher_disagreement_score":0.005427852,"threshold_uncertainty_score":0.01585716,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.008009256037607566,"score_gpt":0.2736690119654247,"score_spread":0.2656597559278172,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}