{"id":"W3170371439","doi":"10.1609/aaai.v36i7.20758","title":"Control-Oriented Model-Based Reinforcement Learning with Implicit Differentiation","year":2022,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian Institute for Advanced Research; Vector Institute; University of Toronto; Université de Montréal","funders":"Compute Canada","keywords":"Reinforcement learning; Bellman equation; Computer science; Context (archaeology); Function (biology); Task (project management); Class (philosophy); Value (mathematics); Likelihood function; Maximum likelihood; Reinforcement; Control (management); Artificial intelligence; Mathematical optimization; Machine learning; Estimation theory; Mathematics; Statistics; Economics; Algorithm","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002428796,0.001122132,0.0012609,0.0003722899,0.0003767336,0.001307743,0.001782962,0.001445082,0.002204542],"category_scores_gemma":[0.007099803,0.0005770535,0.0005074397,0.0003746897,0.001743744,0.001654649,0.002072338,0.002477608,0.0003989725],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001220791,"about_ca_system_score_gemma":0.001560108,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00273188,"about_ca_topic_score_gemma":0.00244148,"domain_scores_codex":[0.9990988,0.0003620143,0.00004833284,0.0001545251,0.0002294897,0.0001068423],"domain_scores_gemma":[0.9969928,0.001892639,0.0003550711,0.0002750686,0.0003333686,0.0001509991],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005634649,0.00005302559,0.0002734507,0.00004463925,0.00002137648,0.00004831324,0.00005910481,0.9586526,0.0009495644,0.0245665,0.0002773982,0.01499763],"study_design_scores_gemma":[0.000007685009,0.00001656059,0.00001289933,0.000002881046,0.00000212588,0.000004368246,0.000001128098,0.994749,0.000164729,0.004964666,0.00007170239,0.000002221924],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01518968,0.0001006874,0.9822948,0.0001934835,0.00002084894,0.00003786975,0.0000150797,0.0002697975,0.001877714],"genre_scores_gemma":[0.9021251,0.00008327632,0.09509165,0.0001568371,0.00002327066,0.000136935,0.00003976133,0.00006596928,0.002277312],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00273188,"threshold_uncertainty_score":0.01284486,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03202434377017425,"score_gpt":0.2550913103624891,"score_spread":0.2230669665923148,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}