{"id":"W2912832158","doi":"10.1109/tcyb.2018.2890046","title":"Optimal Output Regulation of Linear Discrete-Time Systems With Unknown Dynamics Using Reinforcement Learning","year":2019,"lang":"en","type":"article","venue":"IEEE Transactions on Cybernetics","topic":"Adaptive Dynamic Programming Control","field":"Computer Science","cited_by":132,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Fundamental Research Funds for the Central Universities; Higher Education Discipline Innovation Project; Northeastern University; National Natural Science Foundation of China","keywords":"Reinforcement learning; Optimization problem; Optimal control; Mathematical optimization; Control theory (sociology); Computer science; Discrete time and continuous time; Discrete optimization; Noise (video); System dynamics; Control (management); Mathematics; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001248136,0.001004818,0.001138707,0.0003463698,0.0003685315,0.0009465899,0.0008513303,0.0009120809,0.0009280166],"category_scores_gemma":[0.002781096,0.0004210241,0.0004764307,0.0002851896,0.001289396,0.0006935576,0.000867344,0.001189473,0.0001387926],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008810821,"about_ca_system_score_gemma":0.00101608,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007227295,"about_ca_topic_score_gemma":0.00365142,"domain_scores_codex":[0.9994428,0.0001875714,0.00002570843,0.0001155219,0.0001395337,0.00008883641],"domain_scores_gemma":[0.9987803,0.0007924307,0.0001774246,0.00005639395,0.000151441,0.00004200041],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003732865,0.00003341221,0.0002211682,0.000040036,0.00002007607,0.00003956485,0.00003936256,0.9826458,0.000896015,0.004500087,0.0002174173,0.01130986],"study_design_scores_gemma":[0.000006569626,0.00001404764,0.00002535249,0.000002318456,0.00000228209,0.000002882994,0.000001895652,0.9987404,0.0001421611,0.0009932074,0.00006698896,0.000001936651],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02270144,0.0002964036,0.9739825,0.0001955541,0.00004311317,0.00002922588,0.00001396632,0.0002914231,0.002446434],"genre_scores_gemma":[0.969745,0.0001511656,0.02837825,0.00008098689,0.00003340365,0.00008026934,0.00003009314,0.00003002895,0.001470758],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007227295,"threshold_uncertainty_score":0.01437044,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.009236036983152505,"score_gpt":0.2210814403115666,"score_spread":0.2118454033284141,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}