{"id":"W3017059342","doi":"10.1049/iet-cta.2019.0397","title":"Integral reinforcement learning solutions for a synchronisation system with constrained policies","year":2020,"lang":"en","type":"article","venue":"IET Control Theory and Applications","topic":"Adaptive Dynamic Programming Control","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Reinforcement learning; Computer science; Control theory (sociology); Control engineering; Mathematical optimization; Artificial intelligence; Mathematics; Control (management); Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000701164,0.0006665305,0.000696756,0.0003529418,0.0003249522,0.0008297461,0.0006132422,0.0009320732,0.001833319],"category_scores_gemma":[0.001758015,0.000264194,0.0003578732,0.000303078,0.001085834,0.0005095612,0.00113282,0.0008552693,0.0001623262],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008871969,"about_ca_system_score_gemma":0.0009951954,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006656437,"about_ca_topic_score_gemma":0.003879043,"domain_scores_codex":[0.9997346,0.000075417,0.000012997,0.00006234272,0.00006327274,0.0000513286],"domain_scores_gemma":[0.9992742,0.0003814499,0.0001620961,0.00002717825,0.0001115219,0.00004358566],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003189778,0.00001538609,0.0001925782,0.00003416138,0.00001135932,0.00005990252,0.00005624295,0.9701245,0.001027832,0.02294272,0.0002428647,0.005260642],"study_design_scores_gemma":[0.000007614204,0.00001289189,0.00004105901,0.000002249847,0.00000198665,0.000004992146,0.000005324912,0.9965383,0.0001074834,0.003124087,0.0001517587,0.000002412582],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05948328,0.0002993206,0.9293725,0.0003797363,0.00005078112,0.00005727286,0.00005569858,0.0001619051,0.01013946],"genre_scores_gemma":[0.9727644,0.0001327272,0.0223827,0.00004403029,0.00001271777,0.00009878842,0.00003455647,0.00001701394,0.004512914],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006656437,"threshold_uncertainty_score":0.01323539,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01138973012840052,"score_gpt":0.2271804721164862,"score_spread":0.2157907419880857,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}