{"id":"W4315489024","doi":"10.1109/cdc51059.2022.9992565","title":"Thompson-Sampling Based Reinforcement Learning for Networked Control of Unknown Linear Systems","year":2022,"lang":"en","type":"article","venue":"2022 IEEE 61st Conference on Decision and Control (CDC)","topic":"Adaptive Dynamic Programming Control","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Linear-quadratic-Gaussian control; Reinforcement learning; Logarithm; Regret; Network packet; Bounded function; Control theory (sociology); Generalization; Controller (irrigation); Sampling (signal processing); Gaussian; Discrete mathematics; Mathematics; Computer science; Optimal control; Control (management); Mathematical optimization; Artificial intelligence; Mathematical analysis; Statistics; Telecommunications","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002691185,0.00105473,0.001776884,0.00043635,0.0004117309,0.000805938,0.001206625,0.001127569,0.001891902],"category_scores_gemma":[0.009905707,0.0004514746,0.0005272389,0.00049272,0.001913607,0.0009938774,0.001234623,0.001858536,0.0002279275],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002076686,"about_ca_system_score_gemma":0.001448711,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01200772,"about_ca_topic_score_gemma":0.007603865,"domain_scores_codex":[0.9988664,0.0005289462,0.00004686345,0.0001892004,0.0002377377,0.0001307048],"domain_scores_gemma":[0.9941339,0.004674593,0.0003749353,0.0002041465,0.000417302,0.0001951307],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006917918,0.0000291972,0.0004548732,0.00004038566,0.00002464113,0.00003572124,0.00003144338,0.9805854,0.0002773746,0.00997386,0.0003394683,0.008138421],"study_design_scores_gemma":[0.000005778068,0.0000140409,0.0000373475,0.000002354185,0.000001689335,0.000002150447,0.000001602788,0.9966109,0.00005082961,0.003212943,0.00005865307,0.000001724515],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03831405,0.0007781439,0.9566457,0.0006250141,0.00007470426,0.00005764924,0.00004640245,0.0003120167,0.003146353],"genre_scores_gemma":[0.9642838,0.0003294629,0.03212383,0.0002109598,0.00007806326,0.0001285338,0.00009762794,0.00006022214,0.002687505],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01200772,"threshold_uncertainty_score":0.02387565,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03500695118355131,"score_gpt":0.2775513208392075,"score_spread":0.2425443696556562,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}