{"id":"W4387269125","doi":"10.23919/epe23ecceeurope58414.2023.10264605","title":"Exploring the Effectiveness of Different State Spaces and Reward Functions in Reinforcement Learning-based Control of a DC/DC Buck Converter","year":2023,"lang":"en","type":"article","venue":"","topic":"Microgrid Control and Optimization","field":"Engineering","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Reinforcement learning; Converters; Buck converter; Computer science; Reliability (semiconductor); Stability (learning theory); Power electronics; Control (management); State (computer science); Control theory (sociology); Power (physics); State space; Artificial intelligence; Mathematics; Physics; Algorithm; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002602453,0.00008779068,0.0001845541,0.0001095454,0.00002272504,0.00001107647,0.00003788054,0.0000165812,0.00001681095],"category_scores_gemma":[0.00002533713,0.00005968487,0.00002964132,0.0001602154,0.0000246993,0.00006393543,0.00001052831,0.00007606486,0.000002607649],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002000658,"about_ca_system_score_gemma":0.000006151683,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007618052,"about_ca_topic_score_gemma":0.00004875245,"domain_scores_codex":[0.9994566,0.00007818971,0.0001808718,0.00007996634,0.00008198275,0.0001223749],"domain_scores_gemma":[0.9995229,0.0003089254,0.00002990257,0.00008755543,0.00003075652,0.000019955],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001255682,0.000007463186,0.008600404,0.0001905985,0.00004779529,4.862618e-7,0.0002272507,0.9784446,0.01068753,0.00004230993,0.00000631427,0.001619622],"study_design_scores_gemma":[0.001772681,0.0001075915,0.05969157,0.00009125818,0.00002106819,1.607058e-7,0.0001630595,0.9250442,0.01294814,0.00001635901,0.00006800811,0.00007590023],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8628803,0.00008213854,0.1363433,0.00004435085,0.0001096558,0.0003194802,0.000002793417,0.00008553474,0.000132361],"genre_scores_gemma":[0.999653,0.0001432123,0.000006209809,0.00000579543,0.000006363929,0.0001030544,0.000008108589,0.0000109566,0.00006326332],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1367727,"threshold_uncertainty_score":0.2433878,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01321741693769975,"score_gpt":0.1889584888631335,"score_spread":0.1757410719254337,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}