{"id":"W4387269125","doi":"10.23919/epe23ecceeurope58414.2023.10264605","title":"Exploring the Effectiveness of Different State Spaces and Reward Functions in Reinforcement Learning-based Control of a DC/DC Buck Converter","year":2023,"lang":"en","type":"article","venue":"","topic":"Microgrid Control and Optimization","field":"Engineering","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Reinforcement learning; Converters; Buck converter; Computer science; Reliability (semiconductor); Stability (learning theory); Power electronics; Control (management); State (computer science); Control theory (sociology); Power (physics); State space; Artificial intelligence; Mathematics; Physics; Algorithm; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001636869,0.0005259662,0.0004539458,0.0002889569,0.0002275205,0.000685869,0.0004069576,0.0005070046,0.0006227727],"category_scores_gemma":[0.003824222,0.0002211223,0.000228467,0.0001480573,0.0005914702,0.000686663,0.00050372,0.0005912759,0.00006275422],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005630752,"about_ca_system_score_gemma":0.000659415,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004046736,"about_ca_topic_score_gemma":0.002678922,"domain_scores_codex":[0.9997087,0.0001398464,0.00001599407,0.00003793951,0.00005353688,0.00004400741],"domain_scores_gemma":[0.9982838,0.001290383,0.0001235291,0.00006306113,0.0001699982,0.00006922166],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001861523,0.0001047793,0.000614254,0.0000473267,0.00001958031,0.00002628754,0.00002661809,0.9837494,0.002131422,0.001745946,0.00007140353,0.01127681],"study_design_scores_gemma":[0.00001196986,0.0000912414,0.0001432957,0.000003930999,0.000005679098,0.00000364569,0.000005753587,0.9984031,0.0008292869,0.0004630484,0.00003569175,0.000003356403],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7231372,0.0006180569,0.2687562,0.0004861441,0.00003552943,0.0001146999,0.00003840332,0.0002477237,0.006565922],"genre_scores_gemma":[0.993053,0.00004513844,0.006551856,0.00002205214,0.00000181099,0.00001836567,0.000006538959,0.000004418354,0.0002967423],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004046736,"threshold_uncertainty_score":0.008656681,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01321741693769975,"score_gpt":0.1889584888631335,"score_spread":0.1757410719254337,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}