{"id":"W4416429290","doi":"10.1109/tnnls.2025.3626050","title":"Universal Stabilization for Maximum Entropy Optimization in Reinforcement Learning","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Neural Networks and Learning Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"China Scholarship Council; National Natural Science Foundation of China","keywords":"Entropy (arrow of time); Kullback–Leibler divergence; Reinforcement learning; Principle of maximum entropy; Upper and lower bounds; Monotonic function; Maximum entropy spectral estimation; Divergence (linguistics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002425568,0.001301921,0.001232971,0.0006599583,0.0006349146,0.001237265,0.001009749,0.00124082,0.002094595],"category_scores_gemma":[0.009910637,0.0006090451,0.0006394418,0.0004289735,0.002235478,0.001462811,0.00208064,0.002080881,0.0004819985],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001452046,"about_ca_system_score_gemma":0.00158744,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002439484,"about_ca_topic_score_gemma":0.001986654,"domain_scores_codex":[0.9990841,0.0003517355,0.00005425868,0.000217425,0.0001934904,0.0000988854],"domain_scores_gemma":[0.9969607,0.002154011,0.0003250647,0.0001758566,0.0002607667,0.0001235577],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006957455,0.0000394477,0.0005406508,0.00008354759,0.00003500423,0.00005526521,0.00008947573,0.9453074,0.001969675,0.0296436,0.0007480304,0.02141833],"study_design_scores_gemma":[0.000006894947,0.00002095698,0.00004494766,0.000009478508,0.000003576748,0.000007379688,0.00000427463,0.9880547,0.0004428542,0.01118017,0.0002194929,0.000005227216],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.009750555,0.0002219624,0.9874355,0.0001987352,0.00002689563,0.0000358988,0.00001679052,0.0003156197,0.001998045],"genre_scores_gemma":[0.8543645,0.0003112506,0.1413876,0.0003282705,0.00007478068,0.0002855744,0.00009270924,0.0002674569,0.002887796],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.002439484,"threshold_uncertainty_score":0.01282775,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01099109538066441,"score_gpt":0.2301446614918618,"score_spread":0.2191535661111974,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}