{"id":"W4414128997","doi":"10.1007/978-3-032-04558-4_8","title":"Learning to Optimize Entropy in the Soft Actor-Critic","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Reinforcement learning; Benchmarking; Regularization (linguistics); Entropy (arrow of time); Artificial neural network; Simulated annealing; Source code","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication","open_science"],"consensus_categories":[],"category_scores_codex":[0.001607645,0.0005568492,0.0005371438,0.001199129,0.0003388392,0.00118571,0.006637934,0.0002765206,0.00002348629],"category_scores_gemma":[0.0007167882,0.0004484074,0.0001388399,0.00131402,0.0003691297,0.0004762807,0.002038803,0.001943699,0.00008268622],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004236271,"about_ca_system_score_gemma":0.0006077435,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002333096,"about_ca_topic_score_gemma":0.00002223381,"domain_scores_codex":[0.9954804,0.0001301616,0.0006364762,0.001422692,0.001390277,0.0009399636],"domain_scores_gemma":[0.9961615,0.001726131,0.0002035996,0.001578957,0.0001915715,0.0001382783],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000003646909,0.00000972859,0.0000684187,0.00002668438,0.000005210667,0.00006996774,0.002032591,0.8824871,0.00002280052,0.01412354,0.00005888693,0.1010915],"study_design_scores_gemma":[0.0002492059,0.0002524234,0.0001147761,0.0005959191,0.000007519996,0.00002683913,0.000001583399,0.9809591,0.0000957485,0.01027492,0.006862025,0.0005599125],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00002627181,0.0001231734,0.9886744,0.003267117,0.001520007,0.0006090696,7.089321e-7,0.0001486382,0.005630631],"genre_scores_gemma":[0.1992876,0.0000699934,0.7817931,0.01252679,0.0006280174,0.00004462047,0.000008455723,0.00005958553,0.005581906],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.2068813,"threshold_uncertainty_score":0.9998512,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01357638618321766,"score_gpt":0.2525143598403673,"score_spread":0.2389379736571496,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}