{"id":"W7117631606","doi":"","title":"A Comedy of Estimators: On KL Regularization in RL Training of LLMs","year":2025,"lang":"","type":"article","venue":"ArXiv.org","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Lawrence Livermore National Laboratory; Fonds de recherche du Québec – Nature et technologies; Office of Science; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research; Laboratory Directed Research and Development; U.S. Department of Energy","keywords":"Regularization (linguistics); Estimator; Divergence (linguistics); Reinforcement learning; Asynchronous communication","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00943846,0.001750939,0.001707646,0.0008085342,0.0007855016,0.002270418,0.002456147,0.002596346,0.002723556],"category_scores_gemma":[0.06802106,0.001122817,0.000803848,0.0006182984,0.003224787,0.003879782,0.003622076,0.005533666,0.00103339],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001720812,"about_ca_system_score_gemma":0.002662237,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005417113,"about_ca_topic_score_gemma":0.006544407,"domain_scores_codex":[0.9945849,0.003246404,0.0002718827,0.0007236783,0.0008430046,0.0003302281],"domain_scores_gemma":[0.9727729,0.02218526,0.0009443149,0.002190716,0.001499937,0.0004068528],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000417045,0.0001817492,0.003609139,0.0002852861,0.0001419222,0.0001599274,0.0003381273,0.8220253,0.003727077,0.04768782,0.004158793,0.1172678],"study_design_scores_gemma":[0.00003607781,0.00006362858,0.0001649942,0.00005566114,0.00001252643,0.00002486378,0.00002605486,0.9801469,0.001770285,0.01703474,0.0006485833,0.00001556136],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02088855,0.0007906464,0.973912,0.0008397939,0.00007713833,0.00007172846,0.00004677798,0.001347312,0.002026051],"genre_scores_gemma":[0.64315,0.0006490426,0.350246,0.00145148,0.0001369423,0.0004151345,0.0002612246,0.001061723,0.002628356],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00943846,"threshold_uncertainty_score":0.04991591,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06209032547014128,"score_gpt":0.2915530558813591,"score_spread":0.2294627304112178,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}