{"id":"W7106845376","doi":"10.48448/338a-5431","title":"RLHF Algorithms Ranked: An Extensive Evaluation Across Diverse Tasks, Rewards, and Hyperparameters","year":2025,"lang":"","type":"other","venue":"Open MIND","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Hyperparameter; Benchmarking; Benchmark (surveying); Helpfulness; Inefficiency; Reinforcement learning; Pruning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007848623,0.003324142,0.001736864,0.002865396,0.001488155,0.00195971,0.002892436,0.00351024,0.007610122],"category_scores_gemma":[0.02799437,0.0006208313,0.001495925,0.002263783,0.001097933,0.002959169,0.001891851,0.00377562,0.004791316],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002629872,"about_ca_system_score_gemma":0.002815105,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01701833,"about_ca_topic_score_gemma":0.02818595,"domain_scores_codex":[0.9942104,0.002309793,0.000584234,0.001057583,0.001398722,0.0004393358],"domain_scores_gemma":[0.9856205,0.009253458,0.0004827138,0.001931551,0.002281036,0.0004308039],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001407985,0.001334488,0.005378033,0.002604844,0.0006939345,0.0002926473,0.0002990797,0.2446391,0.003795484,0.00387723,0.1306516,0.6050256],"study_design_scores_gemma":[0.0009604527,0.001690975,0.005227105,0.0006112023,0.0003639506,0.0003716028,0.0006357562,0.9140514,0.02006113,0.008708138,0.04713276,0.0001854913],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4542631,0.04648157,0.2489437,0.006223988,0.003765971,0.003168577,0.02504007,0.1096731,0.10244],"genre_scores_gemma":[0.5509663,0.005215908,0.3746643,0.002007732,0.0003688028,0.001538171,0.04105185,0.00645909,0.01772786],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01701833,"threshold_uncertainty_score":0.04150796,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09332913924956755,"score_gpt":0.4097889817312186,"score_spread":0.3164598424816511,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}