{"id":"W4402670301","doi":"10.18653/v1/2024.findings-acl.297","title":"Disentangling Length from Quality in Direct Preference Optimization","year":2024,"lang":"en","type":"article","venue":"","topic":"Rough Sets and Fuzzy Logic","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Office of Naval Research; Canadian Institute for Advanced Research","keywords":"Preference; Quality (philosophy); Computer science; Mathematics; Statistics; Physics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008193716,0.001449918,0.001622935,0.0007470172,0.0006183762,0.001646427,0.001928465,0.001995219,0.001858557],"category_scores_gemma":[0.03673175,0.0008398552,0.0007466947,0.0006707924,0.002153919,0.003863213,0.002666938,0.003263081,0.0004470601],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001423512,"about_ca_system_score_gemma":0.001924009,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002998856,"about_ca_topic_score_gemma":0.004021242,"domain_scores_codex":[0.9962962,0.002305661,0.000151524,0.000468219,0.000531499,0.000246836],"domain_scores_gemma":[0.9768575,0.01842614,0.00152248,0.001604578,0.0009336398,0.0006556459],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0003224003,0.0002603256,0.004432638,0.0002077279,0.0001281409,0.00009961972,0.000244231,0.8836044,0.002801373,0.01821539,0.001542342,0.08814146],"study_design_scores_gemma":[0.00002964715,0.00010469,0.0002374158,0.0000179416,0.00001218458,0.00001968609,0.00001868197,0.9844307,0.0005123662,0.01430783,0.0002958325,0.00001296878],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0918256,0.0008175616,0.9025261,0.001137935,0.00005690455,0.0001222795,0.00009335065,0.0007705397,0.002649742],"genre_scores_gemma":[0.8693803,0.0002306063,0.126923,0.0005078061,0.00008962675,0.0002460864,0.0001583024,0.0002659339,0.002198308],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008193716,"threshold_uncertainty_score":0.04333305,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05934964987187338,"score_gpt":0.2897378848251219,"score_spread":0.2303882349532485,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}