{"id":"W6979368922","doi":"","title":"Difficulty-Based Preference Data Selection by DPO Implicit Reward Gap","year":2025,"lang":"en","type":"article","venue":"ArXiv.org","topic":"Recommender Systems and Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Bundesministerium für Bildung und Forschung; Government of Canada; Canadian Institute for Advanced Research","keywords":"Preference; Selection (genetic algorithm); Preference learning; Reinforcement learning; Model selection; Revealed preference","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003031811,0.0001699816,0.0001973906,0.00009760771,0.0001688122,0.0001534946,0.001769483,0.0001014224,0.00001026981],"category_scores_gemma":[0.00004101564,0.0001475146,0.00004221035,0.0005508313,0.00002136425,0.0004767214,0.0005177128,0.0001850319,0.00003957629],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006445318,"about_ca_system_score_gemma":0.0000994111,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003774576,"about_ca_topic_score_gemma":0.00005089631,"domain_scores_codex":[0.9984792,0.00009548638,0.0002875752,0.0006807618,0.0001667616,0.0002901405],"domain_scores_gemma":[0.9982855,0.00007651918,0.0001052276,0.001385519,0.00008256239,0.00006461642],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00001176261,0.0002502215,0.3481092,0.0001098531,0.00006886516,0.000004229591,0.0001201422,0.000007982985,0.02824692,0.007904782,0.5884826,0.02668346],"study_design_scores_gemma":[0.001125366,0.0003281469,0.2481339,0.0003689779,0.0000424201,0.00001569058,0.00003562674,0.03759395,0.1254039,0.002468544,0.5834932,0.000990283],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.131248,0.0001549972,0.8588505,0.003780407,0.0004645835,0.0003598771,0.00002164853,0.000884897,0.004235137],"genre_scores_gemma":[0.9915174,0.00001725238,0.006369706,0.0008919226,0.00005858831,0.00005312398,0.00005275979,0.000009016603,0.001030203],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8602695,"threshold_uncertainty_score":0.6015469,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1030329214442541,"score_gpt":0.3095350622575327,"score_spread":0.2065021408132787,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}