{"id":"W4404782689","doi":"10.18653/v1/2024.emnlp-main.626","title":"ORPO: Monolithic Preference Optimization without Reference Model","year":2024,"lang":"en","type":"article","venue":"","topic":"Multi-Criteria Decision Making","field":"Decision Sciences","cited_by":76,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea; Korea Advanced Institute of Science and Technology","keywords":"Preference; Computer science; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002211317,0.001151509,0.001059522,0.0006003295,0.0005174386,0.001653413,0.002295779,0.0009752728,0.01195398],"category_scores_gemma":[0.009720072,0.0004519596,0.00107417,0.0007834529,0.0005431724,0.002367243,0.002138522,0.002004412,0.00330144],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008172153,"about_ca_system_score_gemma":0.0019135,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003925769,"about_ca_topic_score_gemma":0.007249124,"domain_scores_codex":[0.9983058,0.0005416826,0.00009188195,0.0004111507,0.000444448,0.0002050415],"domain_scores_gemma":[0.9977691,0.001047317,0.0001301207,0.0005577483,0.0003902076,0.0001054503],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006862542,0.0003793786,0.003191878,0.000348155,0.0001589001,0.0001465743,0.0001636749,0.2621579,0.009864938,0.03012367,0.01748936,0.6752893],"study_design_scores_gemma":[0.00006602239,0.0001196093,0.0003030304,0.00001812137,0.00002383954,0.00004542541,0.00004347149,0.9751209,0.0034741,0.01687073,0.003895842,0.00001895622],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0268316,0.0002782322,0.9551708,0.0002882504,0.000107221,0.0001863656,0.0003916521,0.008775313,0.007970543],"genre_scores_gemma":[0.3883216,0.0001469836,0.5999838,0.0007379609,0.00008891377,0.0004553151,0.001312215,0.002057211,0.006895974],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01195398,"threshold_uncertainty_score":0.03999007,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5224362640997373,"score_gpt":0.47926384642419,"score_spread":0.0431724176755473,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}