{"id":"W4403963877","doi":"10.48550/arxiv.2410.04350","title":"TIS-DPO: Token-level Importance Sampling for Direct Preference Optimization With Estimated Weights","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Institute for Catastrophic Loss Reduction","keywords":"Preference; Security token; Sampling (signal processing); Statistics; Computer science; Mathematics; Computer network; Telecommunications","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003455181,0.001614448,0.001838486,0.0009329871,0.0006125514,0.001039346,0.002183722,0.001454859,0.004658149],"category_scores_gemma":[0.0133182,0.0007262546,0.001161706,0.001293176,0.000899689,0.002172577,0.001848736,0.002761295,0.001115438],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001186892,"about_ca_system_score_gemma":0.002505361,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003616475,"about_ca_topic_score_gemma":0.00590761,"domain_scores_codex":[0.9978696,0.001016851,0.000141606,0.0004393893,0.0003191295,0.000213401],"domain_scores_gemma":[0.9953241,0.003218424,0.0002543642,0.0004745785,0.000502761,0.0002258533],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007979455,0.0005188924,0.004012502,0.0003735593,0.0001576985,0.0001626888,0.0001844312,0.588258,0.005404628,0.01795003,0.007949363,0.3742303],"study_design_scores_gemma":[0.0000312117,0.00006807454,0.0001341012,0.000006454607,0.000009034951,0.00001524948,0.00001428755,0.9919705,0.0006928821,0.006632832,0.0004185642,0.000006732679],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01865527,0.0002157526,0.9784681,0.0002017069,0.00006719417,0.000194346,0.0001748952,0.001135655,0.0008870871],"genre_scores_gemma":[0.455054,0.0001734364,0.5374133,0.0005850617,0.0001596956,0.0009516555,0.00146922,0.0004992933,0.003694323],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004658149,"threshold_uncertainty_score":0.01827294,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2090452637902154,"score_gpt":0.2221080567722163,"score_spread":0.01306279298200086,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}