{"id":"W4404402193","doi":"10.48550/arxiv.2411.06824","title":"Combining Domain and Alignment Vectors to Achieve Better Knowledge-Safety Trade-offs in LLMs","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Digital Rights Management and Security","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Samsung; Alliance de recherche numérique du Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Domain (mathematical analysis); Computer science; Business; Risk analysis (engineering); Mathematics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004044167,0.001478838,0.0009308436,0.002132971,0.0006403792,0.002349967,0.001819501,0.001802608,0.004079356],"category_scores_gemma":[0.01791183,0.000638375,0.000943818,0.00164254,0.0009209088,0.005177455,0.003831808,0.002532754,0.002141346],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001365099,"about_ca_system_score_gemma":0.001936139,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00331779,"about_ca_topic_score_gemma":0.005155263,"domain_scores_codex":[0.9973789,0.0009674706,0.0001787902,0.0006012148,0.0006393157,0.0002342664],"domain_scores_gemma":[0.9933046,0.003248352,0.0004805645,0.001674059,0.0009779871,0.0003146126],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006776727,0.0006869038,0.01709536,0.0004845657,0.0002191893,0.0003671715,0.0009959035,0.4783824,0.02844561,0.01915723,0.01018023,0.4433078],"study_design_scores_gemma":[0.00004426927,0.0002069613,0.001458731,0.00007094545,0.00006274135,0.0001086489,0.0004114843,0.9463198,0.02121266,0.0223908,0.007679058,0.00003396476],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2541997,0.001021473,0.7181523,0.001039763,0.0001406058,0.0002062711,0.0008724888,0.0145787,0.009788689],"genre_scores_gemma":[0.6656865,0.0002905988,0.3254809,0.0004131382,0.00003664546,0.0001577277,0.002788044,0.00161561,0.003530812],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004079356,"threshold_uncertainty_score":0.02138782,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03512000690300871,"score_gpt":0.1879398007447607,"score_spread":0.152819793841752,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}