{"id":"W4416035462","doi":"10.18653/v1/2025.emnlp-main.1430","title":"Improving Large Language Model Safety with Contrastive Representation Learning","year":2025,"lang":"en","type":"article","venue":"","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; National Supercomputing Centre Singapore; National Science Foundation; Centro Svizzero di Calcolo Scientifico; Bundesministerium für Bildung und Forschung; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung","keywords":"Representation (politics); Feature (linguistics); Natural language; Language identification; Language model; Quality (philosophy)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003449441,0.0001200147,0.0001472901,0.0001081578,0.0002758843,0.0001169164,0.0003819427,0.00004706953,0.00001629847],"category_scores_gemma":[0.0003180314,0.00009919272,0.00003281318,0.0004358765,0.00002704179,0.0005681282,0.0003104605,0.0003353584,0.000008327121],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006940074,"about_ca_system_score_gemma":0.0001115487,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001793653,"about_ca_topic_score_gemma":0.00003902907,"domain_scores_codex":[0.9988754,0.0001049871,0.0001649033,0.0003951289,0.0001925964,0.0002669594],"domain_scores_gemma":[0.9992795,0.0002055639,0.00009302241,0.0003036258,0.00007990121,0.00003836218],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005293444,0.00002421786,0.004566447,0.00001380495,0.00003126366,0.00001805386,0.001934952,0.7545225,0.0009417362,0.1933247,0.00005485237,0.04451459],"study_design_scores_gemma":[0.0007585178,0.0000245311,0.000837692,0.0000216708,0.000009440026,0.000002197659,0.0007597367,0.9963124,0.0006640319,0.0004041143,0.00008562711,0.0001200249],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.003471911,0.00002304818,0.9650943,0.0004128997,0.0000860444,0.0001455821,5.606331e-7,0.0003833602,0.03038226],"genre_scores_gemma":[0.8112033,0.000001215016,0.1831743,0.0002956686,0.00002344453,0.000007771828,0.000004449449,0.000007445817,0.005282349],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8077314,"threshold_uncertainty_score":0.4044962,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.005280031098914127,"score_gpt":0.2679396322120302,"score_spread":0.262659601113116,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}