{"id":"W4398836342","doi":"10.48550/arxiv.2405.14577","title":"Representation Noising: A Defence Mechanism Against Harmful Finetuning","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Research Nova Scotia; Alliance de recherche numérique du Canada; Killam Trusts; Bundesministerium für Bildung und Forschung; Canadian Institute for Advanced Research","keywords":"Representation (politics); Fine-tuning; Computer science; Artificial intelligence; Political science; Physics; Law; Politics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01062543,0.001794373,0.001228595,0.001126092,0.001049292,0.003210547,0.004214445,0.00364699,0.003119728],"category_scores_gemma":[0.06314617,0.0009552845,0.001724364,0.0006476071,0.004639267,0.008114931,0.009364796,0.007555915,0.001553811],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001084275,"about_ca_system_score_gemma":0.002062496,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001119473,"about_ca_topic_score_gemma":0.001958417,"domain_scores_codex":[0.9893637,0.005123812,0.000512787,0.001721484,0.002477068,0.0008010929],"domain_scores_gemma":[0.9549478,0.01432557,0.003211211,0.02477985,0.002003372,0.0007323622],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001233313,0.0005629587,0.01098362,0.0004185314,0.000425523,0.0006276474,0.002630599,0.2602928,0.04491837,0.1943582,0.01740603,0.4661423],"study_design_scores_gemma":[0.00009437041,0.000321723,0.0006681046,0.0000958536,0.00009265541,0.0003503721,0.0001765851,0.8246236,0.02073061,0.1413038,0.01143783,0.0001043653],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.04266332,0.00028354,0.9427269,0.002059107,0.0001632347,0.0001620967,0.0001017594,0.008205624,0.003634601],"genre_scores_gemma":[0.7030601,0.0002042139,0.2842716,0.002075736,0.0002172231,0.0003078167,0.00038083,0.001747443,0.007734944],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01062543,"threshold_uncertainty_score":0.05619329,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3478307503312511,"score_gpt":0.2981831477082125,"score_spread":0.04964760262303858,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}