{"id":"W4400375395","doi":"10.48550/arxiv.2407.02551","title":"Breach By A Thousand Leaks: Unsafe Information Leakage in `Safe' AI Responses","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Government of Canada; Canadian Institute for Advanced Research; Alfred P. Sloan Foundation","keywords":"Leakage (economics); Computer security; Sense (electronics); Computer science; Business; Engineering; Electrical engineering; Economics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0008111293,0.0004324935,0.0004215935,0.000926951,0.0001623541,0.0004216814,0.00206309,0.000462165,0.00003517904],"category_scores_gemma":[0.0001983963,0.0005032948,0.0001812246,0.001299364,0.0001225294,0.001428379,0.004248801,0.002244406,0.0003581779],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004898256,"about_ca_system_score_gemma":0.0004349412,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005962301,"about_ca_topic_score_gemma":0.00007226269,"domain_scores_codex":[0.9974897,0.0004496237,0.00040176,0.0009798047,0.0001975049,0.0004816005],"domain_scores_gemma":[0.9980453,0.0002835568,0.0002789593,0.001138794,0.0001143786,0.0001389725],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002347875,0.00009516151,0.003109125,0.0004158341,0.0001091923,0.0007656988,0.002912899,0.7950374,0.00004091885,0.1835827,0.005238677,0.008457649],"study_design_scores_gemma":[0.0007944917,0.00007541125,0.0009737992,0.0003712416,0.00005721249,0.00001617298,0.0002226825,0.9374683,0.00006300591,0.04703841,0.0121259,0.0007932917],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1151057,0.0002408419,0.8734754,0.001189071,0.001163543,0.0004557865,0.00004242319,0.0006213519,0.007705872],"genre_scores_gemma":[0.9946281,0.0001748407,0.001658942,0.0003273992,0.00006892101,0.000002308918,0.00004671174,0.00002672438,0.003066044],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8795224,"threshold_uncertainty_score":0.9997419,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02937721900323098,"score_gpt":0.2062563259642631,"score_spread":0.1768791069610322,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}