{"id":"W4400375395","doi":"10.48550/arxiv.2407.02551","title":"Breach By A Thousand Leaks: Unsafe Information Leakage in `Safe' AI Responses","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Government of Canada; Canadian Institute for Advanced Research; Alfred P. Sloan Foundation","keywords":"Leakage (economics); Computer security; Sense (electronics); Computer science; Business; Engineering; Electrical engineering; Economics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02165768,0.0008716073,0.001041508,0.001072287,0.001305241,0.004482776,0.002199499,0.003147062,0.004007841],"category_scores_gemma":[0.09933148,0.0005544907,0.001179578,0.0007188445,0.006346369,0.009724374,0.006719064,0.006122256,0.000969081],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002089313,"about_ca_system_score_gemma":0.002066623,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00117787,"about_ca_topic_score_gemma":0.001045206,"domain_scores_codex":[0.9718111,0.01633533,0.001167479,0.002432226,0.006869672,0.001384081],"domain_scores_gemma":[0.8944706,0.06875,0.004984587,0.02593277,0.004157417,0.001704673],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002772614,0.0007334508,0.01636289,0.000709939,0.0003878702,0.000753111,0.004454303,0.2436286,0.02571466,0.4635539,0.01194644,0.2289821],"study_design_scores_gemma":[0.00006733605,0.0004603298,0.001592045,0.0001413242,0.00009393234,0.0003471987,0.0006025253,0.5221058,0.02680806,0.4404846,0.007211882,0.00008491682],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1512351,0.0005373022,0.8272126,0.005381178,0.0001307437,0.0001972342,0.0003962146,0.003310597,0.01159913],"genre_scores_gemma":[0.9214912,0.0001367041,0.07479664,0.001013328,0.00006215789,0.0001125202,0.0002676797,0.0002762738,0.001843518],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02165768,"threshold_uncertainty_score":0.1145381,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02937721900323098,"score_gpt":0.2062563259642631,"score_spread":0.1768791069610322,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}