{"id":"W4411120102","doi":"10.18653/v1/2025.naacl-long.187","title":"CluSanT: Differentially Private and Semantically Coherent Text Sanitization","year":2025,"lang":"en","type":"article","venue":"","topic":"Digital and Cyber Forensics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Victoria","funders":"","keywords":"Computer science; Natural language processing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004023965,0.001158042,0.001540183,0.001568049,0.001994014,0.002907769,0.002872417,0.002307425,0.009975325],"category_scores_gemma":[0.0126296,0.0006185978,0.001090006,0.001319201,0.002651666,0.007143465,0.008281735,0.003231027,0.007745646],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001313714,"about_ca_system_score_gemma":0.002330374,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009163413,"about_ca_topic_score_gemma":0.001694369,"domain_scores_codex":[0.9951755,0.001331865,0.0003022411,0.0007021385,0.001990707,0.0004975727],"domain_scores_gemma":[0.9905977,0.00277352,0.0004911486,0.004771698,0.00107825,0.0002877509],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004100716,0.0006228625,0.002165713,0.0005364326,0.0003004463,0.001008971,0.0008011113,0.03704821,0.05028855,0.2282414,0.1920438,0.4828419],"study_design_scores_gemma":[0.000558589,0.0004315741,0.0009215638,0.00008345226,0.000131875,0.0008278494,0.0003554477,0.5099856,0.06555291,0.3612738,0.05976887,0.0001084095],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04995015,0.001566238,0.8772976,0.005096999,0.001221335,0.0006419513,0.003232514,0.04315158,0.01784162],"genre_scores_gemma":[0.5546666,0.0008083634,0.393155,0.001893097,0.0009193519,0.0006360129,0.008975705,0.003162613,0.03578328],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009975325,"threshold_uncertainty_score":0.03337073,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.004422774677562558,"score_gpt":0.2056662444107658,"score_spread":0.2012434697332033,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}