{"id":"W4404867262","doi":"10.1038/s41598-024-81170-y","title":"De-identification is not enough: a comparison between de-identified and synthetic clinical notes","year":2024,"lang":"en","type":"article","venue":"Scientific Reports","topic":"Privacy-Preserving Technologies in Data","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Manitoba","funders":"National Human Genome Research Institute; U.S. National Library of Medicine; National Institute on Aging; Natural Sciences and Engineering Research Council of Canada; University of Texas Health Science Center at Houston; University of Manitoba; National Cancer Institute; National Science Foundation; National Institutes of Health; Cancer Prevention and Research Institute of Texas","keywords":"Computer science; Identification (biology); Inference; Synthetic data; Task (project management); Machine learning; Artificial intelligence; Generative model; Generative grammar; Safeguarding; Data mining; Data science; Medicine","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01286867,0.0005692852,0.0005951886,0.0009064232,0.0006347613,0.002227103,0.001407112,0.001585869,0.001741702],"category_scores_gemma":[0.06183525,0.000272447,0.0006222494,0.0008797248,0.00162147,0.002459524,0.002313145,0.001690135,0.0006421015],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001132642,"about_ca_system_score_gemma":0.001353874,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001150187,"about_ca_topic_score_gemma":0.001025493,"domain_scores_codex":[0.9905663,0.005263203,0.0008211874,0.001188651,0.001872912,0.0002876605],"domain_scores_gemma":[0.929117,0.05010737,0.00295363,0.01388603,0.003275993,0.0006599571],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.006816394,0.001397446,0.0375724,0.001752684,0.0006201508,0.001847167,0.002244829,0.4907259,0.01875663,0.07815079,0.01760957,0.3425061],"study_design_scores_gemma":[0.0003002057,0.001220869,0.009107979,0.0003415307,0.000109405,0.003007993,0.001382765,0.8764102,0.03229553,0.05451637,0.0211494,0.0001577329],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5814947,0.003472194,0.3936625,0.003831062,0.0007042654,0.0007353659,0.004045577,0.001929366,0.01012494],"genre_scores_gemma":[0.8982973,0.0005822823,0.09349163,0.0006123947,0.00009363966,0.0001967531,0.005074469,0.0001259447,0.001525619],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01286867,"threshold_uncertainty_score":0.06805688,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0863253993477095,"score_gpt":0.373935892853746,"score_spread":0.2876104935060365,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}