{"id":"W2484962737","doi":"10.1016/j.jbi.2016.07.015","title":"A unified framework for evaluating the risk of re-identification of text de-identification tools","year":2016,"lang":"en","type":"article","venue":"Journal of Biomedical Informatics","topic":"Electronic Health Records Systems","field":"Health Professions","cited_by":28,"is_retracted":false,"has_abstract":true,"ca_institutions":"Children's Hospital of Eastern Ontario; Privacy Analytics (Canada); University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs; University of Ottawa","keywords":"Identification (biology); Identifier; Computer science; Context (archaeology); Data mining; Set (abstract data type); Information retrieval","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1157001,0.003340968,0.002642058,0.01993566,0.00212527,0.0101391,0.003608212,0.003181827,0.001234716],"category_scores_gemma":[0.2527611,0.0008977807,0.002688888,0.006363197,0.00332283,0.00816489,0.00592503,0.002923401,0.0004149871],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006799032,"about_ca_system_score_gemma":0.005221884,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01116093,"about_ca_topic_score_gemma":0.007930821,"domain_scores_codex":[0.8670694,0.065402,0.01544198,0.01061623,0.03899762,0.002472701],"domain_scores_gemma":[0.6958768,0.2015135,0.03801108,0.01696964,0.04533566,0.002293475],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001800145,0.001245937,0.1714788,0.002317495,0.002657286,0.0005149776,0.00514844,0.2408331,0.01293327,0.04590697,0.007588726,0.5075749],"study_design_scores_gemma":[0.0001522836,0.003125564,0.06942666,0.000816158,0.0006670165,0.00105301,0.001817943,0.873886,0.01120316,0.03015903,0.007198607,0.0004945553],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1360396,0.002673799,0.8474319,0.001359744,0.0001272773,0.002323536,0.001682143,0.00242114,0.005941014],"genre_scores_gemma":[0.5366108,0.0003293452,0.4590988,0.0001876776,0.00008557091,0.001736289,0.001198996,0.0001756459,0.0005768591],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.1157001,"threshold_uncertainty_score":0.6118879,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1309630934617878,"score_gpt":0.4942414625981897,"score_spread":0.3632783691364019,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}