{"id":"W3021380283","doi":"10.1093/jamia/ocaa038","title":"Using word embeddings to improve the privacy of clinical notes","year":2020,"lang":"en","type":"article","venue":"Journal of the American Medical Informatics Association","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"St. Michael's Hospital; Institute for Clinical Evaluative Sciences; Vector Institute; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Readability; Security token; Word embedding; Recall; Artificial intelligence; Word (group theory); Natural language processing; Precision and recall; Machine learning; Information retrieval; Embedding; Data science; Computer security","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005229417,0.001223216,0.0008064532,0.0017397,0.0009644548,0.002787851,0.001146444,0.001510708,0.002670257],"category_scores_gemma":[0.0418205,0.0005048956,0.0008530449,0.001834399,0.00192569,0.008049322,0.00431833,0.002489654,0.002106786],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007993074,"about_ca_system_score_gemma":0.001664542,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001008344,"about_ca_topic_score_gemma":0.0009653953,"domain_scores_codex":[0.9916028,0.003658992,0.001012199,0.001591656,0.001779197,0.0003551018],"domain_scores_gemma":[0.9765589,0.009087411,0.003300481,0.008491111,0.002217431,0.000344558],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001483737,0.0004957345,0.01703292,0.0007648452,0.0002606522,0.0006539859,0.002175314,0.07903878,0.02458541,0.06259383,0.01275261,0.7981623],"study_design_scores_gemma":[0.000197619,0.001026996,0.006092629,0.0003862726,0.0002785241,0.002562279,0.001806573,0.6648151,0.06207597,0.2149621,0.04558839,0.0002075771],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1069282,0.001274437,0.8811201,0.002154703,0.0004728051,0.0002296108,0.001213469,0.002418744,0.004187917],"genre_scores_gemma":[0.6649458,0.0009583417,0.3238395,0.0007787675,0.0004355494,0.0002259552,0.003118016,0.0004007285,0.005297371],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005229417,"threshold_uncertainty_score":0.02765614,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05535004671577025,"score_gpt":0.4073642148405306,"score_spread":0.3520141681247604,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}