{"id":"W4293574676","doi":"10.2196/38154","title":"An Efficient Method for Deidentifying Protected Health Information in Chinese Electronic Health Records: Algorithm Development and Validation","year":2022,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Overfitting; Artificial intelligence; Health records; Data mining; Machine learning; Protected health information; Information extraction; Health informatics; Deep learning; Artificial neural network; Health care; Health policy; HRHIS","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002671469,0.001169279,0.001303467,0.002162447,0.0008441741,0.0009800274,0.002229416,0.001520377,0.003886593],"category_scores_gemma":[0.006367458,0.0004730909,0.0009704373,0.001409348,0.000527706,0.001698051,0.001337113,0.001476898,0.00118695],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001547413,"about_ca_system_score_gemma":0.004364615,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02223785,"about_ca_topic_score_gemma":0.01491994,"domain_scores_codex":[0.9989288,0.0001693095,0.0001399069,0.0002865082,0.0003699025,0.0001055603],"domain_scores_gemma":[0.9974701,0.0009654511,0.0001823009,0.0002145906,0.001092568,0.00007503665],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002461681,0.0002400431,0.006121744,0.0002656883,0.0001106403,0.0002712354,0.0001343559,0.1265074,0.008946885,0.002784026,0.006273396,0.8480984],"study_design_scores_gemma":[0.00003899144,0.00003829366,0.0008336197,0.00001285121,0.0000251923,0.00009904384,0.00003501306,0.9926883,0.004524657,0.0008416864,0.0008509207,0.00001147536],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04247153,0.0005946534,0.9499977,0.0003986957,0.0001009641,0.0006218506,0.0003205563,0.004284441,0.001209485],"genre_scores_gemma":[0.240467,0.0005946062,0.7509639,0.0003262147,0.00007857389,0.000938217,0.002288343,0.0002014265,0.004141702],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02223785,"threshold_uncertainty_score":0.04421681,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01750684231635612,"score_gpt":0.369468775256509,"score_spread":0.3519619329401529,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}