{"id":"W4293574676","doi":"10.2196/38154","title":"An Efficient Method for Deidentifying Protected Health Information in Chinese Electronic Health Records: Algorithm Development and Validation","year":2022,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Overfitting; Artificial intelligence; Health records; Data mining; Machine learning; Protected health information; Information extraction; Health informatics; Deep learning; Artificial neural network; Health care; Health policy; HRHIS","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00662068,0.0001784488,0.0003195497,0.0004613992,0.0006549516,0.000153538,0.0006184179,0.00007434959,0.00001577113],"category_scores_gemma":[0.0002572991,0.000167527,0.00002958535,0.0008712496,0.00001724809,0.0008529194,0.0003377273,0.0008227177,0.000003464535],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009565766,"about_ca_system_score_gemma":0.002491058,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002735222,"about_ca_topic_score_gemma":0.00008122341,"domain_scores_codex":[0.9961168,0.0005349081,0.001404455,0.0001855996,0.001097206,0.0006610474],"domain_scores_gemma":[0.998347,0.0001889978,0.0006753803,0.0003261093,0.00009601362,0.000366537],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001122745,0.00008078146,0.0004423913,0.0006704002,0.000004856268,4.478914e-7,0.05033024,0.005099781,1.829794e-7,0.001472803,0.0001011663,0.9417858],"study_design_scores_gemma":[0.0008140433,0.0005128014,0.002689065,0.00006300883,5.521943e-7,0.00004738182,0.00133723,0.9844082,0.000004387558,0.0003124248,0.009648479,0.000162452],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.06243568,0.00006409689,0.9314163,0.003444514,0.0002208692,0.002203693,0.000007139356,0.0001953776,0.00001230239],"genre_scores_gemma":[0.05216479,0.00001924936,0.9412892,0.004601922,0.00004025125,0.001558027,0.0003076285,0.00001327925,0.000005657052],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9793084,"threshold_uncertainty_score":0.6831552,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01750684231635612,"score_gpt":0.369468775256509,"score_spread":0.3519619329401529,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}