{"id":"W3012211566","doi":"10.2196/17622","title":"Re-examination of Rule-Based Methods in Deidentification of Electronic Health Records: Algorithm Development and Validation","year":2020,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Key Research and Development Program of China; National Natural Science Foundation of China","keywords":"Computer science; Rule-based system; Task (project management); Machine learning; Artificial intelligence; Data mining; Health informatics; Set (abstract data type); Ensemble learning; Health records; Reliability (semiconductor); Health care","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05385492,0.0012395,0.00137503,0.00342726,0.0008602861,0.003317144,0.003306465,0.001731838,0.001514111],"category_scores_gemma":[0.08507132,0.0004498016,0.001396021,0.002335968,0.0008237985,0.003505217,0.001767742,0.003930377,0.0009546361],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001701812,"about_ca_system_score_gemma":0.003477766,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004834312,"about_ca_topic_score_gemma":0.004286431,"domain_scores_codex":[0.9767445,0.01289542,0.001851396,0.00310183,0.005038793,0.000368041],"domain_scores_gemma":[0.8918744,0.07478445,0.002984933,0.009989761,0.01950403,0.0008624393],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0003675782,0.0006161793,0.03044624,0.0005997437,0.0007086438,0.0002067345,0.0004520228,0.09330528,0.003775998,0.004973129,0.004151464,0.860397],"study_design_scores_gemma":[0.00006840185,0.000406579,0.00484951,0.000517441,0.0002290512,0.0003899634,0.0002848091,0.9543344,0.01785942,0.009260718,0.01173669,0.00006309291],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.06490213,0.005836594,0.9212145,0.002050739,0.0003290716,0.0005446759,0.0003538418,0.002347286,0.002421157],"genre_scores_gemma":[0.306238,0.001916323,0.6884232,0.0007180189,0.0001229075,0.0002907014,0.001002914,0.0001783657,0.00110962],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.05385492,"threshold_uncertainty_score":0.2848154,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03716599547608451,"score_gpt":0.3833501845760687,"score_spread":0.3461841890999842,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}