{"id":"W1685527554","doi":"10.1016/j.jbi.2015.09.004","title":"Hidden Markov model using Dirichlet process for de-identification","year":2015,"lang":"en","type":"article","venue":"Journal of Biomedical Informatics","topic":"Bayesian Methods and Mixture Models","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"ca_institutions":"Memorial University of Newfoundland","funders":"U.S. National Library of Medicine","keywords":"Hidden Markov model; Computer science; Conditional random field; Artificial intelligence; Dirichlet process; Feature (linguistics); Context (archaeology); Identification (biology); Machine learning; Dirichlet distribution; Hierarchical Dirichlet process; Vocabulary; Pattern recognition (psychology); Field (mathematics); Bayesian probability; Topic model; Latent Dirichlet allocation; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006551606,0.001446182,0.002371051,0.001871257,0.000966816,0.001674894,0.003796329,0.002639499,0.005026999],"category_scores_gemma":[0.01735434,0.001330311,0.002349272,0.002057023,0.001171572,0.003174611,0.002125508,0.005213071,0.002915351],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002002862,"about_ca_system_score_gemma":0.002251397,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01167802,"about_ca_topic_score_gemma":0.01170142,"domain_scores_codex":[0.9957052,0.002218079,0.0002222156,0.001000703,0.0006122061,0.0002416403],"domain_scores_gemma":[0.9916412,0.006525911,0.0003571486,0.0007042864,0.0006371879,0.0001343613],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003214042,0.0001222349,0.001607171,0.0002286806,0.0001887352,0.0002065931,0.0002448983,0.7467468,0.002038749,0.1050058,0.008340626,0.1349484],"study_design_scores_gemma":[0.00001100715,0.00001017452,0.0001094582,0.00001357654,0.000009640124,0.00002660949,0.000008610305,0.9647065,0.0004403343,0.03327473,0.001372417,0.00001699796],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002102376,0.000342233,0.9958439,0.0002386393,0.00006553751,0.00004625983,0.0002569034,0.0005350452,0.0005690504],"genre_scores_gemma":[0.2352874,0.001325224,0.7495505,0.0006138837,0.0003740103,0.0008979883,0.003797875,0.0006152777,0.007537898],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01167802,"threshold_uncertainty_score":0,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06895016093744245,"score_gpt":0.36020037196107,"score_spread":0.2912502110236276,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}