{"id":"W4210833614","doi":"10.2196/36119","title":"Identifying COVID-19 Outbreaks From Contact-Tracing Interview Forms for Public Health Departments: Development of a Natural Language Processing Pipeline","year":2022,"lang":"en","type":"article","venue":"JMIR Public Health and Surveillance","topic":"Data-Driven Disease Surveillance","field":"Medicine","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institute of General Medical Sciences; National Institute on Drug Abuse; National Institute of Diabetes and Digestive and Kidney Diseases; National Heart, Lung, and Blood Institute; U.S. National Library of Medicine; National Institute on Alcohol Abuse and Alcoholism; National Institutes of Health","keywords":"Outbreak; Pipeline (software); Computer science; Recall; Precision and recall; Named-entity recognition; Contact tracing; Coronavirus disease 2019 (COVID-19); Natural language processing; Artificial intelligence; Data mining; Infectious disease (medical specialty); Medicine; Disease; Psychology; Engineering; Virology; Pathology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.004830491,0.0003411162,0.001136383,0.0003249728,0.000915712,0.0001305574,0.0003190549,0.00005919849,0.000105416],"category_scores_gemma":[0.001036668,0.0003127205,0.0001377109,0.0006030162,0.00007379572,0.0003875396,0.0003285238,0.0004028005,0.000002889652],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001186998,"about_ca_system_score_gemma":0.007558582,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000318007,"about_ca_topic_score_gemma":0.001614448,"domain_scores_codex":[0.9953047,0.0005754876,0.001476555,0.0007750244,0.0007247821,0.00114346],"domain_scores_gemma":[0.9962203,0.0002868405,0.0009793523,0.0004896618,0.0002375578,0.001786295],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0004665688,0.0007423373,0.1528847,0.009234075,0.0001784474,0.0000265702,0.01999521,0.00000144309,0.00008173519,0.00009487764,0.007347787,0.8089463],"study_design_scores_gemma":[0.01077674,0.000758863,0.1716959,0.0003339524,0.000006328622,0.0001329912,0.01990833,0.004042628,0.000009174603,0.0000560061,0.7914473,0.0008317267],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7825155,0.0908528,0.058047,0.05538226,0.0008031341,0.007138497,0.004329782,0.0007180434,0.0002129539],"genre_scores_gemma":[0.970731,0.0001777944,0.003652123,0.01755529,0.0001034195,0.0008369019,0.006757552,0.00005437795,0.0001315246],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8081146,"threshold_uncertainty_score":0.9999325,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08075919963231584,"score_gpt":0.3833390910134271,"score_spread":0.3025798913811113,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}