{"id":"W4210833614","doi":"10.2196/36119","title":"Identifying COVID-19 Outbreaks From Contact-Tracing Interview Forms for Public Health Departments: Development of a Natural Language Processing Pipeline","year":2022,"lang":"en","type":"article","venue":"JMIR Public Health and Surveillance","topic":"Data-Driven Disease Surveillance","field":"Medicine","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institute of General Medical Sciences; National Institute on Drug Abuse; National Institute of Diabetes and Digestive and Kidney Diseases; National Heart, Lung, and Blood Institute; U.S. National Library of Medicine; National Institute on Alcohol Abuse and Alcoholism; National Institutes of Health","keywords":"Outbreak; Pipeline (software); Computer science; Recall; Precision and recall; Named-entity recognition; Contact tracing; Coronavirus disease 2019 (COVID-19); Natural language processing; Artificial intelligence; Data mining; Infectious disease (medical specialty); Medicine; Disease; Psychology; Engineering; Virology; Pathology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004763545,0.001038414,0.0004377062,0.004614257,0.0006165269,0.001458213,0.0011107,0.0008015899,0.003335128],"category_scores_gemma":[0.01625946,0.0005441456,0.001077883,0.001953008,0.0003790894,0.002551275,0.001714127,0.001453671,0.002770512],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001628136,"about_ca_system_score_gemma":0.00320712,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02558567,"about_ca_topic_score_gemma":0.03200082,"domain_scores_codex":[0.9979049,0.0007171663,0.0003137165,0.0005867484,0.0003546575,0.0001228709],"domain_scores_gemma":[0.990417,0.005539472,0.0008071633,0.0007519695,0.002293071,0.0001914274],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005133061,0.0006059884,0.09987921,0.001489529,0.0002766661,0.001049592,0.002856384,0.03777051,0.04122081,0.003451276,0.05586977,0.7550169],"study_design_scores_gemma":[0.000120585,0.0003639009,0.0832717,0.0003760381,0.0002469002,0.0008809632,0.003379449,0.804731,0.04677358,0.01037276,0.04931623,0.0001669449],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2325795,0.0006872434,0.6682335,0.003006999,0.0002235077,0.00439941,0.04570988,0.037784,0.0073759],"genre_scores_gemma":[0.2994102,0.0003629678,0.6421757,0.000403114,0.0000756126,0.001406118,0.05248943,0.0004098502,0.003267008],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02558567,"threshold_uncertainty_score":0.05087346,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08075919963231584,"score_gpt":0.3833390910134271,"score_spread":0.3025798913811113,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}