{"id":"W3110873124","doi":"10.2196/24008","title":"Family History Extraction From Synthetic Clinical Narratives Using Natural Language Processing: Overview and Evaluation of a Challenge Data Set and Solutions for the 2019 National NLP Clinical Challenges (n2c2)/Open Health Natural Language Processing (OHNLP) Competition","year":2020,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Genomics and Rare Diseases","field":"Biochemistry, Genetics and Molecular Biology","cited_by":20,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Center for Advancing Translational Sciences; U.S. National Library of Medicine; National Institute of General Medical Sciences; National Institutes of Health","keywords":"Natural language processing; Natural history; Computer science; Set (abstract data type); Narrative; Natural (archaeology); Artificial intelligence; Data set; Natural language; Data science; Medicine; Linguistics; History; Programming language","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02094486,0.002612709,0.001205415,0.003554475,0.002391729,0.002726105,0.003917539,0.003448373,0.004252743],"category_scores_gemma":[0.05059613,0.0006816432,0.002331201,0.002201331,0.001934748,0.003529975,0.006632628,0.003357711,0.003556123],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0026994,"about_ca_system_score_gemma":0.005344688,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01355309,"about_ca_topic_score_gemma":0.02023738,"domain_scores_codex":[0.9773046,0.0117933,0.002831636,0.00407747,0.003332244,0.0006608044],"domain_scores_gemma":[0.9424846,0.0376082,0.001663333,0.007034924,0.008917828,0.002291111],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.004170373,0.004469349,0.05673887,0.01307375,0.001507954,0.007201359,0.01041095,0.02861947,0.02896644,0.007197524,0.3993413,0.4383026],"study_design_scores_gemma":[0.002904455,0.00309059,0.09821668,0.00274255,0.001210112,0.008975975,0.0155115,0.2968119,0.07344999,0.02628562,0.4698181,0.0009825416],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4386179,0.01562039,0.2320441,0.02097873,0.003607497,0.01297261,0.2282308,0.02924486,0.01868315],"genre_scores_gemma":[0.2149608,0.001928677,0.3208156,0.002897027,0.0005031888,0.005934175,0.4465071,0.001461945,0.004991515],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02094486,"threshold_uncertainty_score":0.1107683,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2907145640907986,"score_gpt":0.4810624134638092,"score_spread":0.1903478493730106,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}