{"id":"W4385848332","doi":"10.1016/j.patter.2023.100887","title":"Enhancing phenotype recognition in clinical notes using large language models: PhenoBCBERT and PhenoGPT","year":2023,"lang":"en","type":"article","venue":"Patterns","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":65,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"U.S. National Library of Medicine; National Human Genome Research Institute; Intellectual and Developmental Disabilities Research Center; CHEO Research Institute; University of Pennsylvania; National Institutes of Health; Eunice Kennedy Shriver National Institute of Child Health and Human Development; Children's Hospital of Philadelphia","keywords":"Phenotype; Leverage (statistics); Computer science; Vocabulary; Heuristic; Ontology; Natural language processing; Scope (computer science); Artificial intelligence; Computational biology; Machine learning; Biology; Gene; Genetics; Linguistics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002797356,0.0009283393,0.0006619075,0.002259144,0.0003489507,0.001762365,0.00113166,0.0009399331,0.001908731],"category_scores_gemma":[0.01149133,0.0003823699,0.001346093,0.001136544,0.0004528223,0.002301759,0.001413652,0.00141011,0.000941752],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001314594,"about_ca_system_score_gemma":0.002336393,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01401886,"about_ca_topic_score_gemma":0.02613016,"domain_scores_codex":[0.9984488,0.0004746586,0.0002003932,0.0003853809,0.0004328795,0.0000578162],"domain_scores_gemma":[0.991847,0.005880296,0.0005293118,0.0005851997,0.0009160364,0.0002420534],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001160223,0.0008536029,0.05144164,0.001252796,0.0005712566,0.002367169,0.001010057,0.3306701,0.02112205,0.02195842,0.04826863,0.5193241],"study_design_scores_gemma":[0.00006157662,0.00006096181,0.002483299,0.00006525283,0.00007018616,0.0003432042,0.00008687903,0.9720939,0.00524867,0.01028804,0.009153128,0.0000448365],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09563865,0.001036231,0.8466375,0.004676316,0.0003088176,0.0008939957,0.01533319,0.03096535,0.00451],"genre_scores_gemma":[0.3804777,0.0006613707,0.5962039,0.001473864,0.0001213285,0.0006157347,0.01692276,0.0007938981,0.002729514],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01401886,"threshold_uncertainty_score":0.02787453,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07803578452570843,"score_gpt":0.3586311829778933,"score_spread":0.2805953984521849,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}