{"id":"W4378472501","doi":"10.1038/s41598-023-35482-0","title":"Constructing a disease database and using natural language processing to capture and standardize free text clinical information","year":2023,"lang":"en","type":"article","venue":"Scientific Reports","topic":"Topic Modeling","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; Public Health Ontario","funders":"Institute of Health Services and Policy Research; Canadian Institutes of Health Research","keywords":"Computer science; Benchmark (surveying); Infectious disease (medical specialty); Disease; Natural language processing; Protocol (science); Data science; Artificial intelligence; Machine learning; Medicine; Pathology; Alternative medicine","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004221495,0.0009336469,0.000920711,0.008484377,0.0007719381,0.002783677,0.001794124,0.001202923,0.002828307],"category_scores_gemma":[0.01609753,0.0004975005,0.001562686,0.004316907,0.0006336935,0.003335052,0.002344788,0.001741652,0.002187385],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001107441,"about_ca_system_score_gemma":0.002999752,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005606965,"about_ca_topic_score_gemma":0.005196098,"domain_scores_codex":[0.9963098,0.001047038,0.0007218186,0.001098243,0.00070109,0.000122011],"domain_scores_gemma":[0.9883704,0.007277076,0.001001898,0.001477466,0.001626333,0.0002468979],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004799953,0.001010749,0.02785054,0.002206634,0.0002647527,0.001961303,0.002321795,0.02929774,0.04439659,0.01511718,0.03873565,0.8363572],"study_design_scores_gemma":[0.0002604995,0.0009118688,0.04697814,0.0006337545,0.0005331936,0.004395212,0.004796748,0.6253747,0.07460582,0.08275139,0.1584039,0.0003547735],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0490334,0.0008269033,0.9038112,0.002166633,0.0002830964,0.002198596,0.02959539,0.008734561,0.003350141],"genre_scores_gemma":[0.1746404,0.0006429165,0.7597154,0.0005153547,0.0002077434,0.001736716,0.0604992,0.000255599,0.001786568],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008484377,"threshold_uncertainty_score":0.02232569,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02654819782817638,"score_gpt":0.3204444185690802,"score_spread":0.2938962207409038,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}