{"id":"W4385709848","doi":"10.1093/jamiaopen/ooad062","title":"Automated identification of unstandardized medication data: a scalable and flexible data standardization pipeline using RxNorm on GEMINI multicenter hospital data","year":2023,"lang":"en","type":"article","venue":"JAMIA Open","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; St. Michael's Hospital","funders":"Alliance de recherche numérique du Canada","keywords":"Standardization; Computer science; Identifier; Pharmacy; Coding (social sciences); Data mining; Information retrieval; Medicine; Statistics; Mathematics; Family medicine","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02291794,0.001665647,0.001127704,0.006035862,0.001097759,0.003496294,0.002256845,0.0006959593,0.001897202],"category_scores_gemma":[0.04283873,0.0008868027,0.001532432,0.004723921,0.001059669,0.004105562,0.004252796,0.001468244,0.001811263],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002627307,"about_ca_system_score_gemma":0.006855585,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01926284,"about_ca_topic_score_gemma":0.01506424,"domain_scores_codex":[0.9860137,0.003490947,0.002213792,0.003454697,0.004375889,0.0004510507],"domain_scores_gemma":[0.9682873,0.01021647,0.004793179,0.008882719,0.007109439,0.0007107822],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002162022,0.0006583183,0.140184,0.001460255,0.0007950412,0.001102731,0.003911375,0.02062934,0.04967035,0.00834034,0.08722878,0.6838574],"study_design_scores_gemma":[0.0006620859,0.0008763178,0.120377,0.000666213,0.0004134311,0.00147264,0.002450626,0.4801412,0.1960168,0.01717509,0.1791541,0.0005945335],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1334763,0.001609107,0.5519601,0.003147989,0.0001779025,0.002416237,0.04159261,0.2594373,0.006182528],"genre_scores_gemma":[0.2147084,0.0005385302,0.712531,0.0007375658,0.00009932744,0.0008361132,0.06495054,0.003678691,0.001919738],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02291794,"threshold_uncertainty_score":0.1212031,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09741173152302603,"score_gpt":0.4051766285713367,"score_spread":0.3077648970483107,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}