{"id":"W4283364896","doi":"10.1101/2022.02.16.22268694","title":"Automated identification of unstandardized medication data: A scalable and flexible data standardization pipeline using RxNorm on GEMINI multicenter hospital data","year":2022,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; St. Michael's Hospital","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institutes of Health Research; Alliance de recherche numérique du Canada; Canadian Frailty Network; University of Toronto; University Health Network","keywords":"Standardization; Computer science; Identifier; Pharmacy; False positive paradox; Data mining; Coding (social sciences); Information retrieval; Medicine; Artificial intelligence; Statistics; Mathematics; Family medicine","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02240972,0.001656929,0.001209887,0.006868535,0.001176415,0.004009985,0.002349998,0.0007028171,0.001900716],"category_scores_gemma":[0.04501295,0.000944917,0.001615386,0.005283065,0.001086197,0.004568312,0.004704499,0.001565053,0.001714444],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002587423,"about_ca_system_score_gemma":0.006839572,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02057021,"about_ca_topic_score_gemma":0.01604202,"domain_scores_codex":[0.9845441,0.003805523,0.002504772,0.00397389,0.004698724,0.000472947],"domain_scores_gemma":[0.9645962,0.01176866,0.004757347,0.01009351,0.007982344,0.000802031],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001914018,0.0006043417,0.1349377,0.001530945,0.0008782936,0.001147138,0.004191599,0.02136965,0.04694799,0.01022459,0.1023834,0.6738703],"study_design_scores_gemma":[0.0006014692,0.0006943723,0.09930225,0.000669539,0.0004076873,0.001230543,0.002709725,0.5255401,0.1660272,0.02128032,0.1809623,0.000574565],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1197678,0.001604548,0.5429134,0.00336185,0.0002042885,0.002239239,0.0443072,0.2792849,0.006316784],"genre_scores_gemma":[0.2179647,0.0005139289,0.7088275,0.0008024559,0.0001056116,0.0007956206,0.06495728,0.004092445,0.001940494],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02240972,"threshold_uncertainty_score":0.1185154,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07059790562919259,"score_gpt":0.3699112617216725,"score_spread":0.2993133560924799,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}