{"id":"W3010725448","doi":"10.1186/s12911-020-1068-5","title":"Methods to improve the quality of smoking records in a primary care EMR database: exploring multiple imputation and pattern-matching algorithms","year":2020,"lang":"en","type":"article","venue":"BMC Medical Informatics and Decision Making","topic":"Data-Driven Disease Surveillance","field":"Medicine","cited_by":17,"is_retracted":false,"has_abstract":true,"ca_institutions":"Health Sciences Centre; University of Alberta; University of Calgary","funders":"Canadian Institutes of Health Research; Alberta Innovates; Public Health Agency; Public Health Agency of Canada","keywords":"Missing data; Imputation (statistics); Medicine; Health informatics; Medical record; Data quality; Health care; Data mining; Population; Matching (statistics); Database; Family medicine; Computer science; Public health; Environmental health; Nursing; Machine learning","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00243821,0.0001466681,0.0004555389,0.0001070079,0.00007317454,0.00005175674,0.0001560933,0.00006867076,0.00001428102],"category_scores_gemma":[0.00396252,0.0001026933,0.00005123345,0.0002653378,0.00005287524,0.0003046069,0.0005727772,0.0002865846,0.00000133135],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000500656,"about_ca_system_score_gemma":0.0001509979,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007559481,"about_ca_topic_score_gemma":0.00008550132,"domain_scores_codex":[0.9975514,0.0001561263,0.001088425,0.0002015469,0.0008039552,0.0001985053],"domain_scores_gemma":[0.9958183,0.003224204,0.0002690159,0.0002920719,0.0000989247,0.0002975412],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002000085,0.00001342926,0.02786362,0.001494297,0.00001238092,0.000006773945,0.006603192,0.0001156393,0.00007709127,0.00001070549,0.00003150092,0.9635714],"study_design_scores_gemma":[0.004219594,0.00020519,0.1823885,0.004259282,0.00005294474,0.00002363984,0.02284339,0.7846413,0.0001344726,0.0003123552,0.0005710963,0.0003482328],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4612711,0.0002352263,0.5379997,0.00004284103,0.00008082276,0.0002477283,0.00005856089,0.00001781481,0.00004616994],"genre_scores_gemma":[0.5179281,0.0000941622,0.4800607,0.001755297,0.00006401401,0.00002440315,0.00006238738,0.00001077145,2.134404e-7],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9632231,"threshold_uncertainty_score":0.4743793,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1033849024432202,"score_gpt":0.4127406139001553,"score_spread":0.3093557114569351,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}