{"id":"W2791458756","doi":"10.2196/medinform.8960","title":"Characterizing and Managing Missing Structured Data in Electronic Health Records: Data Analysis","year":2018,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Statistical Methods and Bayesian Inference","field":"Mathematics","cited_by":174,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"U.S. National Library of Medicine; National Institute of Environmental Health Sciences; National Institute of Allergy and Infectious Diseases; National Human Genome Research Institute; University of Pennsylvania; National Institutes of Health; Pennsylvania Department of Health","keywords":"Missing data; Imputation (statistics); Health records; Computer science; Data science; Electronic health record; Data mining; Health care; Machine learning","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1618055,0.001556426,0.002786629,0.006906434,0.002669608,0.005935527,0.006235376,0.002975062,0.001615854],"category_scores_gemma":[0.3850677,0.001786325,0.003955049,0.01226353,0.003473086,0.008335262,0.005702194,0.005043277,0.0007409051],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002817116,"about_ca_system_score_gemma":0.009961076,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003691247,"about_ca_topic_score_gemma":0.003521664,"domain_scores_codex":[0.7957572,0.1683882,0.01341859,0.006478245,0.01477504,0.001182708],"domain_scores_gemma":[0.4848964,0.4240991,0.03891189,0.03155181,0.01918129,0.001359564],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0004572458,0.0006800584,0.1853109,0.009158053,0.003380133,0.0006694454,0.007679041,0.1260181,0.002272693,0.07612578,0.02304954,0.565199],"study_design_scores_gemma":[0.0003459515,0.00088503,0.04770602,0.009884449,0.001261775,0.001473706,0.006127028,0.439486,0.01023312,0.441892,0.0400249,0.0006798966],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01498847,0.002085174,0.9748694,0.004452876,0.0001264079,0.0009405634,0.0012791,0.0006104618,0.0006476479],"genre_scores_gemma":[0.1216429,0.002163111,0.8708085,0.0009797481,0.0002506851,0.001648132,0.002115493,0.0001677112,0.0002236778],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.1618055,"threshold_uncertainty_score":0.8557194,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08677746776859628,"score_gpt":0.4334871571936644,"score_spread":0.3467096894250681,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}