{"id":"W3210133931","doi":"10.1097/ede.0000000000001432","title":"On the Nature of Informative Presence Bias in Analyses of Electronic Health Records","year":2021,"lang":"en","type":"article","venue":"Epidemiology","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"National Institute of Environmental Health Sciences","keywords":"Health records; Confounding; Electronic health record; Scope (computer science); Non-response bias; Cohort; Information bias; Medical record","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2960759,0.001118683,0.002584801,0.00266506,0.003039595,0.006815774,0.003658122,0.004612824,0.003113383],"category_scores_gemma":[0.614203,0.001420024,0.003552538,0.002783853,0.01522889,0.01150482,0.006422359,0.007387494,0.0004059534],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003233786,"about_ca_system_score_gemma":0.003911053,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003783766,"about_ca_topic_score_gemma":0.003736285,"domain_scores_codex":[0.6927683,0.2751328,0.00674002,0.01188146,0.01085044,0.00262691],"domain_scores_gemma":[0.1636963,0.7983841,0.01434472,0.01909989,0.003600734,0.0008742981],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00160182,0.0002644092,0.1153608,0.001729119,0.002711359,0.001705586,0.004413692,0.1132747,0.0009272713,0.6235772,0.005813949,0.1286201],"study_design_scores_gemma":[0.0002883346,0.0003331298,0.01286955,0.001270879,0.0006557841,0.0008929608,0.0007083688,0.1599002,0.001422192,0.8160086,0.005495931,0.0001539994],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.09983601,0.007370951,0.8506959,0.02846346,0.0005821906,0.0006341201,0.0005735117,0.0004594977,0.01138431],"genre_scores_gemma":[0.8561464,0.002350405,0.1319827,0.005554053,0.0009941505,0.0007671522,0.0004168397,0.0001697229,0.001618567],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.7039242,"threshold_uncertainty_score":0.8680638,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1350699049802592,"score_gpt":0.4530477399609975,"score_spread":0.3179778349807383,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}