{"id":"W2943628838","doi":"10.1101/622761","title":"The effect of number of healthcare visits on study sample selection and prevalence estimates in electronic health record data","year":2019,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Medical Coding and Health Information","field":"Health Professions","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kellogg's (Canada)","funders":"National Center for Advancing Translational Sciences; National Institute of Diabetes and Digestive and Kidney Diseases; Centers for Disease Control and Prevention; National Institutes of Health","keywords":"Sample (material); Electronic health record; Medicine; Health care; Sample size determination; SNOMED CT; Health data; Family medicine; Statistics; Terminology; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.008408627,0.0003347227,0.0008835287,0.0001871401,0.0004863677,0.00002079567,0.0006004598,0.0004029976,0.00002086605],"category_scores_gemma":[0.002567115,0.0002576202,0.00003654611,0.0004425959,0.00005103978,0.0001422023,0.0005510272,0.002145518,0.00002132008],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005943013,"about_ca_system_score_gemma":0.002421707,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004021011,"about_ca_topic_score_gemma":0.0005612075,"domain_scores_codex":[0.9946113,0.001863104,0.001534272,0.0006574847,0.0005311099,0.000802712],"domain_scores_gemma":[0.9935531,0.002904405,0.001609435,0.001412782,0.0003117066,0.0002085837],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0003580635,0.00008759306,0.9803538,0.01802807,0.00004669177,3.40367e-7,0.0001483047,0.00002919131,0.0001156373,0.0004384742,0.0002747875,0.0001190933],"study_design_scores_gemma":[0.002053067,0.002728552,0.975177,0.008186582,0.00007549636,1.234041e-8,0.00007019394,0.00918867,0.0005988182,0.00001183759,0.001518139,0.0003915901],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9918364,0.00114055,0.0003546982,0.0009947454,0.0007046555,0.004587253,0.0002874204,0.00008921892,0.000005016312],"genre_scores_gemma":[0.99658,0.002363229,0.0004227029,0.0001725578,0.0001228227,0.000289556,0.00000241086,0.00004423438,0.000002509365],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.009841482,"threshold_uncertainty_score":0.9999876,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07582095941604945,"score_gpt":0.4052297692592078,"score_spread":0.3294088098431583,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}