{"id":"W2943628838","doi":"10.1101/622761","title":"The effect of number of healthcare visits on study sample selection and prevalence estimates in electronic health record data","year":2019,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Medical Coding and Health Information","field":"Health Professions","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kellogg's (Canada)","funders":"National Center for Advancing Translational Sciences; National Institute of Diabetes and Digestive and Kidney Diseases; Centers for Disease Control and Prevention; National Institutes of Health","keywords":"Sample (material); Electronic health record; Medicine; Health care; Sample size determination; SNOMED CT; Health data; Family medicine; Statistics; Terminology; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.6025243,0.0009214955,0.001638937,0.002318554,0.002454083,0.004647457,0.003320565,0.003584856,0.003765499],"category_scores_gemma":[0.8253902,0.001597422,0.00566901,0.003631555,0.006042703,0.005836618,0.00466631,0.003676332,0.0007633465],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002525105,"about_ca_system_score_gemma":0.003083006,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002965199,"about_ca_topic_score_gemma":0.003906282,"domain_scores_codex":[0.157869,0.7632921,0.03266939,0.01656806,0.02720845,0.002392969],"domain_scores_gemma":[0.04286491,0.901256,0.02098028,0.02758147,0.006469433,0.0008478595],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.007021081,0.0004882148,0.8569035,0.00254926,0.007118417,0.0004036465,0.00467693,0.003206757,0.001356489,0.005550571,0.008263156,0.1024619],"study_design_scores_gemma":[0.001370981,0.004892705,0.914264,0.003992128,0.006494748,0.001496229,0.00194509,0.027767,0.008846318,0.01267945,0.01597809,0.0002733356],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6306905,0.017448,0.2818962,0.02775082,0.002875279,0.009000336,0.004868624,0.001000576,0.02446979],"genre_scores_gemma":[0.9141309,0.0007402849,0.07258933,0.005243602,0.0005069295,0.004737972,0.000716566,0.0003593183,0.0009752113],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3974757,"threshold_uncertainty_score":0.4901584,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07582095941604945,"score_gpt":0.4052297692592078,"score_spread":0.3294088098431583,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}