{"id":"W4321351204","doi":"10.1002/sim.9685","title":"Impute‐then‐exclude versus exclude‐then‐impute: Lessons when imputing a variable used both in cohort creation and as an independent variable in the analysis model","year":2023,"lang":"en","type":"article","venue":"Statistics in Medicine","topic":"Statistical Methods and Bayesian Inference","field":"Mathematics","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Institute for Clinical Evaluative Sciences; Sunnybrook Hospital; University of Toronto","funders":"Canadian Institutes of Health Research; Ontario Ministry of Health and Long-Term Care; Heart and Stroke Foundation of Canada","keywords":"Missing data; Imputation (statistics); Statistics; Sample size determination; Variable (mathematics); Random variable; Computer science; Mathematics; Medicine; Econometrics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3804743,0.002130252,0.004691964,0.002465629,0.002169766,0.006456438,0.007797923,0.004849097,0.003450504],"category_scores_gemma":[0.6344141,0.002447204,0.005568061,0.003885154,0.006503616,0.007279988,0.006160449,0.01445809,0.0008525943],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002278779,"about_ca_system_score_gemma":0.006744858,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008383493,"about_ca_topic_score_gemma":0.01006877,"domain_scores_codex":[0.6377845,0.3399073,0.005363295,0.006837366,0.008649953,0.001457675],"domain_scores_gemma":[0.2726597,0.6755845,0.00822499,0.03361044,0.008519338,0.001401099],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002464454,0.0005356178,0.07610411,0.001707371,0.005961111,0.001432484,0.01038747,0.1208501,0.00072883,0.2971576,0.02318156,0.4594895],"study_design_scores_gemma":[0.0008143933,0.001142057,0.01165528,0.001469111,0.0007872205,0.001069934,0.002183317,0.4746118,0.003776119,0.475721,0.02630534,0.0004644406],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02355164,0.001465912,0.9598555,0.0108342,0.0003800208,0.0006086171,0.0003377505,0.0005308304,0.002435588],"genre_scores_gemma":[0.2277024,0.001133041,0.7617737,0.005136517,0.0005627738,0.001085003,0.0004691837,0.0008238064,0.001313519],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6195257,"threshold_uncertainty_score":0.7639855,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09486507976885547,"score_gpt":0.4341099087394667,"score_spread":0.3392448289706113,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}