{"id":"W4205665932","doi":"10.17504/protocols.io.b3nfqmbn","title":"Benchmarking missing-values approaches for predictive models on health databases v2","year":2022,"lang":"en","type":"preprint","venue":"","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Missing data; Imputation (statistics); Benchmarking; Computer science; Predictive modelling; Machine learning; Discriminative model; Benchmark (surveying); Artificial intelligence; Data mining; Random forest","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.002098469,0.0005201282,0.00068987,0.0003581893,0.0008897852,0.0002954708,0.002366048,0.0001457788,0.00006827841],"category_scores_gemma":[0.0002380132,0.0005020539,0.0002394591,0.0002450502,0.00005779152,0.0003278677,0.004455624,0.001695904,0.000002653788],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006209499,"about_ca_system_score_gemma":0.001428101,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001845739,"about_ca_topic_score_gemma":0.00004786588,"domain_scores_codex":[0.9948002,0.0007883456,0.0006989036,0.002053284,0.0008987885,0.0007605054],"domain_scores_gemma":[0.9956437,0.001043995,0.0006547823,0.002253478,0.0001064042,0.0002976882],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003775304,0.0001377322,0.0002707617,0.001278081,0.00005737042,0.000003731748,0.002959024,0.719308,1.048295e-7,0.1216413,0.007695331,0.1466109],"study_design_scores_gemma":[0.0001738143,0.0005098922,0.0002654491,0.0003181088,0.000008814342,0.000004880012,0.0001537647,0.9435491,0.000005246991,0.04955133,0.00505665,0.0004029741],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.000222644,0.001050843,0.97164,0.01507308,0.001741145,0.002147101,0.0003524774,0.0008598063,0.006912913],"genre_scores_gemma":[0.08666651,0.0001913148,0.903875,0.004282963,0.0008941541,0.001559221,0.001589675,0.0001129234,0.0008282597],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.2242411,"threshold_uncertainty_score":0.9997431,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2279906903560777,"score_gpt":0.3799372270832587,"score_spread":0.1519465367271809,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}