{"id":"W4229039606","doi":"10.1101/2022.05.03.490388","title":"A real data-driven simulation strategy to select an imputation method for mixed-type trait data","year":2022,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Ecology and Vegetation Dynamics Studies","field":"Environmental Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Guelph","funders":"","keywords":"Missing data; Imputation (statistics); Categorical variable; Random forest; Statistics; Trait; Type I and type II errors; Multivariate statistics; Computer science; Mean squared error; Data mining; Mathematics; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01657291,0.0006826103,0.0009684067,0.001090077,0.0006829053,0.0009164311,0.002243779,0.001210936,0.002818072],"category_scores_gemma":[0.03302428,0.000634214,0.001145188,0.0008388854,0.0007774953,0.001159766,0.001420708,0.00182453,0.0004947067],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006839992,"about_ca_system_score_gemma":0.001563087,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002962818,"about_ca_topic_score_gemma":0.003219706,"domain_scores_codex":[0.9966371,0.002471228,0.0001599164,0.0003129337,0.0002711135,0.000147625],"domain_scores_gemma":[0.9756898,0.01904724,0.001179219,0.001418182,0.002222311,0.0004432185],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003768413,0.0001316762,0.009511572,0.00009431326,0.0001748565,0.0001467285,0.000161254,0.9470517,0.001921226,0.0196738,0.0008714324,0.01988459],"study_design_scores_gemma":[0.00002472506,0.00002620323,0.0001410486,0.000007026778,0.000008656051,0.00001226298,0.000006721888,0.99675,0.0004766346,0.002347864,0.0001925387,0.000006406912],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03281287,0.00003144325,0.9659516,0.0001263542,0.00002199198,0.0001815726,0.0001055785,0.0004066831,0.0003618348],"genre_scores_gemma":[0.3004222,0.00003853753,0.6972592,0.0001358963,0.00002375447,0.001013853,0.0004744835,0.0001251071,0.0005070187],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01657291,"threshold_uncertainty_score":0.08764696,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05013011175079522,"score_gpt":0.3267265911972929,"score_spread":0.2765964794464977,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}