{"id":"W4412049114","doi":"10.1016/j.jbi.2025.104873","title":"Accounting for population structure in deep learning models for genomic analysis","year":2025,"lang":"en","type":"article","venue":"Journal of Biomedical Informatics","topic":"Genetic Associations and Epidemiology","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Alberta Children's Hospital; University of Calgary","funders":"National Institute of Mental Health; Natural Sciences and Engineering Research Council of Canada; National Institutes of Health; Alberta Innovates; Alberta Children's Hospital Foundation; Calgary Foundation","keywords":"Artificial intelligence; Confounding; Deep learning; Population; Computer science; Machine learning; Single-nucleotide polymorphism; Computational biology; Biology; Genotype; Genetics; Statistics; Medicine; Mathematics; Gene","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006352015,0.001186865,0.001064873,0.000996488,0.0008242312,0.001590599,0.002668456,0.002044824,0.003211234],"category_scores_gemma":[0.01985755,0.0008810763,0.001432331,0.0010729,0.001681047,0.002483108,0.001922118,0.004429688,0.0005310686],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002839707,"about_ca_system_score_gemma":0.002366731,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01278055,"about_ca_topic_score_gemma":0.01695635,"domain_scores_codex":[0.9985404,0.0007637131,0.00007355615,0.0002987323,0.0001823399,0.0001412871],"domain_scores_gemma":[0.9888833,0.008972069,0.0006041045,0.0006134195,0.0007085082,0.0002186068],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004992743,0.00003718868,0.00470398,0.00004916125,0.00009449907,0.00009131891,0.00007807851,0.9503158,0.0002646148,0.0183795,0.001234909,0.02470094],"study_design_scores_gemma":[0.000007487256,0.000009465292,0.0002718924,0.00001339596,0.00001200669,0.00001488821,0.000006066743,0.9718462,0.0001343778,0.02729893,0.0003791293,0.00000623731],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03457996,0.0007316014,0.9609153,0.001477048,0.00008421006,0.00006488206,0.0003280008,0.0006486583,0.001170424],"genre_scores_gemma":[0.7314727,0.0009247681,0.2592336,0.001178748,0.0001755995,0.0005860937,0.001091857,0.0003024511,0.005034196],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01278055,"threshold_uncertainty_score":0.03359312,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01066905425319228,"score_gpt":0.2878423783043185,"score_spread":0.2771733240511262,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}