{"id":"W4410872475","doi":"10.1101/2025.05.28.25328511","title":"PheCode-guided multi-modal topic modeling of electronic health records improves disease incidence prediction and GWAS discovery from UK Biobank","year":2025,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; McGill University","funders":"Fonds de recherche du Québec – Nature et technologies; Alliance de recherche numérique du Canada; Canada First Research Excellence Fund; Natural Sciences and Engineering Research Council of Canada; McGill University","keywords":"Biobank; Health records; Modal; Data science; Disease; Incidence (geometry); Genome-wide association study; Data discovery; Medicine; Computer science; Internal medicine; Bioinformatics; Political science; World Wide Web; Health care; Biology; Mathematics; Genetics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00491341,0.0007197127,0.0009346401,0.001809275,0.0005467284,0.001765865,0.0009231946,0.001164426,0.003408168],"category_scores_gemma":[0.02445936,0.0004862412,0.001974923,0.001722192,0.0003745285,0.001539715,0.001975433,0.001580873,0.001693422],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007060873,"about_ca_system_score_gemma":0.001471012,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01141168,"about_ca_topic_score_gemma":0.02326848,"domain_scores_codex":[0.9974342,0.001362565,0.0001586876,0.0006652909,0.0002473291,0.0001319328],"domain_scores_gemma":[0.9873111,0.01023145,0.0006272544,0.001066918,0.0005876846,0.0001756038],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002408562,0.0005421389,0.1891331,0.001577614,0.002021341,0.0008103354,0.002547827,0.2131647,0.01091852,0.02976461,0.0968636,0.4502476],"study_design_scores_gemma":[0.0002738729,0.0001125843,0.02243741,0.0001603792,0.0002675671,0.0002944224,0.0002187895,0.9152316,0.003407226,0.03785674,0.01964536,0.00009406833],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1860437,0.00163695,0.7536072,0.004300804,0.0002175107,0.000337466,0.0349777,0.01377818,0.005100478],"genre_scores_gemma":[0.6350632,0.0009773499,0.2869205,0.001300496,0.0003850391,0.0007132547,0.069089,0.001331398,0.004219803],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01141168,"threshold_uncertainty_score":0.02598488,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02675848719481826,"score_gpt":0.3081720775947838,"score_spread":0.2814135903999655,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}