{"id":"W4312214325","doi":"10.1101/2022.12.12.520180","title":"Leveraging a machine learning derived surrogate phenotype to improve power for genome-wide association studies of partially missing phenotypes in population biobanks","year":2022,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Genetic Associations and Epidemiology","field":"Biochemistry, Genetics and Molecular Biology","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Medical Research Council; Natural Sciences and Engineering Research Council of Canada; National Institutes of Health","keywords":"Biobank; Missing data; Imputation (statistics); Inference; Genome-wide association study; Computer science; Population; Statistical power; Computational biology; Artificial intelligence; Machine learning; Data mining; Statistics; Bioinformatics; Biology; Genetics; Genotype; Single-nucleotide polymorphism; Mathematics; Medicine; Gene","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03072913,0.000737236,0.001474573,0.00122081,0.0004679254,0.001663745,0.001469753,0.001325147,0.002499382],"category_scores_gemma":[0.06147068,0.0006079415,0.001621261,0.001216075,0.001865254,0.001574497,0.00268005,0.001898147,0.0005897195],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004122749,"about_ca_system_score_gemma":0.0008540428,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009295772,"about_ca_topic_score_gemma":0.001161304,"domain_scores_codex":[0.9878588,0.01013423,0.0002538254,0.0009667028,0.0006281742,0.0001584131],"domain_scores_gemma":[0.9368199,0.05194846,0.002424651,0.007003914,0.001264564,0.0005385447],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00169979,0.0003608369,0.07191139,0.0005178596,0.001702945,0.0009413308,0.0005785122,0.655284,0.0133508,0.05943068,0.007529142,0.1866928],"study_design_scores_gemma":[0.0001354431,0.0001722061,0.003568305,0.00003998922,0.00009898785,0.0001915361,0.0000288015,0.9550925,0.002857324,0.03592174,0.001856919,0.00003617144],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.07381929,0.0003597016,0.9220309,0.0009248347,0.0001084031,0.00005591247,0.000379167,0.001291307,0.001030318],"genre_scores_gemma":[0.6682234,0.0002354976,0.328138,0.0007276855,0.0001518484,0.0001633718,0.0009890031,0.0003654942,0.001005627],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.03072913,"threshold_uncertainty_score":0.1625131,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01857247223814745,"score_gpt":0.2608246647387107,"score_spread":0.2422521925005633,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}