{"id":"W4392972987","doi":"10.1101/2024.03.18.585541","title":"Machine learning methods applied to classify complex diseases using genomic data","year":2024,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Genetics, Bioinformatics, and Biomedical Research","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Canadian Institutes of Health Research; National Institutes of Health; Genentech; IXICO; H. Lundbeck A/S; Servier; Eisai; Northern California Institute for Research and Education; F. Hoffmann-La Roche; University of Southern California; Biogen; Eli Lilly and Company; Bristol-Myers Squibb; BioClinica; U.S. Department of Defense; Meso Scale Diagnostics; Alzheimer's Disease Neuroimaging Initiative; Novartis Pharmaceuticals Corporation; Pfizer; Alzheimer's Association","keywords":"Artificial intelligence; Computer science; Machine learning; Computational biology; Data science; Biology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","open_science"],"consensus_categories":[],"category_scores_codex":[0.001267112,0.0006514267,0.0006367825,0.0003257353,0.000219894,0.0004085869,0.001882076,0.0006834104,0.00005550955],"category_scores_gemma":[0.0007289535,0.0006371854,0.0001735053,0.0003691279,0.0002600439,0.000006604087,0.009717804,0.001059985,0.0001436693],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001566905,"about_ca_system_score_gemma":0.001055204,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005447297,"about_ca_topic_score_gemma":0.000004270071,"domain_scores_codex":[0.9960565,0.0002431098,0.0007365274,0.001548016,0.0005531846,0.0008626903],"domain_scores_gemma":[0.9961271,0.00005233433,0.0002461356,0.002517075,0.0002810197,0.000776376],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0000673488,0.0000780212,0.0005316478,0.0007741293,0.0002729997,0.00001390561,0.000007100446,0.0004662801,0.9960546,0.00005049877,0.001542064,0.000141351],"study_design_scores_gemma":[0.001182159,0.0003915733,0.01692822,0.0005300261,0.0007779149,1.95118e-7,0.00003019919,0.1108877,0.3768848,0.00002725437,0.4892657,0.003094243],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8152256,0.01599366,0.1507423,0.001225429,0.003860392,0.003736352,0.008327146,0.0006583018,0.0002308605],"genre_scores_gemma":[0.8519402,0.001023855,0.1446992,0.0005182671,0.001418397,0.00007863037,0.00007586909,0.0002095684,0.00003604364],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6191699,"threshold_uncertainty_score":0.9996079,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06061037757877705,"score_gpt":0.3344229349369701,"score_spread":0.273812557358193,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}