{"id":"W4389991380","doi":"10.1038/s41467-023-44035-y","title":"Open access repository-scale propagated nearest neighbor suspect spectral library for untargeted metabolomics","year":2023,"lang":"en","type":"article","venue":"Nature Communications","topic":"Metabolomics and Mass Spectrometry Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":80,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"National Center for Complementary and Integrative Health; National Institute of General Medical Sciences; National Institute on Aging; Biotechnology and Biological Sciences Research Council; National Institutes of Health; Universiteit Antwerpen; National Research Foundation of Korea; Conselho Nacional de Desenvolvimento Científico e Tecnológico; Agence Nationale de la Recherche; Deutsche Forschungsgemeinschaft; National Cancer Institute; National Research Foundation; Netherlands eScience Center; US-UK Fulbright Commission; Joint Genome Institute; Ministry of Science and ICT, South Korea; Fonds National de la Recherche Luxembourg; Ministry of Innovative Development of the Republic of Uzbekistan; Gordon and Betty Moore Foundation; U.S. Department of Energy; National Science Foundation","keywords":"Suspect; Computer science; Scale (ratio); Metabolomics; k-nearest neighbors algorithm; Computational biology; Data mining; World Wide Web; Data science; Bioinformatics; Biology; Artificial intelligence; Geography; Cartography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002335424,0.000203344,0.000280681,0.0001289599,0.0005950968,0.0004463542,0.003983631,0.0002871426,0.0000175553],"category_scores_gemma":[0.0002061074,0.0001865723,0.0001438663,0.0007909762,0.0001379667,0.00005189059,0.004061447,0.0004004607,0.00000948135],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001692974,"about_ca_system_score_gemma":0.0001746053,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002651783,"about_ca_topic_score_gemma":0.0002907188,"domain_scores_codex":[0.9987108,0.0001203281,0.0002956245,0.0004358393,0.0001021502,0.0003352284],"domain_scores_gemma":[0.9977022,0.00008833702,0.0001573214,0.001815359,0.0001450987,0.00009164034],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0003601668,0.0003535183,0.02459149,0.00004095111,0.0006679745,0.000003658172,0.00009960634,0.00003216049,0.5421386,0.03483264,0.3962393,0.0006398824],"study_design_scores_gemma":[0.0007932585,0.0001264958,0.03278388,0.000009252245,0.00008949811,0.000007150934,0.0001021908,0.0004593592,0.2177627,0.001153904,0.7463722,0.0003402062],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8485668,0.03785817,0.0005595483,0.03752899,0.002320233,0.006241854,0.002025411,0.0007746196,0.06412438],"genre_scores_gemma":[0.9721857,0.00546744,0.01279797,0.0006425054,0.0003088954,0.0004253198,0.003773028,0.00007383434,0.00432531],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3501328,"threshold_uncertainty_score":0.76082,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03412720650927243,"score_gpt":0.3515244981108545,"score_spread":0.3173972916015821,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}