{"id":"W4317425811","doi":"10.1038/s41467-022-35734-z","title":"Annotation of natural product compound families using molecular networking topology and structural similarity fingerprinting","year":2023,"lang":"en","type":"article","venue":"Nature Communications","topic":"Metabolomics and Mass Spectrometry Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":90,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University; University of New Brunswick","funders":"National Center for Complementary and Integrative Health; National Institutes of Health; Natural Sciences and Engineering Research Council of Canada; Government of Canada; U.S. Department of Health and Human Services","keywords":"Bottleneck; Annotation; Computer science; Similarity (geometry); Fragmentation (computing); Computational biology; Mass spectrometry; Identification (biology); Data mining; Pattern recognition (psychology); Chemistry; Artificial intelligence; Biology; Chromatography","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002248198,0.0001072372,0.0001612332,0.00009664806,0.0002557146,0.00001659745,0.0002870112,0.0001170887,6.887473e-7],"category_scores_gemma":[0.0002416429,0.0001059664,0.00004721338,0.0003120087,0.0002096811,0.000005594537,0.0005941971,0.000305596,2.140463e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000008864404,"about_ca_system_score_gemma":0.00002509718,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002869553,"about_ca_topic_score_gemma":0.0001349138,"domain_scores_codex":[0.9992573,0.00009504074,0.0002027174,0.0002061776,0.00007675734,0.000162053],"domain_scores_gemma":[0.9990524,0.00005537084,0.0001337163,0.000596565,0.0001413518,0.00002060425],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","study_design_scores_codex":[0.00001679712,0.00001375684,0.02829999,0.00002773406,0.0001313803,8.426031e-7,0.0001676714,0.0002253047,0.9630599,0.005250993,0.0002028482,0.002602796],"study_design_scores_gemma":[0.001853393,0.0002587403,0.4951322,0.0001240072,0.0003857966,0.0001239002,0.002484498,0.06341203,0.3548616,0.006878611,0.0729578,0.001527409],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.978412,0.02011766,0.0001100172,0.0006775812,0.0002651789,0.0001337103,0.0000120391,0.00001908163,0.0002526895],"genre_scores_gemma":[0.9887007,0.002075579,0.008851266,0.00009459844,0.00008238571,0.00000536414,0.0001626514,0.00001116819,0.00001631126],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6081982,"threshold_uncertainty_score":0.4321183,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02166204898805064,"score_gpt":0.3207216103799703,"score_spread":0.2990595613919197,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}