{"id":"W3161047411","doi":"10.12688/f1000research.52549.1","title":"Supervised topic modeling for predicting molecular substructure from mass spectrometry","year":2021,"lang":"en","type":"preprint","venue":"F1000Research","topic":"Metabolomics and Mass Spectrometry Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; University of Toronto","funders":"National Institute of General Medical Sciences; National Institutes of Health","keywords":"Metabolomics; Preprocessor; Bottleneck; Computational biology; Principal component analysis; Computer science; Set (abstract data type); Mass spectrum; Modular design; Chemical space; Pattern recognition (psychology); Biological system; Data mining; Artificial intelligence; Mass spectrometry; Bioinformatics; Chemistry; Drug discovery; Biology; Chromatography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0004799444,0.0004618508,0.0006589435,0.0002215661,0.0001985663,0.0002553572,0.000868863,0.0006914886,0.0001386685],"category_scores_gemma":[0.0004548409,0.000474214,0.0004748906,0.0002193512,0.00007712008,0.000004891294,0.001786114,0.0008491676,0.000002772592],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000074372,"about_ca_system_score_gemma":0.0003524677,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002663757,"about_ca_topic_score_gemma":0.00006454365,"domain_scores_codex":[0.9966845,0.0001136489,0.0004671485,0.001367884,0.0005652392,0.0008015924],"domain_scores_gemma":[0.9979619,0.00004883924,0.0001082879,0.001203154,0.0004849258,0.0001928561],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0001295931,0.00004696472,0.003441121,0.0002480428,0.0008345589,0.00002752122,0.00009452543,0.002959257,0.9911284,0.0002684497,0.0003210587,0.0005004572],"study_design_scores_gemma":[0.00219553,0.0003771968,0.001694438,0.0001902066,0.0003076454,0.000008706327,0.001400131,0.08688776,0.8882679,0.0136471,0.003552479,0.001470887],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8932642,0.01639762,0.08724092,0.0003778214,0.0007022497,0.0008677273,0.0004681028,0.00003799798,0.0006433227],"genre_scores_gemma":[0.9397559,0.002292111,0.05226302,0.0001252211,0.001456279,0.0002775002,0.003281533,0.0001202338,0.0004282499],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1028605,"threshold_uncertainty_score":0.9997709,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02947631917098903,"score_gpt":0.3038835450687069,"score_spread":0.2744072258977178,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}