{"id":"W3161047411","doi":"10.12688/f1000research.52549.1","title":"Supervised topic modeling for predicting molecular substructure from mass spectrometry","year":2021,"lang":"en","type":"preprint","venue":"F1000Research","topic":"Metabolomics and Mass Spectrometry Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; University of Toronto","funders":"National Institute of General Medical Sciences; National Institutes of Health","keywords":"Metabolomics; Preprocessor; Bottleneck; Computational biology; Principal component analysis; Computer science; Set (abstract data type); Mass spectrum; Modular design; Chemical space; Pattern recognition (psychology); Biological system; Data mining; Artificial intelligence; Mass spectrometry; Bioinformatics; Chemistry; Drug discovery; Biology; Chromatography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004674396,0.001584927,0.001365711,0.002235384,0.0007627305,0.001574295,0.00273529,0.002371271,0.002994645],"category_scores_gemma":[0.01147223,0.0008854509,0.002571573,0.001583823,0.001028233,0.001903287,0.00169754,0.00255829,0.001920287],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001619487,"about_ca_system_score_gemma":0.001129303,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007116407,"about_ca_topic_score_gemma":0.009492857,"domain_scores_codex":[0.9981624,0.0008570909,0.0001080212,0.0005154538,0.0002347325,0.0001221762],"domain_scores_gemma":[0.993817,0.004687417,0.0004178036,0.0003308896,0.0006024531,0.0001444158],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009622779,0.0003023142,0.01096943,0.0008664045,0.0006442225,0.0003425626,0.0006761923,0.5556198,0.01575631,0.02838843,0.02872002,0.3567519],"study_design_scores_gemma":[0.00001893817,0.00001885564,0.0005602707,0.00002080114,0.00001960925,0.00002769794,0.00002398758,0.9842859,0.0011895,0.01219753,0.001622627,0.00001426549],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01685625,0.001373277,0.9763173,0.001056386,0.0001420354,0.0001448537,0.001177402,0.002014971,0.0009176247],"genre_scores_gemma":[0.4569847,0.001743283,0.5234503,0.0008937362,0.0009118935,0.001098568,0.008546304,0.0008124579,0.005558797],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007116407,"threshold_uncertainty_score":0.02472085,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02947631917098903,"score_gpt":0.3038835450687069,"score_spread":0.2744072258977178,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}