{"id":"W4394925712","doi":"10.26434/chemrxiv-2024-pz45l","title":"Machine Learning in Complex Organic Mixtures: Applying Domain Knowledge Allows for Meaningful Performance with Small Datasets.","year":2024,"lang":"en","type":"preprint","venue":"ChemRxiv","topic":"Metabolomics and Mass Spectrometry Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"Canada First Research Excellence Fund; Alberta Innovates; Canada Research Chairs","keywords":"Leverage (statistics); Computer science; Domain (mathematical analysis); Machine learning; Artificial intelligence; Domain knowledge; Data mining; Data science; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006121986,0.00128363,0.0009447208,0.001132397,0.0006276138,0.002256088,0.001210362,0.001845582,0.001366183],"category_scores_gemma":[0.01480132,0.0005799776,0.0009315152,0.001529218,0.001446648,0.002989284,0.001838514,0.003310127,0.0009021337],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008914159,"about_ca_system_score_gemma":0.001086681,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001928601,"about_ca_topic_score_gemma":0.001945111,"domain_scores_codex":[0.9979547,0.001186431,0.00008026517,0.0004573383,0.000260074,0.00006117269],"domain_scores_gemma":[0.99239,0.005978792,0.0002486395,0.0009928229,0.0002514195,0.0001383279],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009506917,0.0007442499,0.01871316,0.001457358,0.0009400989,0.0003954627,0.0002928874,0.6070219,0.02540223,0.02397431,0.01869748,0.3014101],"study_design_scores_gemma":[0.00006930637,0.0001207579,0.003689056,0.00007694816,0.00005480902,0.00009457631,0.00008001114,0.9137161,0.01142453,0.06003373,0.01060286,0.00003736261],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.143888,0.0112574,0.8196304,0.007213214,0.0003636746,0.0003171751,0.005604472,0.004030633,0.00769484],"genre_scores_gemma":[0.5788394,0.003772338,0.403286,0.001044906,0.0004426712,0.0003672786,0.01036376,0.0003081549,0.001575551],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006121986,"threshold_uncertainty_score":0.03237659,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02500619470500674,"score_gpt":0.2590000551128164,"score_spread":0.2339938604078096,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}