{"id":"W4251502010","doi":"10.21203/rs.3.rs-15502/v1","title":"Systematic evaluation of supervised machine learning for sample origin prediction using metagenomic sequencing data","year":2020,"lang":"en","type":"preprint","venue":"Research Square","topic":"Gut microbiota and health","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Public Health Agency of Canada","funders":"","keywords":"Metagenomics; Machine learning; Sample (material); Computer science; Artificial intelligence; Data mining; Biology; Chromatography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01507531,0.002426644,0.002674166,0.002283534,0.00133634,0.001652656,0.002983798,0.001938101,0.001491162],"category_scores_gemma":[0.03933476,0.0007607333,0.002114559,0.002171037,0.0009891372,0.002094875,0.002169268,0.001482571,0.0009158092],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001072946,"about_ca_system_score_gemma":0.003867412,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006607385,"about_ca_topic_score_gemma":0.01046516,"domain_scores_codex":[0.9902917,0.005570533,0.0007474959,0.001854318,0.001227221,0.0003087154],"domain_scores_gemma":[0.9392062,0.04737239,0.001575382,0.005798821,0.005219444,0.0008276959],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.008632952,0.003164985,0.05267568,0.004777255,0.005861353,0.0005153998,0.0003555738,0.3077759,0.02105234,0.002252752,0.02646753,0.5664684],"study_design_scores_gemma":[0.0006879789,0.0008712608,0.00549348,0.0001242046,0.0007350219,0.0001420693,0.0001433849,0.9787068,0.00941015,0.001990581,0.001650357,0.00004461576],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8288214,0.009871448,0.1342988,0.0008329532,0.0005737891,0.0008710654,0.00894611,0.01333464,0.002449876],"genre_scores_gemma":[0.7363393,0.0009323031,0.231803,0.0003466205,0.0002227472,0.0004398261,0.0276475,0.0009626813,0.001305942],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01507531,"threshold_uncertainty_score":0.07972682,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.402589624165527,"score_gpt":0.4827112108342927,"score_spread":0.08012158666876579,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}