{"id":"W6939411770","doi":"10.6084/m9.figshare.13364270","title":"Additional file 5 of Systematic evaluation of supervised machine learning for sample origin prediction using metagenomic sequencing data","year":2020,"lang":"en","type":"article","venue":"Figshare","topic":"Forensic and Genetic Research","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Public Health Agency of Canada","funders":"","keywords":"Sample (material); Random forest; Metagenomics; Geographic coordinate system; Sample mean and sample covariance; Longitude; Supervised learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.007529639,0.001821983,0.001681708,0.003253379,0.001232856,0.001916267,0.003423433,0.001534881,0.7849209],"category_scores_gemma":[0.07563155,0.0008855522,0.0022159,0.004448263,0.0005331887,0.002361753,0.001632089,0.001343224,0.1401436],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001342513,"about_ca_system_score_gemma":0.00305793,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007772759,"about_ca_topic_score_gemma":0.02006617,"domain_scores_codex":[0.9974744,0.0007196332,0.000396021,0.000611898,0.0005981601,0.0002000268],"domain_scores_gemma":[0.9188544,0.06859673,0.002195677,0.003446487,0.006024212,0.0008824808],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001026428,0.000165852,0.004737628,0.008370812,0.0003017189,0.0001110089,0.0001230925,0.001797736,0.0003972163,0.0007168054,0.9664544,0.01579718],"study_design_scores_gemma":[0.02041535,0.001380262,0.06434672,0.009728459,0.001588255,0.001061775,0.0009016535,0.02874936,0.006184923,0.02221081,0.842776,0.0006564171],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.000364439,0.00003053617,0.001011523,0.00008662784,0.00003810834,0.0001710674,0.9961985,0.001697998,0.0004011946],"genre_scores_gemma":[0.01283028,0.0001334453,0.02114115,0.0005725549,0.0001288384,0.003328064,0.9535407,0.002902202,0.005422726],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7849209,"threshold_uncertainty_score":0.3067842,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3184958879889506,"score_gpt":0.3577501234210463,"score_spread":0.03925423543209577,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}