{"id":"W6920730838","doi":"10.6084/m9.figshare.13364270.v1","title":"Additional file 5 of Systematic evaluation of supervised machine learning for sample origin prediction using metagenomic sequencing data","year":2020,"lang":"en","type":"article","venue":"Figshare","topic":"Forensic and Genetic Research","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Public Health Agency of Canada","funders":"","keywords":"Sample (material); Random forest; Metagenomics; Geographic coordinate system; Sample mean and sample covariance; Longitude; Supervised learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0001671809,0.00006674598,0.0001441465,0.00002231196,0.00003932009,0.000007437462,0.0001839588,0.00005406029,0.576437],"category_scores_gemma":[0.01217394,0.00006316034,0.00005136892,0.00006826477,0.000008850042,0.000005734862,0.0001204109,0.00004283302,0.00001339739],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001832659,"about_ca_system_score_gemma":0.0003235984,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001073915,"about_ca_topic_score_gemma":0.000004956546,"domain_scores_codex":[0.999127,0.0001043629,0.0002257608,0.0002038392,0.0002470823,0.00009195958],"domain_scores_gemma":[0.9989893,0.0002625236,0.0001468731,0.000229397,0.000334601,0.00003729741],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003023049,0.000008620314,0.000007117359,0.005114458,0.00009904028,1.32151e-7,0.00004145268,0.005870865,0.04131177,4.792935e-7,0.9471538,0.0003620451],"study_design_scores_gemma":[0.0003458387,0.0001835382,0.00004831946,0.002168607,0.00007121443,0.000002694528,0.00009961061,0.9548162,0.01344139,0.000008433259,0.0287333,0.00008085908],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.001625106,0.0002889649,0.0002372786,0.000005062663,0.000007067176,0.0004874507,0.9973029,0.000003822615,0.00004230648],"genre_scores_gemma":[0.05289239,0.000001469939,0.0039864,0.0000127671,0.0000954513,0.0002117295,0.9427623,0.00001159286,0.00002585993],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.9489453,"threshold_uncertainty_score":0.9961469,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3184958879889506,"score_gpt":0.3577501234210463,"score_spread":0.03925423543209577,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}