{"id":"W4388839981","doi":"10.1101/2023.11.20.567879","title":"Metagenome profiling and containment estimation through abundance-corrected k-mer sketching with sylph","year":2023,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Metagenomics; Genome; RefSeq; Computational biology; Profiling (computer programming); Computer science; Biology; Genetics; Gene","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002469126,0.001345446,0.0012606,0.002173395,0.0007785525,0.002223777,0.001661955,0.001351487,0.01045385],"category_scores_gemma":[0.01156893,0.001252151,0.001442098,0.001666826,0.0006974569,0.002462878,0.002302711,0.002145177,0.007151712],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000540406,"about_ca_system_score_gemma":0.0009749468,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001365829,"about_ca_topic_score_gemma":0.003256637,"domain_scores_codex":[0.9983957,0.0002969786,0.0001171152,0.0005835729,0.0005120882,0.00009448464],"domain_scores_gemma":[0.9968011,0.001415139,0.000356682,0.0009216216,0.0003718314,0.0001335935],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002863469,0.000302844,0.02606422,0.00297203,0.0008786637,0.0009186301,0.001963926,0.06065372,0.2433425,0.02104709,0.06351852,0.5754744],"study_design_scores_gemma":[0.0002194015,0.0003210574,0.008909021,0.0001961611,0.0001193901,0.0007301318,0.0004177649,0.7429727,0.158861,0.03285962,0.05412926,0.0002644898],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03717953,0.0007081353,0.8595738,0.0003091145,0.0001968023,0.0001872834,0.01302775,0.08636847,0.002449034],"genre_scores_gemma":[0.09617379,0.0002789379,0.8789996,0.0002192737,0.00006012106,0.0003145746,0.01486779,0.007130642,0.001955212],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01045385,"threshold_uncertainty_score":0.03497159,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01515476721249977,"score_gpt":0.2256776294218983,"score_spread":0.2105228622093985,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}