{"id":"W3041800561","doi":"10.1073/pnas.1903436117","title":"Mismatch-tolerant, alignment-free sequence classification using multiple spaced seeds and multiindex Bloom filters","year":2020,"lang":"en","type":"article","venue":"Proceedings of the National Academy of Sciences","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":17,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia; Canada's Michael Smith Genome Sciences Centre","funders":"National Human Genome Research Institute; Genome British Columbia","keywords":"Bloom filter; Bloom; Sequence (biology); Multiple sequence alignment; Computer science; Biology; Artificial intelligence; Speech recognition; Computational biology; Sequence alignment; Genetics; Algorithm; Peptide sequence; Gene","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003005615,0.0008170475,0.0009912121,0.002062305,0.001168031,0.001984983,0.002208987,0.001609448,0.002001756],"category_scores_gemma":[0.01069606,0.0008151811,0.0008375491,0.002258013,0.0007266887,0.003810101,0.001412957,0.001278504,0.001794367],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002418882,"about_ca_system_score_gemma":0.003021193,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004938039,"about_ca_topic_score_gemma":0.01057771,"domain_scores_codex":[0.9973416,0.0004782721,0.0003355893,0.0007226688,0.0009544707,0.0001674079],"domain_scores_gemma":[0.9939836,0.002348844,0.000792659,0.001593685,0.001097203,0.0001839727],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00174915,0.0003299235,0.01865544,0.0005784903,0.0002282143,0.0002484082,0.0007317445,0.07117906,0.2125773,0.02552983,0.01208996,0.6561025],"study_design_scores_gemma":[0.00008242052,0.000222357,0.002831262,0.00006955498,0.00004487151,0.0003469143,0.0001287677,0.8024366,0.1490861,0.02997219,0.01467957,0.00009926361],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03081738,0.0002992791,0.9566905,0.0002097564,0.00004126077,0.0001387348,0.0008899337,0.009766905,0.001146146],"genre_scores_gemma":[0.1268617,0.0001336355,0.8689535,0.000191848,0.0000214449,0.0002173986,0.001668318,0.0003232229,0.00162892],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004938039,"threshold_uncertainty_score":0.01755023,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09356698315401978,"score_gpt":0.3006610690749802,"score_spread":0.2070940859209605,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}