{"id":"W4416373688","doi":"10.1016/j.isci.2025.114029","title":"Run-length compressed metagenomic read classification with SMEM-finding and tagging","year":2025,"lang":"en","type":"article","venue":"iScience","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"National Human Genome Research Institute; Vlaamse regering; Natural Sciences and Engineering Research Council of Canada; Fonds Wetenschappelijk Onderzoek; National Institutes of Health; National Science Foundation","keywords":"Metagenomics; Identifier; Task (project management); Class (philosophy); Identification (biology); Pattern recognition (psychology)","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001326285,0.00008153462,0.00008069155,0.00003706827,0.000186787,0.0000379101,0.0001392644,0.00003098158,0.000001748448],"category_scores_gemma":[0.00002019983,0.00006714183,0.00001493537,0.00009903118,0.0001581584,0.000001153417,0.000102394,0.00003396668,0.000001095839],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000007682187,"about_ca_system_score_gemma":0.00004651963,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001422196,"about_ca_topic_score_gemma":0.000020454,"domain_scores_codex":[0.9994025,0.00001621,0.00008768791,0.0002954425,0.00005528444,0.0001428718],"domain_scores_gemma":[0.9996933,0.0000129561,0.00004333216,0.0001892822,0.00003290688,0.0000282022],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","study_design_scores_codex":[0.000012683,0.000009239035,0.008000208,0.000008036175,0.00002086822,3.540039e-7,0.00007560631,0.00007474112,0.9842858,0.001014453,0.0001104927,0.006387537],"study_design_scores_gemma":[0.001031457,0.0002889991,0.5263042,0.00006475788,0.00008604682,0.00001401989,0.001027209,0.003222564,0.34758,0.0005206382,0.1193437,0.0005164202],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9890427,0.002218596,0.00267075,0.0003461118,0.00007865929,0.0001028012,0.000004331856,0.00000370865,0.005532343],"genre_scores_gemma":[0.997097,0.0002902159,0.00176541,0.0002072269,0.00002380197,0.00001134945,0.000003343641,0.000004332294,0.0005973412],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6367058,"threshold_uncertainty_score":0.2737964,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02076258872854633,"score_gpt":0.2647410485507156,"score_spread":0.2439784598221692,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}