{"id":"W4408068552","doi":"10.1101/2025.02.25.640119","title":"Run-length compressed metagenomic read classification with SMEM-finding and tagging","year":2025,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Metagenomics; Computer science; Computational biology; Information retrieval; Artificial intelligence; Biology; Genetics; Gene","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008065988,0.0009022756,0.0009918115,0.002213335,0.0006566353,0.001630338,0.001953143,0.001131579,0.003789817],"category_scores_gemma":[0.00527728,0.0003776943,0.0009648293,0.002563577,0.0005861514,0.00176052,0.001451698,0.00107458,0.003283734],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007319064,"about_ca_system_score_gemma":0.001632501,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003146883,"about_ca_topic_score_gemma":0.005274427,"domain_scores_codex":[0.9991605,0.00007496423,0.00007834979,0.0002810104,0.0003169138,0.00008822194],"domain_scores_gemma":[0.9975842,0.0006705109,0.0002852589,0.0007553726,0.0005437529,0.0001609057],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00221371,0.0004223025,0.02288791,0.0007838079,0.0001868726,0.0005250111,0.000677656,0.03864811,0.1521206,0.006823466,0.02677272,0.747938],"study_design_scores_gemma":[0.0001288366,0.0002745293,0.00806809,0.00008436172,0.00006346792,0.0004628912,0.0003543376,0.8394626,0.121264,0.0137445,0.01599233,0.0001001813],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1516615,0.0008505729,0.7766705,0.0004739075,0.0003064777,0.0002554361,0.006446516,0.06045826,0.002876826],"genre_scores_gemma":[0.2543773,0.0001639301,0.725163,0.000321955,0.0001285514,0.0002797674,0.01448963,0.001638148,0.003437761],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003789817,"threshold_uncertainty_score":0.01267821,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01824713739215438,"score_gpt":0.2281811957736552,"score_spread":0.2099340583815008,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}