{"id":"W2120895683","doi":"10.1186/1471-2105-12-8","title":"Flexible taxonomic assignment of ambiguous sequencing reads","year":2011,"lang":"en","type":"article","venue":"BMC Bioinformatics","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":32,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Institute of Genetics","keywords":"Metagenomics; Computer science; Set (abstract data type); False positive paradox; Data mining; Computational biology; Matching (statistics); Phylogenetic tree; Biology; Information retrieval; Artificial intelligence; Statistics; Genetics; Mathematics; Gene","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000150311,0.0001282988,0.0001536102,0.00003692467,0.00004730259,0.000007101367,0.0001726687,0.00008607834,0.00001668169],"category_scores_gemma":[0.00002391977,0.0001165983,0.00008119791,0.00004473682,0.00007833375,0.000001396982,0.0001381156,0.00003758718,0.00001739816],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002060739,"about_ca_system_score_gemma":0.0001067271,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005758119,"about_ca_topic_score_gemma":0.00001788607,"domain_scores_codex":[0.9992523,0.0000128161,0.0003544921,0.0001088717,0.00007893569,0.0001925704],"domain_scores_gemma":[0.9993821,0.00000454915,0.0001775742,0.0003308026,0.00005303007,0.00005195473],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0004061924,0.0003118805,0.05043782,0.0009042756,0.0006902136,0.000002836388,0.006354378,0.002944452,0.9001698,0.003407267,0.006486344,0.02788449],"study_design_scores_gemma":[0.0008989128,0.000863075,0.02013887,0.00003749692,0.00007225456,0.00002499014,0.002563742,0.003585581,0.9603426,0.0003895792,0.01056043,0.0005223918],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8807783,0.0005437234,0.07073376,0.00000568317,0.0002566512,0.0002942971,0.00004656133,0.0000100015,0.04733104],"genre_scores_gemma":[0.8700585,0.0001620763,0.1293368,0.00007699159,0.00004888615,0.0000136959,0.00001581327,0.00001214041,0.0002751237],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.06017282,"threshold_uncertainty_score":0.4754743,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.062467156920103,"score_gpt":0.231665825668583,"score_spread":0.1691986687484799,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}