{"id":"W4292388628","doi":"10.1101/2022.08.19.504330","title":"Implications of taxonomic bias for microbial differential-abundance analysis","year":2022,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Gut microbiota and health","field":"Biochemistry, Genetics and Molecular Biology","cited_by":25,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"National Institute of General Medical Sciences; U.S. Department of Energy; Office of Science; National Science Foundation; National Institutes of Health; Research Nova Scotia","keywords":"Interpretability; Relative species abundance; Abundance (ecology); Commit; Microbiome; Computer science; Ecology; Data science; Statistics; Data mining; Biology; Bioinformatics; Artificial intelligence; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0003069536,0.0003736497,0.0006150766,0.0002739737,0.0002089584,0.00005495063,0.0006651039,0.000392549,0.0001134647],"category_scores_gemma":[0.00007041468,0.0004361862,0.000547429,0.0003782134,0.0001057546,0.000003651841,0.0006356237,0.0002918136,0.000003515279],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001240584,"about_ca_system_score_gemma":0.0007287383,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007164216,"about_ca_topic_score_gemma":0.00002954128,"domain_scores_codex":[0.997815,0.0001055985,0.0006002858,0.0009588436,0.00009723309,0.0004231011],"domain_scores_gemma":[0.9974086,0.00002616585,0.0006651469,0.001453721,0.0003075387,0.000138771],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00005545757,0.0001488096,0.01183464,0.0002644587,0.0007894112,3.505594e-7,0.000002832156,0.000181322,0.9848766,0.0002579928,0.001587137,0.000001036525],"study_design_scores_gemma":[0.0006821898,0.0001524171,0.3294158,0.00003920321,0.001061367,9.194848e-9,0.000003007027,0.00006120354,0.6223834,0.000002114725,0.04547662,0.0007226682],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9784117,0.0007914845,0.01507554,0.0002033377,0.0005299231,0.0009601255,0.00397969,0.00004047395,0.000007729885],"genre_scores_gemma":[0.9934931,0.000373953,0.005078894,0.0001540466,0.000317495,0.0004491665,0.00003315517,0.00007888838,0.00002134552],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3624931,"threshold_uncertainty_score":0.999809,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02290430979130305,"score_gpt":0.2542282694697396,"score_spread":0.2313239596784366,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}