{"id":"W4407670178","doi":"10.1101/2025.02.13.638109","title":"Commonly used compositional data analysis implementations are not advantageous in microbial differential abundance analyses benchmarked against biological ground truth","year":2025,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Institute of Population and Public Health","funders":"","keywords":"Ground truth; Implementation; Abundance (ecology); Differential (mechanical device); Computer science; Environmental science; Data mining; Ecology; Biology; Artificial intelligence; Physics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02319468,0.002266907,0.001348066,0.002340118,0.001401293,0.004278326,0.002474855,0.001790993,0.002190509],"category_scores_gemma":[0.06733341,0.0006050288,0.002457745,0.002086098,0.001972894,0.003180559,0.003301055,0.00279175,0.001953296],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001149744,"about_ca_system_score_gemma":0.00178531,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002494477,"about_ca_topic_score_gemma":0.004075068,"domain_scores_codex":[0.9868868,0.005427729,0.001078823,0.003072964,0.003071555,0.0004622095],"domain_scores_gemma":[0.975585,0.01377358,0.001807103,0.005322739,0.003062089,0.0004495977],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003353912,0.001003903,0.1612429,0.00536592,0.004078068,0.0004254649,0.001494549,0.2253419,0.1329926,0.02808875,0.02621752,0.4103943],"study_design_scores_gemma":[0.0002727789,0.0009702867,0.03847124,0.0006694857,0.0004336836,0.0005981937,0.0009598924,0.7638643,0.1030468,0.0532494,0.03713335,0.0003305514],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3045487,0.004500083,0.6479839,0.002043493,0.001119311,0.0004203107,0.01020429,0.02182773,0.007352141],"genre_scores_gemma":[0.5120506,0.0006763038,0.465328,0.001365519,0.0001401436,0.0006213672,0.01589515,0.003020823,0.0009020954],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9768053,"threshold_uncertainty_score":0.1226667,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06003124855312948,"score_gpt":0.3039418420017381,"score_spread":0.2439105934486087,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}