{"id":"W4382344242","doi":"10.1186/s40168-023-01562-6","title":"MetaPro: a scalable and reproducible data processing and analysis pipeline for metatranscriptomic investigation of microbial communities","year":2023,"lang":"en","type":"article","venue":"Microbiome","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":24,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; Hospital for Sick Children","funders":"Canadian Institutes of Health Research; University of Toronto","keywords":"Pipeline (software); Scalability; Computer science; Visualization; Modular design; Sequence assembly; Annotation; Benchmark (surveying); Metagenomics; Data mining; Computational biology; Biology; Information retrieval; Artificial intelligence; Database; Gene; Cartography","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004666579,0.0001035121,0.0002221997,0.0001365239,0.0001131997,0.0000270789,0.0001535682,0.00005056789,0.00000120888],"category_scores_gemma":[0.00002040213,0.00009930345,0.0000407763,0.0003098591,0.0001856195,0.000002699106,0.0002398183,0.00002589528,2.992287e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000002394366,"about_ca_system_score_gemma":0.00003041921,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00008305217,"about_ca_topic_score_gemma":0.0001481702,"domain_scores_codex":[0.9992837,0.00003040406,0.0002012633,0.0003191782,0.00002732245,0.000138084],"domain_scores_gemma":[0.9993587,0.00001353419,0.00007883011,0.0004399513,0.00008274692,0.00002624453],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00002831769,0.000007466916,0.004350324,0.00013984,0.000306996,6.962135e-8,0.0004123935,0.00001914875,0.9930802,0.000003131298,0.0009019239,0.0007502105],"study_design_scores_gemma":[0.0009071415,0.000108357,0.0127222,0.00002514636,0.001093055,0.000008265013,0.0009377271,0.003424454,0.9684899,0.0001528428,0.01184781,0.0002830882],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9924878,0.005545113,0.0008932072,0.0001887468,0.00003200505,0.0001909884,0.0006483577,0.000005214049,0.000008493615],"genre_scores_gemma":[0.9919712,0.0006034413,0.005407545,0.00006128359,0.00003707269,0.00001339309,0.001634662,0.00001363337,0.0002577569],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02459026,"threshold_uncertainty_score":0.4049477,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05093893882885635,"score_gpt":0.2775620213990623,"score_spread":0.226623082570206,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}