{"id":"W4399276129","doi":"10.1101/2024.05.29.596415","title":"Lineage-specific microbial protein prediction enables large-scale exploration of protein ecology within the human gut","year":2024,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Gut microbiota and health","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"McMaster University","funders":"","keywords":"Metagenomics; Biology; Computational biology; Spurious relationship; Lineage (genetic); Microbiome; Genome; Gene; Evolutionary biology; Ecology; Genetics; Machine learning; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009433105,0.0004171748,0.0003633976,0.000837811,0.0002996723,0.0009169949,0.0003094285,0.0004054822,0.001612876],"category_scores_gemma":[0.001346596,0.0002821952,0.0004384684,0.0005525542,0.0002227544,0.0005601195,0.0008502201,0.0004876512,0.0006705668],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002603271,"about_ca_system_score_gemma":0.0004242077,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001338544,"about_ca_topic_score_gemma":0.002311646,"domain_scores_codex":[0.999711,0.0000833024,0.00001267262,0.0001068953,0.00006081134,0.00002543266],"domain_scores_gemma":[0.9995857,0.0001990193,0.00006900182,0.00005751403,0.00004774833,0.00004102212],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","study_design_scores_codex":[0.001379787,0.0002386405,0.1245728,0.0003625982,0.0002525493,0.000449288,0.0002750015,0.06100136,0.6570927,0.003109263,0.003808782,0.1474573],"study_design_scores_gemma":[0.00004603883,0.000188605,0.04515887,0.00003990387,0.00006513094,0.0004450722,0.0001761968,0.7824349,0.1601438,0.005671765,0.005577107,0.00005247651],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7712009,0.0006680245,0.2189367,0.000366093,0.00002862544,0.00003266251,0.003192491,0.004281473,0.00129308],"genre_scores_gemma":[0.7744625,0.0002524334,0.2209798,0.0001114926,0.00002218425,0.00003405758,0.002883181,0.0003499144,0.0009045185],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.001612876,"threshold_uncertainty_score":0.005395591,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01505157386639813,"score_gpt":0.2303637787117178,"score_spread":0.2153122048453197,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}