{"id":"W4224060344","doi":"10.1016/j.jmb.2022.167589","title":"MarkerML – Marker Feature Identification in Metagenomic Datasets Using Interpretable Machine Learning","year":2022,"lang":"en","type":"article","venue":"Journal of Molecular Biology","topic":"Gene expression and cancer classification","field":"Biochemistry, Genetics and Molecular Biology","cited_by":9,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"Council of Scientific and Industrial Research, India; TDC Research; Tata Consultancy Services","keywords":"Machine learning; Identification (biology); Artificial intelligence; Computer science; Context (archaeology); Random forest; Leverage (statistics); Metagenomics; Microbiome; Data science; Biology; Bioinformatics; Ecology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002300943,0.001371154,0.0007788906,0.002588897,0.0005927238,0.001792101,0.001119461,0.001085253,0.002029561],"category_scores_gemma":[0.00785493,0.0003913103,0.001658054,0.001789146,0.0003952022,0.001592827,0.001474925,0.001301196,0.001466455],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005349065,"about_ca_system_score_gemma":0.0008232202,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001621449,"about_ca_topic_score_gemma":0.002942706,"domain_scores_codex":[0.9986483,0.0003083137,0.0001487811,0.0005516303,0.0002195155,0.0001233819],"domain_scores_gemma":[0.9970285,0.0013471,0.0003491276,0.0008629265,0.0003105795,0.0001017871],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003902747,0.0009591936,0.1107084,0.002129397,0.002033243,0.001319248,0.000672242,0.06758411,0.1610282,0.00629462,0.0459263,0.5974423],"study_design_scores_gemma":[0.0002740399,0.0007188357,0.05021897,0.000224234,0.0005166485,0.0007398823,0.0003176779,0.7652738,0.1138984,0.02688673,0.04075386,0.0001768864],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3197049,0.001519714,0.4955551,0.0009338433,0.0003835944,0.0003448051,0.09663194,0.08265542,0.002270665],"genre_scores_gemma":[0.451124,0.0002852747,0.4220244,0.0003157122,0.00008815874,0.0005363648,0.1227408,0.001801776,0.001083654],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.002588897,"threshold_uncertainty_score":0.01216871,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01001247581386115,"score_gpt":0.2880977949613673,"score_spread":0.2780853191475062,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}