{"id":"W4224060344","doi":"10.1016/j.jmb.2022.167589","title":"MarkerML – Marker Feature Identification in Metagenomic Datasets Using Interpretable Machine Learning","year":2022,"lang":"en","type":"article","venue":"Journal of Molecular Biology","topic":"Gene expression and cancer classification","field":"Biochemistry, Genetics and Molecular Biology","cited_by":9,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"Council of Scientific and Industrial Research, India; TDC Research; Tata Consultancy Services","keywords":"Machine learning; Identification (biology); Artificial intelligence; Computer science; Context (archaeology); Random forest; Leverage (statistics); Metagenomics; Microbiome; Data science; Biology; Bioinformatics; Ecology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008372947,0.0001130824,0.000178403,0.0002103257,0.00008780064,0.00001673208,0.0003146923,0.00008978938,0.0001014479],"category_scores_gemma":[0.0001066547,0.0001084532,0.0001102741,0.0001707005,0.00003243071,0.000006689524,0.0002163003,0.0003512523,0.000001106712],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006993944,"about_ca_system_score_gemma":0.0001047063,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001085692,"about_ca_topic_score_gemma":0.000004204632,"domain_scores_codex":[0.9985191,0.000599084,0.0003655977,0.0002327439,0.0001183542,0.0001650845],"domain_scores_gemma":[0.9992009,0.000008847923,0.0004475748,0.0002388566,0.00005418093,0.00004957739],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.000260999,0.00004754347,0.004168484,0.000005480867,0.00005510892,0.00001709831,0.000020637,0.00282817,0.9898757,0.00002834651,0.001010746,0.001681702],"study_design_scores_gemma":[0.00249833,0.000864194,0.01009771,0.00003254184,0.0001267872,0.0009806008,0.0004870938,0.005809573,0.505512,0.0004273291,0.4726407,0.0005231493],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9731613,0.006294689,0.01945778,0.0003493906,0.0004402041,0.0001184027,0.00006128645,0.000003146171,0.000113734],"genre_scores_gemma":[0.9977201,0.0001783544,0.001154836,0.0002430663,0.00005557243,0.000008880378,0.0003921517,0.00001767565,0.0002294168],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4843637,"threshold_uncertainty_score":0.4422594,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01001247581386115,"score_gpt":0.2880977949613673,"score_spread":0.2780853191475062,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}