{"id":"W2951725148","doi":"10.1371/journal.pone.0206409","title":"An open-source k-mer based machine learning tool for fast and accurate subtyping of HIV-1 genomes","year":2018,"lang":"en","type":"article","venue":"PLoS ONE","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":99,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo; Western University","funders":"Natural Sciences and Engineering Research Council of Canada; Genome Canada; Ontario Genomics; Canadian Institutes of Health Research; Government of Canada; Ontario Genomics Institute","keywords":"Subtyping; Computer science; Workflow; Software; Machine learning; Artificial intelligence; Data mining; False positive paradox; Database","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001466974,0.00010257,0.0001693028,0.00002111302,0.0001390476,0.00002850622,0.0001675916,0.00005156525,0.00001362422],"category_scores_gemma":[0.00005023389,0.00009862115,0.0000228441,0.00003046886,0.00008779007,0.000001449795,0.0001480349,0.00003468377,0.000001562707],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000003101153,"about_ca_system_score_gemma":0.00002363263,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001338463,"about_ca_topic_score_gemma":0.00002378665,"domain_scores_codex":[0.9993868,0.00002811316,0.000139309,0.0002429784,0.00005123445,0.0001515707],"domain_scores_gemma":[0.9995534,0.00001673853,0.00008225733,0.0001893206,0.0001195685,0.00003869856],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0001388006,0.0001515831,0.02435672,0.00004373578,0.0001382048,9.045447e-8,0.0001062253,0.00005622547,0.9730154,0.00001216002,0.00002454343,0.001956311],"study_design_scores_gemma":[0.001236682,0.00171693,0.01119884,0.00003576036,0.0001066866,8.722051e-7,0.00009636985,0.01041059,0.9678402,0.00003476588,0.007052578,0.0002697211],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9960064,0.0009925694,0.002396649,0.00008774699,0.00001366192,0.0002731063,0.000070296,0.000003310931,0.000156249],"genre_scores_gemma":[0.9900104,0.0001713255,0.009052962,0.0001376973,0.0001490943,0.00002713882,0.00008676243,0.00002348808,0.0003411326],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01315788,"threshold_uncertainty_score":0.4021654,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03275651240184892,"score_gpt":0.2516375323218455,"score_spread":0.2188810199199966,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}