{"id":"W3162676821","doi":"10.1101/2021.05.13.444008","title":"DeLUCS: Deep Learning for Unsupervised Clustering of DNA Sequences","year":2021,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University; University of Waterloo","funders":"","keywords":"Cluster analysis; Biology; DNA sequencing; Genome; Computational biology; Homology (biology); Identifier; Sequence (biology); Taxonomic rank; Genetics; Pattern recognition (psychology); Artificial intelligence; DNA; Computer science; Gene","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001457138,0.002048542,0.001367059,0.00233554,0.001026438,0.001486136,0.004200706,0.002122333,0.005940686],"category_scores_gemma":[0.003601508,0.001021364,0.0016769,0.001792127,0.00141097,0.001684042,0.002697469,0.002978721,0.003339442],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002342816,"about_ca_system_score_gemma":0.00239538,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01818784,"about_ca_topic_score_gemma":0.02288765,"domain_scores_codex":[0.9988021,0.000241896,0.0000698187,0.0004204048,0.0003223428,0.0001435682],"domain_scores_gemma":[0.9986981,0.0003907158,0.0001156209,0.0003269777,0.0003658144,0.0001026831],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003480377,0.000198187,0.002358273,0.0003048757,0.000233174,0.0001558122,0.0001650099,0.340071,0.01063302,0.01511004,0.03303155,0.5973909],"study_design_scores_gemma":[0.00001303054,0.00001705723,0.0001096888,0.00001007623,0.000004561367,0.00001603338,0.00001229476,0.9901136,0.002312659,0.005339364,0.002043111,0.000008608733],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01151616,0.0004126577,0.9669366,0.0002718013,0.0001208593,0.0001482068,0.0009913447,0.01808667,0.001515753],"genre_scores_gemma":[0.1425686,0.0002483378,0.8407905,0.0006742883,0.0001080651,0.0004455715,0.007786028,0.00126389,0.006114728],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01818784,"threshold_uncertainty_score":0.03616399,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01464032369067582,"score_gpt":0.2222306482535994,"score_spread":0.2075903245629236,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}