{"id":"W2517033688","doi":"10.1109/tcbb.2016.2598752","title":"Machine Learned Replacement of N-Labels for Basecalled Sequences in DNA Barcoding","year":2016,"lang":"en","type":"article","venue":"IEEE/ACM Transactions on Computational Biology and Bioinformatics","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Guelph","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Foundation for Innovation; Ontario Ministry of Research, Innovation and Science; Government of Canada; Ontario Genomics Institute; Genome Canada","keywords":"Barcode; Correctness; DNA barcoding; Word error rate; Computer science; Sequence (biology); Sequence database; Sanger sequencing; Biology; Automation; Artificial intelligence; Computational biology; DNA sequencing; Pattern recognition (psychology); DNA; Genetics; Algorithm; Gene; Evolutionary biology; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002128524,0.000780746,0.0006438328,0.0009475788,0.0008741845,0.0007271203,0.001842711,0.001066893,0.00132582],"category_scores_gemma":[0.01334211,0.0003194874,0.0005543958,0.0009185834,0.0007826986,0.001721928,0.0007738147,0.001670087,0.001447613],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008371897,"about_ca_system_score_gemma":0.001155163,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003160991,"about_ca_topic_score_gemma":0.004611759,"domain_scores_codex":[0.9973718,0.0007297779,0.0001761902,0.0009242684,0.0006787976,0.0001191319],"domain_scores_gemma":[0.990916,0.004187429,0.001047246,0.002066488,0.001649976,0.0001329248],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006123384,0.0003003353,0.01204875,0.0002512451,0.00007693325,0.0001771784,0.0004087913,0.0603403,0.04929319,0.004963703,0.003601716,0.8679256],"study_design_scores_gemma":[0.00004742075,0.0002194008,0.003908526,0.00009075089,0.00006542591,0.0003710599,0.00009292798,0.8523666,0.1195143,0.01093015,0.01229849,0.00009490208],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09458275,0.0004894462,0.8958903,0.0002826513,0.0002629143,0.00009991927,0.0003537746,0.006437139,0.001601064],"genre_scores_gemma":[0.2985003,0.000141333,0.6958491,0.0003227509,0.00009076463,0.000129862,0.0009809412,0.0005625977,0.003422369],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003160991,"threshold_uncertainty_score":0.01125687,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02494258416889964,"score_gpt":0.2860665525876618,"score_spread":0.2611239684187621,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}