{"id":"W2908220830","doi":"10.7717/peerj.6142","title":"Fully automated sequence alignment methods are comparable to, and much faster than, traditional methods in large data sets: an example with hepatitis B virus","year":2019,"lang":"en","type":"article","venue":"PeerJ","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Phylogenetic tree; Multiple sequence alignment; Alignment-free sequence analysis; Computer science; Sequence alignment; Sequence (biology); Tree (set theory); Genome; Computational biology; Inference; Data mining; Biology; Artificial intelligence; Genetics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008852906,0.001831986,0.001881825,0.00227449,0.001888698,0.003892835,0.001736734,0.00169947,0.004445515],"category_scores_gemma":[0.03056107,0.0008528442,0.001678692,0.005012883,0.001172765,0.005887576,0.001945931,0.002562339,0.003755455],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009695787,"about_ca_system_score_gemma":0.001454255,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003937125,"about_ca_topic_score_gemma":0.006894521,"domain_scores_codex":[0.988506,0.00469712,0.0008850432,0.002350741,0.003229479,0.0003315331],"domain_scores_gemma":[0.9775843,0.01233276,0.001918471,0.003545705,0.004241848,0.0003769629],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001996365,0.0005640661,0.03594374,0.003211813,0.002296049,0.0004393464,0.001946171,0.05567474,0.1020074,0.01632638,0.02650849,0.7530855],"study_design_scores_gemma":[0.0004430095,0.001832026,0.103642,0.001119689,0.0007918891,0.002058394,0.001778489,0.5021276,0.1128057,0.08126873,0.1910256,0.001106817],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2992751,0.01333227,0.6488669,0.002377892,0.000878304,0.0005883841,0.006758153,0.0136243,0.01429871],"genre_scores_gemma":[0.228122,0.002592972,0.7560444,0.0004411995,0.0001738745,0.0004431832,0.007650197,0.002116374,0.002415687],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008852906,"threshold_uncertainty_score":0.04681915,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1235077244675482,"score_gpt":0.3777330263138278,"score_spread":0.2542253018462797,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}