{"id":"W4318193482","doi":"10.1093/bioinformatics/btad057","title":"How to optimally sample a sequence for rapid analysis","year":2023,"lang":"en","type":"article","venue":"Bioinformatics","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"Japan Science and Technology Agency; U.S. National Library of Medicine; Natural Sciences and Engineering Research Council of Canada; National Institutes of Health","keywords":"Sequence (biology); Computer science; Source code; Sampling (signal processing); Sample (material); Sequence analysis; Sequence alignment; Algorithm; Biology; Peptide sequence; Genetics; DNA; Programming language","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001564511,0.0001136344,0.000150473,0.0001143999,0.00009567597,0.00006032978,0.0001780897,0.00005975316,0.000002471783],"category_scores_gemma":[0.0002062109,0.0001037687,0.0001485066,0.0004087079,0.00002556879,0.000001172533,0.0001333505,0.00001909023,0.00001453528],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000007807207,"about_ca_system_score_gemma":0.00003830526,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000004510312,"about_ca_topic_score_gemma":0.00002323595,"domain_scores_codex":[0.9993509,0.000005980386,0.000167463,0.0001399427,0.00008246869,0.000253233],"domain_scores_gemma":[0.9994473,0.00003253659,0.00005809449,0.000284629,0.00009814606,0.00007926452],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000305351,0.00009801931,0.01199743,0.0004652441,0.005573063,0.00000323368,0.00400714,0.02907521,0.693458,0.00152683,0.1361931,0.1172974],"study_design_scores_gemma":[0.0008034395,0.001038294,0.008400133,0.000009469387,0.0004517578,0.000003631277,0.001764919,0.05127095,0.0426251,0.0003380929,0.8925115,0.0007827347],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7489662,0.0002663789,0.2450099,0.002361058,0.0002810531,0.000907314,0.001372897,0.00003817172,0.0007970823],"genre_scores_gemma":[0.7550462,0.0003684284,0.241384,0.001073444,0.0002022441,0.0001606807,0.0007083431,0.00002391518,0.001032702],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7563184,"threshold_uncertainty_score":0.4231564,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03516069343064127,"score_gpt":0.2718266799308762,"score_spread":0.236665986500235,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}