{"id":"W2501117253","doi":"10.1186/s12859-016-1097-3","title":"Evaluating the necessity of PCR duplicate removal from next-generation sequencing data and a comparison of approaches","year":2016,"lang":"en","type":"article","venue":"BMC Bioinformatics","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":178,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institute on Aging; National Institute of Biomedical Imaging and Bioengineering; Canadian Institutes of Health Research; National Institutes of Health; Genentech; IXICO; H. Lundbeck A/S; Eisai; Northern California Institute for Research and Education; University of California, San Diego; Pfizer; Biogen; BioClinica; F. Hoffmann-La Roche; Servier; Brigham Young University; University of Southern California; Novartis Pharmaceuticals Corporation; U.S. Department of Defense; Eli Lilly and Company; Bristol-Myers Squibb; Alzheimer's Disease Neuroimaging Initiative; Meso Scale Diagnostics; Alzheimer's Association; Foundation for the National Institutes of Health","keywords":"Transversion; Computational biology; Reference genome; Biology; DNA sequencing; Concordance; Genetics; Exome; Population; Data mining; Computer science; Bioinformatics; Exome sequencing; Gene; Mutation; Medicine","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.08482776,0.001217886,0.001195987,0.003674785,0.002157393,0.003355269,0.002108899,0.001964098,0.001803868],"category_scores_gemma":[0.1639374,0.0006294568,0.003534363,0.002726663,0.001706549,0.002700476,0.002356919,0.001948399,0.0006324131],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003080441,"about_ca_system_score_gemma":0.002909061,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003319571,"about_ca_topic_score_gemma":0.005211914,"domain_scores_codex":[0.937059,0.02559928,0.007450223,0.01050463,0.01799452,0.001392391],"domain_scores_gemma":[0.737797,0.2127131,0.01258821,0.009506389,0.0258533,0.001542025],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.02005261,0.001986943,0.3419904,0.01563008,0.01222701,0.001154966,0.005753607,0.03651098,0.07890572,0.007124733,0.008911291,0.4697517],"study_design_scores_gemma":[0.001413653,0.01598918,0.4049174,0.003630675,0.01268267,0.006247301,0.007333085,0.1882327,0.2758172,0.02436572,0.05810984,0.001260431],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7417684,0.02006547,0.2204788,0.001323596,0.00110095,0.002818519,0.003174371,0.001660847,0.007609065],"genre_scores_gemma":[0.6673804,0.002580829,0.3213092,0.0009171935,0.0001704267,0.001644573,0.004309527,0.0006499085,0.001037899],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9151722,"threshold_uncertainty_score":0.4486174,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3861978566480501,"score_gpt":0.3493385526642588,"score_spread":0.03685930398379128,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}