{"id":"W3045260398","doi":"10.1101/2020.07.24.212712","title":"Benchmarking challenging small variants with linked and long reads","year":2020,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":60,"is_retracted":false,"has_abstract":true,"ca_institutions":"Terry Fox Research Institute; University of British Columbia","funders":"National Institute of Standards and Technology; National Institutes of Health; German Network for Bioinformatics Infrastructure; Bundesministerium für Bildung und Forschung; U.S. National Library of Medicine; Deutsche Forschungsgemeinschaft","keywords":"Benchmarking; Benchmark (surveying); False positive paradox; Indel; False positives and false negatives; Computer science; Computational biology; Set (abstract data type); Data mining; Biology; Artificial intelligence; Genetics; Gene","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0002446467,0.0005198683,0.0004530144,0.00007549642,0.0001816328,0.0001250309,0.0003305706,0.0004318156,0.00000389898],"category_scores_gemma":[0.00006161704,0.0005221672,0.00008535163,0.0001188934,0.0001153958,0.000001505188,0.0009287178,0.0003928826,0.000003456117],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003098752,"about_ca_system_score_gemma":0.0002280735,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002419036,"about_ca_topic_score_gemma":0.00001114955,"domain_scores_codex":[0.9978687,0.00006626561,0.0003126751,0.001152599,0.0001512489,0.0004485747],"domain_scores_gemma":[0.9985474,0.00001795671,0.0002433196,0.000750958,0.0002077712,0.0002325949],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","study_design_scores_codex":[0.00006730019,0.00004044995,0.0218365,0.0003103739,0.0004938518,0.00005892678,0.00002166335,0.0001283564,0.9768634,0.0001230074,0.00004304814,0.00001314212],"study_design_scores_gemma":[0.00136384,0.0005833948,0.8067533,0.0005704259,0.000379357,2.614316e-7,0.00001361398,0.0007203307,0.1823983,0.000006471834,0.005216285,0.001994368],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9894438,0.005832614,0.003481104,0.0002561258,0.0003913215,0.0004517147,0.00008252751,0.0000348566,0.00002596701],"genre_scores_gemma":[0.9885418,0.002629698,0.007490182,0.0002139463,0.000921711,0.00008199843,0.000001189969,0.0001163051,0.000003203331],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7944651,"threshold_uncertainty_score":0.999723,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01625443660525582,"score_gpt":0.2021607567112065,"score_spread":0.1859063201059507,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}