{"id":"W2558624268","doi":"10.1371/journal.pone.0167047","title":"Simulating Next-Generation Sequencing Datasets from Empirical Mutation and Sequencing Models","year":2016,"lang":"en","type":"article","venue":"PLoS ONE","topic":"Cancer Genomics and Diagnostics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":130,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ontario Institute for Cancer Research","funders":"Ontario Institute for Cancer Research; National Science Foundation","keywords":"Computer science; Benchmarking; Software; Set (abstract data type); Data mining; Genome; DNA sequencing; Scripting language; Computational biology; Biology; Genetics; Gene","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00008048689,0.00009567839,0.0001027988,0.00002032203,0.00007247431,0.00004568053,0.00005613115,0.00008365668,0.00001264024],"category_scores_gemma":[0.0002040064,0.00008284259,0.00001800016,0.00002701937,0.00002704569,0.00001683276,0.00005860268,0.00003556881,0.000004229194],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008061105,"about_ca_system_score_gemma":0.00009290569,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001238984,"about_ca_topic_score_gemma":0.000133384,"domain_scores_codex":[0.9992715,0.00002671572,0.0001610671,0.0002954924,0.0001085849,0.000136607],"domain_scores_gemma":[0.9995576,0.00005788627,0.00006213538,0.0002039281,0.00005209275,0.00006639787],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00001272831,0.00002522746,0.0006739349,0.000007215298,0.00004448176,0.000003801713,0.00008611015,0.001125157,0.9955917,0.00002279796,0.0001307808,0.002276047],"study_design_scores_gemma":[0.0005400235,0.000107217,0.0001329826,0.00007691534,0.00006686639,0.000002770393,0.0000643383,0.1194888,0.8783137,0.0009176299,0.00008474603,0.0002039541],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9863746,0.0004324219,0.01244524,0.0001763353,0.00002793509,0.0001084683,0.000340657,0.00001050015,0.00008387212],"genre_scores_gemma":[0.9910279,0.0002214248,0.006977846,0.0003812554,0.0004606572,0.00001032213,0.000882878,0.0000169515,0.00002076248],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1183637,"threshold_uncertainty_score":0.3378223,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.160761244021002,"score_gpt":0.2840296338623527,"score_spread":0.1232683898413507,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}