{"id":"W2792286521","doi":"10.1101/270157","title":"Best Practices for Benchmarking Germline Small Variant Calls in Human Genomes","year":2018,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Genomics and Rare Diseases","field":"Biochemistry, Genetics and Molecular Biology","cited_by":58,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ontario Institute for Cancer Research","funders":"National Institute of Standards and Technology","keywords":"Benchmarking; Computer science; Context (archaeology); False positive paradox; Data science; Best practice; Data mining; Standardization; Machine learning; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0006498861,0.0005207659,0.0004652483,0.0001889643,0.00019084,0.0002047383,0.0006472708,0.0006702074,0.00002447398],"category_scores_gemma":[0.0002983293,0.000564452,0.000225409,0.0001337424,0.000117648,0.000008056739,0.0006815162,0.000306168,0.00000960166],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009371906,"about_ca_system_score_gemma":0.0006696054,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002104774,"about_ca_topic_score_gemma":0.0001042507,"domain_scores_codex":[0.9973239,0.00009901109,0.0005756478,0.001251119,0.0001536925,0.0005965704],"domain_scores_gemma":[0.9973031,0.00003594999,0.0008481522,0.001118414,0.0004687045,0.0002256965],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00009205059,0.0003712011,0.004850887,0.0004642277,0.0002350031,0.00008167139,0.000007572071,0.0001016315,0.9932622,0.0001542806,0.0003706188,0.000008694523],"study_design_scores_gemma":[0.003632986,0.00137773,0.06467171,0.001125646,0.0008024412,4.156529e-7,0.00002172597,0.001063342,0.6869454,0.00003341051,0.2361412,0.004183938],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9920232,0.004237247,0.001105721,0.00009940747,0.0008453495,0.001028422,0.0005952642,0.00003737166,0.00002802371],"genre_scores_gemma":[0.9871702,0.0008290684,0.008904158,0.0002115505,0.002347177,0.0003521694,0.00001647857,0.0001385888,0.00003064713],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3063168,"threshold_uncertainty_score":0.9996807,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0279454101265618,"score_gpt":0.2696338652257068,"score_spread":0.241688455099145,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}