{"id":"W2792286521","doi":"10.1101/270157","title":"Best Practices for Benchmarking Germline Small Variant Calls in Human Genomes","year":2018,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Genomics and Rare Diseases","field":"Biochemistry, Genetics and Molecular Biology","cited_by":58,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ontario Institute for Cancer Research","funders":"National Institute of Standards and Technology","keywords":"Benchmarking; Computer science; Context (archaeology); False positive paradox; Data science; Best practice; Data mining; Standardization; Machine learning; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1126929,0.002990839,0.002833443,0.02062016,0.002353312,0.01201213,0.007995514,0.003937539,0.006035587],"category_scores_gemma":[0.3283486,0.002048591,0.004757062,0.01648939,0.002238586,0.004908588,0.008469163,0.004033133,0.006760799],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003508038,"about_ca_system_score_gemma":0.00553527,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01119736,"about_ca_topic_score_gemma":0.009799137,"domain_scores_codex":[0.8122407,0.1008648,0.02395024,0.01868866,0.04200898,0.002246571],"domain_scores_gemma":[0.7707816,0.1047328,0.01480539,0.06385498,0.04333879,0.002486284],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001601319,0.0006813263,0.04819973,0.007061186,0.004744889,0.0009303031,0.003698266,0.1090052,0.02070704,0.05494038,0.1129148,0.6355156],"study_design_scores_gemma":[0.000851296,0.0009758145,0.03821226,0.005633781,0.001439478,0.001932913,0.001676291,0.4203478,0.09516981,0.1771582,0.2553765,0.001225834],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02177451,0.005410479,0.8941488,0.002199954,0.001035605,0.001290281,0.01318072,0.05302005,0.007939654],"genre_scores_gemma":[0.09723484,0.001290069,0.8610973,0.0007866822,0.0002166479,0.001696732,0.02741814,0.008958939,0.001300646],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.1126929,"threshold_uncertainty_score":0.5959841,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0279454101265618,"score_gpt":0.2696338652257068,"score_spread":0.241688455099145,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}