{"id":"W4386600212","doi":"10.12688/f1000research.140344.1","title":"NCBench: providing an open, reproducible, transparent, adaptable, and continuous benchmark approach for DNA-sequencing-based variant calling","year":2023,"lang":"en","type":"preprint","venue":"F1000Research","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"National Institute of Standards and Technology; Universität Duisburg-Essen; Deutsche Forschungsgemeinschaft","keywords":"Benchmarking; Computer science; Open source; Workflow; Merge (version control); Open science; Data science; Precision and recall; Data mining; Machine learning; Information retrieval; Database; Software; Operating system","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.002884143,0.0004756944,0.0006620553,0.0001749821,0.0004473402,0.0004364731,0.001371077,0.0004821998,0.00000873404],"category_scores_gemma":[0.0004690927,0.0004722008,0.0001415112,0.0001700382,0.0002113431,0.000003706216,0.001935273,0.0004693754,0.000001350051],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007135589,"about_ca_system_score_gemma":0.001198458,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00112619,"about_ca_topic_score_gemma":0.0003354774,"domain_scores_codex":[0.9956114,0.0002592687,0.0005561393,0.002336137,0.0003704219,0.0008666389],"domain_scores_gemma":[0.9974432,0.0001068678,0.0001832561,0.001535569,0.0004540578,0.0002770471],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00169559,0.0006131267,0.006484778,0.00331601,0.001208029,0.00006119339,0.001079805,0.07988674,0.8917709,0.001012004,0.007346534,0.005525279],"study_design_scores_gemma":[0.01178761,0.01058165,0.01980369,0.001302715,0.001084078,0.0001111967,0.005140977,0.3943532,0.456109,0.01392491,0.07818873,0.007612236],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8332004,0.01226271,0.1315566,0.001277527,0.001144522,0.01433786,0.002849256,0.000100776,0.003270323],"genre_scores_gemma":[0.9517782,0.001312816,0.03955007,0.0001099003,0.0007511923,0.002300502,0.002453092,0.0001984182,0.001545766],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4356619,"threshold_uncertainty_score":0.999773,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.212241329283231,"score_gpt":0.3629359888433186,"score_spread":0.1506946595600876,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}