{"id":"W4410975639","doi":"10.1093/gigascience/giaf049","title":"Analysis-ready VCF at Biobank scale using Zarr","year":2025,"lang":"en","type":"article","venue":"GigaScience","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; Ontario Genomics","funders":"National Human Genome Research Institute; Knut och Alice Wallenbergs Stiftelse; Wellcome Trust; Bill and Melinda Gates Foundation","keywords":"Computer science; Biobank; Scalability; Terabyte; Data mining; Big data; Encoding (memory); Field (mathematics); Database; Bioinformatics; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006241309,0.002170504,0.001155611,0.003454211,0.0009739881,0.00480789,0.003611359,0.001620249,0.03403389],"category_scores_gemma":[0.02305689,0.00130119,0.002121917,0.003315662,0.001116091,0.004692843,0.003577434,0.002033686,0.02576317],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002167088,"about_ca_system_score_gemma":0.00245849,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005539694,"about_ca_topic_score_gemma":0.004097695,"domain_scores_codex":[0.9962663,0.0004207046,0.0005938547,0.001048659,0.001279701,0.0003908579],"domain_scores_gemma":[0.9864691,0.004188468,0.001152571,0.00532019,0.002560963,0.0003086836],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00422377,0.0003005279,0.01687938,0.001886273,0.0003912419,0.001102587,0.0009850529,0.01638294,0.03401254,0.04370333,0.6147441,0.2653883],"study_design_scores_gemma":[0.000979872,0.0004228849,0.01625787,0.000759664,0.0002036124,0.001863074,0.0005259782,0.1295169,0.1421626,0.07243493,0.6341299,0.0007426087],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"software","genre_gemma":"methods","genre_scores_codex":[0.01378947,0.0006726906,0.3496481,0.001108746,0.0005412768,0.0004804785,0.08758287,0.5356587,0.01051763],"genre_scores_gemma":[0.1363042,0.001095709,0.4721348,0.002086881,0.0003516169,0.002796797,0.3189627,0.05319219,0.01307512],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.03403389,"threshold_uncertainty_score":0.1138547,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01347740047657392,"score_gpt":0.2740052126660191,"score_spread":0.2605278121894452,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}