{"id":"W4399555170","doi":"10.1101/2024.06.11.598241","title":"Analysis-ready VCF at Biobank scale using Zarr","year":2024,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"National Institutes of Health; Engineering and Physical Sciences Research Council; Science for Life Laboratory; Robertson Foundation; Bill and Melinda Gates Foundation","keywords":"Computer science; Biobank; Scalability; Terabyte; Big data; Data mining; Encoding (memory); Database; Bioinformatics; Artificial intelligence; Operating system","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007034028,0.001781531,0.00112072,0.003453694,0.0009258214,0.004667528,0.003273115,0.001556781,0.02364144],"category_scores_gemma":[0.02581914,0.001135251,0.001866532,0.002922599,0.00110675,0.004377138,0.003533394,0.001772055,0.0153303],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002056479,"about_ca_system_score_gemma":0.00220507,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005823001,"about_ca_topic_score_gemma":0.004321213,"domain_scores_codex":[0.9960924,0.0005138842,0.0006058334,0.00110203,0.001284405,0.0004013748],"domain_scores_gemma":[0.9831296,0.005617013,0.001301299,0.006677744,0.002920868,0.0003536302],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.005385127,0.0004236927,0.0274024,0.002044654,0.0005060874,0.001527803,0.001544016,0.02954829,0.05189525,0.04562804,0.4623308,0.3717638],"study_design_scores_gemma":[0.001056774,0.0005995741,0.02184234,0.0008379965,0.0002237375,0.001814551,0.0006992561,0.2336727,0.2078271,0.07142863,0.4591057,0.0008916702],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"software","genre_gemma":"methods","genre_scores_codex":[0.02388478,0.0006356651,0.3453571,0.001190375,0.0005131057,0.0004645572,0.06270155,0.5574403,0.007812643],"genre_scores_gemma":[0.2351328,0.0008210482,0.4910192,0.001838009,0.000346798,0.002186405,0.2121125,0.04503542,0.01150788],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.02364144,"threshold_uncertainty_score":0.07908851,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01514163598545507,"score_gpt":0.235462499213503,"score_spread":0.2203208632280479,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}