{"id":"W4386554491","doi":"10.1093/bioinformatics/btad552","title":"μ- PBWT: a lightweight r-indexing of the PBWT for storing and querying UK Biobank data","year":2023,"lang":"en","type":"article","venue":"Bioinformatics","topic":"Genetic Associations and Epidemiology","field":"Biochemistry, Genetics and Molecular Biology","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"National Human Genome Research Institute; European Commission; Natural Sciences and Engineering Research Council of Canada; Japan Society for the Promotion of Science; National Institutes of Health; National Science Foundation","keywords":"Computer science; Biobank; Search engine indexing; Leverage (statistics); Set (abstract data type); Data structure; Haplotype; Data mining; Index (typography); Computation; Theoretical computer science; Information retrieval; Algorithm; Artificial intelligence; Bioinformatics; Biology; Genotype; Genetics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000552047,0.00009138968,0.0001450195,0.00003841198,0.0001438059,0.00001355137,0.0003097325,0.0001194344,0.000001381372],"category_scores_gemma":[0.0004948369,0.00006660442,0.00004906144,0.0001272152,0.00005364178,0.00000579865,0.000468859,0.00004585575,0.000002792412],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000008429649,"about_ca_system_score_gemma":0.00005562053,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001153504,"about_ca_topic_score_gemma":0.0000218331,"domain_scores_codex":[0.9992415,0.00002229422,0.0003186433,0.0001306105,0.0000755723,0.0002113294],"domain_scores_gemma":[0.9991334,0.00006418399,0.0001947724,0.0005194804,0.00005379613,0.00003433747],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001486973,0.0001258792,0.5415142,0.001801556,0.001103031,0.000001449148,0.005052452,0.001935666,0.153095,0.007724059,0.195939,0.09155904],"study_design_scores_gemma":[0.003078562,0.0005564225,0.1845938,0.0002748393,0.0002917073,0.00003772759,0.004042985,0.2646189,0.04986459,0.00208146,0.4893728,0.001186201],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9796332,0.0005178176,0.01650746,0.0008015859,0.0005990958,0.0004898074,0.0003314738,0.00002646502,0.001093164],"genre_scores_gemma":[0.9813676,0.0003603064,0.01710947,0.0001930681,0.000171588,0.00001801734,0.0003062141,0.00001430541,0.0004593548],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3569204,"threshold_uncertainty_score":0.271605,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04623299182789144,"score_gpt":0.2960578176711589,"score_spread":0.2498248258432675,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}