{"id":"W2887369609","doi":"10.1101/386441","title":"Extracting allelic read counts from 250,000 human sequencing runs in Sequence Read Archive","year":2018,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institutes of Health; Canadian Institute for Advanced Research","keywords":"dbSNP; DNA sequencing; Biology; Sequence (biology); Genetics; Human genome; Reference genome; Computational biology; Sequence analysis; Allele; Genomics; Genome; Gene; Genotype; Single-nucleotide polymorphism","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004187655,0.001223234,0.00109705,0.003382709,0.0009768542,0.001683725,0.001135129,0.001095026,0.003819725],"category_scores_gemma":[0.01295472,0.0007136338,0.001327286,0.004585683,0.0004491298,0.0007840934,0.001171263,0.001705704,0.004714884],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004652694,"about_ca_system_score_gemma":0.001288319,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002056648,"about_ca_topic_score_gemma":0.005147671,"domain_scores_codex":[0.9947624,0.000704253,0.0007938128,0.002355909,0.001186696,0.0001970219],"domain_scores_gemma":[0.9947181,0.001574102,0.0005953013,0.00168277,0.001262531,0.0001671992],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003021488,0.0004485687,0.1174336,0.005707699,0.003291033,0.001335855,0.002100403,0.01465002,0.5001109,0.007171426,0.09425183,0.2504771],"study_design_scores_gemma":[0.0004253599,0.0008657454,0.263265,0.0005476072,0.001731699,0.00263497,0.0006580173,0.04800234,0.3367531,0.01764828,0.3270432,0.0004246907],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"methods","genre_scores_codex":[0.2385422,0.003742671,0.1932088,0.000431753,0.0004856512,0.0005412682,0.5345238,0.02139461,0.007129209],"genre_scores_gemma":[0.1518391,0.0008641661,0.292008,0.0005299096,0.0001617743,0.0007530106,0.5465242,0.004563856,0.002756021],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.004187655,"threshold_uncertainty_score":0.0221467,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02936949853081398,"score_gpt":0.2543967648922655,"score_spread":0.2250272663614515,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}