{"id":"W2978660451","doi":"10.1101/788919","title":"Uniform Genomic Data Analysis in the NCI Genomic Data Commons","year":2019,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Cancer Genomics and Diagnostics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":20,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Cancer Institute; National Institutes of Health; University of Texas MD Anderson Cancer Center; Canada's Michael Smith Genome Sciences Centre; U.S. Department of Health and Human Services","keywords":"Workflow; Data sharing; Raw data; Epigenomics; Genomics; Computational biology; Precision medicine; Copy-number variation; DNA methylation; Biology; Computer science; Genome; Database; Gene; Genetics; Medicine; Gene expression","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","open_science"],"consensus_categories":["open_science"],"category_scores_codex":[0.001475115,0.0005606884,0.0006562708,0.0002799603,0.0001288395,0.0003258462,0.006307766,0.0005816572,0.00002403903],"category_scores_gemma":[0.0002662688,0.0005336254,0.0001720926,0.0005612022,0.000136687,0.00001600027,0.008652524,0.0006958497,0.00006363233],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001683176,"about_ca_system_score_gemma":0.001262499,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006285049,"about_ca_topic_score_gemma":0.0008470921,"domain_scores_codex":[0.9963671,0.0001741463,0.0006609847,0.001907606,0.0002821496,0.0006079966],"domain_scores_gemma":[0.986936,0.00009425169,0.0004350252,0.0122359,0.0001478577,0.0001510021],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","study_design_scores_codex":[0.0001657444,0.0005446843,0.1681505,0.00035897,0.003416724,0.00009225489,0.00003323302,0.009885802,0.7985232,0.0005026661,0.01831057,0.00001572106],"study_design_scores_gemma":[0.001663818,0.0001533218,0.6506439,0.0001241671,0.003020526,1.348973e-7,0.00003303363,0.01708957,0.01683834,0.000006822583,0.3079531,0.002473252],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9770933,0.005665472,0.00273403,0.00039376,0.0007909302,0.0009399749,0.01230099,0.00003425884,0.0000473068],"genre_scores_gemma":[0.9928738,0.003078867,0.002003513,0.0007572416,0.0007576321,0.0000548809,0.0003568974,0.0001103076,0.000006794353],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7816848,"threshold_uncertainty_score":0.9997115,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02850292568356836,"score_gpt":0.2536023526968867,"score_spread":0.2250994270133183,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}