{"id":"W4386811376","doi":"10.1101/2023.09.15.558005","title":"Improving the performance of supervised deep learning for regulatory genomics using phylogenetic augmentation","year":2023,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Generalization; Genomics; Phylogenetic tree; Sequence (biology); Artificial intelligence; Machine learning; Set (abstract data type); Source code; Function (biology); Computer science; Deep learning; Code (set theory); Functional genomics; Biological data; Genome; Computational biology; Biology; Bioinformatics; Gene; Genetics; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003083466,0.00151415,0.0008462745,0.0005562474,0.0003960638,0.000856717,0.001830851,0.001438953,0.002941189],"category_scores_gemma":[0.008486327,0.0004676172,0.0008450763,0.0005959021,0.0008581639,0.001710237,0.001552639,0.002843176,0.001308872],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001211121,"about_ca_system_score_gemma":0.001278812,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003758744,"about_ca_topic_score_gemma":0.004693476,"domain_scores_codex":[0.999159,0.0003429553,0.00004255112,0.0002264238,0.000152198,0.00007687559],"domain_scores_gemma":[0.9949368,0.003190604,0.0002460773,0.0007128569,0.0007334569,0.0001801024],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005992732,0.0004414987,0.005702014,0.0002711662,0.0001266287,0.0001108042,0.00009301006,0.7799199,0.01628434,0.004618929,0.007355747,0.1844767],"study_design_scores_gemma":[0.000008279893,0.000022774,0.0001759518,0.000005851072,0.000004941576,0.000005571266,0.000004763012,0.9941812,0.00300005,0.002332836,0.0002541567,0.000003598074],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3351831,0.001797539,0.6348554,0.002391311,0.0003632904,0.0001768852,0.001821855,0.0173667,0.006043769],"genre_scores_gemma":[0.7881991,0.000284705,0.2033334,0.000601068,0.0001006212,0.0002397109,0.004098936,0.000505267,0.002637215],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003758744,"threshold_uncertainty_score":0.01630712,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01859754700293224,"score_gpt":0.2220003104755954,"score_spread":0.2034027634726631,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}