{"id":"W4386560750","doi":"10.3389/fmolb.2023.1257550","title":"A curated census of pathogenic and likely pathogenic UTR variants and evaluation of deep learning models for variant effect prediction","year":2023,"lang":"en","type":"article","venue":"Frontiers in Molecular Biosciences","topic":"Genomics and Rare Diseases","field":"Biochemistry, Genetics and Molecular Biology","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"SickKids Foundation; Hospital for Sick Children; Ontario Genomics","funders":"","keywords":"Untranslated region; Pathogenicity; Genetics; Computational biology; Biology; Confidence interval; Bioinformatics; Gene; Medicine; Internal medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001079083,0.00010748,0.000164764,0.0001501409,0.00006454796,0.00001489954,0.00009894583,0.00009730395,4.087143e-7],"category_scores_gemma":[0.0002806376,0.00009734113,0.00005613223,0.0002792048,0.0001403051,0.000008513432,0.00006598292,0.00003576926,4.481957e-8],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000008920689,"about_ca_system_score_gemma":0.00006880148,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000008064409,"about_ca_topic_score_gemma":0.000003875466,"domain_scores_codex":[0.9988812,0.0001801955,0.0002122147,0.0003484838,0.0002019674,0.0001759087],"domain_scores_gemma":[0.9995503,0.00001497305,0.000144788,0.0001201014,0.0001193055,0.0000505391],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001360213,0.00003079017,0.006732183,0.00007542241,0.00004439082,0.000004690291,0.0001275871,0.009735327,0.9612842,0.00004415635,0.00001987586,0.02176534],"study_design_scores_gemma":[0.002068965,0.001220571,0.05082614,0.00005858911,0.0002209492,0.00001844496,0.0005705598,0.8745167,0.06756485,0.002610272,0.00007265927,0.0002512876],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9656556,0.004241497,0.02933916,0.00001243199,0.0001946556,0.0004298562,0.0001053814,0.000006079451,0.00001536822],"genre_scores_gemma":[0.9967484,0.0005482259,0.002494846,0.000007417816,0.00001215943,0.00004405871,0.0001304879,0.000009874415,0.000004544172],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8937194,"threshold_uncertainty_score":0.3969456,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01283622021230062,"score_gpt":0.2504469162033067,"score_spread":0.2376106959910061,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}