{"id":"W4384028514","doi":"10.1101/2023.07.10.23292474","title":"A curated census of pathogenic and likely pathogenic UTR variants and evaluation of deep learning models for variant effect prediction","year":2023,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Genomics and Rare Diseases","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"SickKids Foundation; Hospital for Sick Children; Ontario Genomics","funders":"","keywords":"Untranslated region; Pathogenicity; Genetics; Confidence interval; Biology; Computational biology; Five prime untranslated region; Bioinformatics; Gene; Medicine; Internal medicine; Messenger RNA","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.004105076,0.0007649413,0.0008937419,0.003719474,0.0007578499,0.0009599102,0.0009564123,0.00106155,0.00131532],"category_scores_gemma":[0.01263839,0.0003386347,0.001364564,0.002073034,0.0005369991,0.0005090417,0.001286955,0.000743197,0.0006905415],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006191998,"about_ca_system_score_gemma":0.00107213,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005587991,"about_ca_topic_score_gemma":0.008757512,"domain_scores_codex":[0.9972597,0.0007413842,0.0003136004,0.0007997372,0.000693264,0.0001922809],"domain_scores_gemma":[0.9923511,0.004453155,0.000585408,0.001066494,0.001216736,0.0003270874],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.002988719,0.0006294069,0.6306534,0.002044279,0.001833355,0.008642386,0.0007855868,0.09654757,0.06802744,0.003686059,0.02527354,0.1588883],"study_design_scores_gemma":[0.0004918518,0.0009930687,0.2810203,0.0005601626,0.001142004,0.00928282,0.0007821873,0.6237513,0.03887925,0.007750847,0.035166,0.0001802333],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9516262,0.0019533,0.02260194,0.0002098132,0.00005601314,0.0001307497,0.02048849,0.001504504,0.001428878],"genre_scores_gemma":[0.8752832,0.0005559105,0.03678489,0.0002284141,0.00004260296,0.00014673,0.0859816,0.0003649869,0.0006117206],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9958949,"threshold_uncertainty_score":0.02170998,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0344840857462884,"score_gpt":0.278662914884624,"score_spread":0.2441788291383356,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}