{"id":"W4402423987","doi":"10.1101/2024.09.09.612016","title":"High-quality peptide evidence for annotating non-canonical open reading frames as human proteins","year":2024,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"RNA and protein synthesis mechanisms","field":"Biochemistry, Genetics and Molecular Biology","cited_by":35,"is_retracted":false,"has_abstract":true,"ca_institutions":"Centre Hospitalier Universitaire de Sherbrooke; Université de Sherbrooke","funders":"National Cancer Institute; National Health and Medical Research Council; National Institutes of Health; University College Cork; European Commission; Curing Kids Cancer; National Human Genome Research Institute; Nederlandse Organisatie voor Wetenschappelijk Onderzoek; Wellcome Trust; Hope Foundation; European Molecular Biology Laboratory; Medical Research Council; Stichting Villa Joep; Dr. Miriam and Sheldon G. Adelson Medical Research Foundation; Alex's Lemonade Stand Foundation for Childhood Cancer; Deutsche Forschungsgemeinschaft; Hyundai Hope On Wheels; Damon Runyon Cancer Research Foundation; National Science Foundation","keywords":"Human proteome project; Annotation; Open reading frame; Proteome; Computational biology; Proteomics; Genomics; Proteogenomics; Reading (process); Human genome; Data science; Genome; Biology; Computer science; Gene; Bioinformatics; Genetics; Political science; Peptide sequence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.002453343,0.0007373539,0.0008246975,0.0001312868,0.000395147,0.000853657,0.001701649,0.001039058,0.00004214776],"category_scores_gemma":[0.001740113,0.0007696463,0.000354679,0.0002158269,0.0001327054,0.00002479769,0.002994573,0.0007229043,0.00006702996],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000141008,"about_ca_system_score_gemma":0.00125044,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004957902,"about_ca_topic_score_gemma":0.00001482664,"domain_scores_codex":[0.9956213,0.0003024791,0.0009175607,0.002013672,0.0003987321,0.0007462025],"domain_scores_gemma":[0.9965712,0.0001391853,0.0006127807,0.001883327,0.0004887372,0.0003047019],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0001370038,0.00007411677,0.0002211569,0.001051925,0.0002491308,0.00002029418,0.000009076909,0.00002123876,0.9933077,0.004351276,0.0005439485,0.00001320182],"study_design_scores_gemma":[0.000395949,0.0003548714,0.001666584,0.00223774,0.0001474553,4.566703e-8,0.000005979509,0.00002545207,0.9899322,0.000251712,0.003961314,0.001020636],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9852077,0.001556598,0.007212953,0.0004449398,0.0009603661,0.004077175,0.000330746,0.0001518609,0.00005764495],"genre_scores_gemma":[0.9470675,0.0001341175,0.04798377,0.0003311016,0.001316133,0.002773552,0.000003429095,0.0002340209,0.0001563632],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.04077082,"threshold_uncertainty_score":0.9994755,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04594483221233545,"score_gpt":0.3305771607848429,"score_spread":0.2846323285725074,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}