{"id":"W4416113856","doi":"10.1371/journal.pcbi.1012929","title":"Zero-shot segmentation using embeddings from a protein language model identifies functional regions in the human proteome","year":2025,"lang":"en","type":"article","venue":"PLoS Computational Biology","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Hospital for Sick Children; University of Toronto","funders":"Canadian Institutes of Health Research; Canada Foundation for Innovation; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"UniProt; Human proteome project; Proteome; Categorization; Protein function prediction; Protein domain; Proteomics; Protein family; Annotation; Function (biology)","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005491004,0.0005924411,0.0004505549,0.0008979608,0.0003605779,0.0006778273,0.000531999,0.0008115832,0.00113412],"category_scores_gemma":[0.002082017,0.000234365,0.0008292936,0.0005997447,0.0005848433,0.001448762,0.001073269,0.00080551,0.0006466216],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005756944,"about_ca_system_score_gemma":0.0006066629,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002635486,"about_ca_topic_score_gemma":0.00282534,"domain_scores_codex":[0.999592,0.00008811866,0.00002350858,0.0001571937,0.00007707476,0.00006206281],"domain_scores_gemma":[0.9992318,0.0003731118,0.0001114245,0.0001008403,0.000125707,0.0000571533],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003226619,0.0005683334,0.03679598,0.001008925,0.0003017316,0.001974454,0.001531848,0.2720625,0.2748441,0.02414103,0.01222055,0.3713239],"study_design_scores_gemma":[0.00002220221,0.0001518544,0.003547061,0.00002058351,0.00002309129,0.0002836865,0.0002152644,0.9631702,0.02095925,0.009306821,0.0022727,0.00002714498],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6494682,0.0006758951,0.3409547,0.0004259598,0.00009095288,0.00007560377,0.001691025,0.004537353,0.002080237],"genre_scores_gemma":[0.8719959,0.0002477911,0.1195669,0.0001513602,0.00002635054,0.00007239267,0.006266836,0.0003721601,0.001300388],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.002635486,"threshold_uncertainty_score":0.005240262,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03058763592084477,"score_gpt":0.3211175182554487,"score_spread":0.2905298823346039,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}