{"id":"W4411009890","doi":"10.1093/molbev/msaf136","title":"Exploring Large Protein Sequence Space through Homology- and Representation-based Hierarchical Clustering","year":2025,"lang":"en","type":"article","venue":"Molecular Biology and Evolution","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canada's Michael Smith Genome Sciences Centre; University of British Columbia","funders":"Human Frontier Science Program","keywords":"Biology; Computational biology; Sequence alignment; Alignment-free sequence analysis; Sequence analysis; Protein function prediction; Hidden Markov model; Homology (biology); Protein sequencing; Sequence (biology); Cluster analysis; Phylogenetic tree; Multiple sequence alignment; Leverage (statistics); Visualization; Pipeline (software); Bioinformatics; Genetics; Computer science; Data mining; Peptide sequence; Artificial intelligence; Gene; Protein function","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001494998,0.001256292,0.00120491,0.006473691,0.001698727,0.002124926,0.001834016,0.0009149666,0.002378031],"category_scores_gemma":[0.004411562,0.0006454145,0.001887788,0.006043419,0.0007950193,0.0019118,0.002175203,0.001514584,0.001774712],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00137236,"about_ca_system_score_gemma":0.002398929,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01147598,"about_ca_topic_score_gemma":0.01548099,"domain_scores_codex":[0.998697,0.0003025015,0.00008607778,0.0004006309,0.0003948943,0.000118888],"domain_scores_gemma":[0.9986777,0.0004644036,0.0001753218,0.0002511952,0.0003373799,0.00009393067],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005766246,0.000602756,0.01470976,0.001336294,0.0005952638,0.0004825452,0.003001514,0.1752232,0.1290629,0.03786161,0.01675327,0.6197943],"study_design_scores_gemma":[0.00004388951,0.0001036569,0.004967086,0.00007225275,0.00008738832,0.0002316542,0.0007528181,0.8938735,0.01914006,0.0681638,0.01246413,0.00009969006],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05314969,0.0006108764,0.934658,0.000275543,0.0000249578,0.0002820991,0.001758127,0.007539768,0.001700944],"genre_scores_gemma":[0.1969588,0.0005278449,0.7934391,0.0001052239,0.00002456403,0.000375822,0.006651849,0.0008364341,0.001080371],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01147598,"threshold_uncertainty_score":0.02281839,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03312948828492843,"score_gpt":0.2995392262859621,"score_spread":0.2664097380010336,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}