{"id":"W4390082939","doi":"10.1101/2023.12.20.572683","title":"Leveraging ancestral sequence reconstruction for protein representation learning","year":2023,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canada's Michael Smith Genome Sciences Centre; University of British Columbia","funders":"","keywords":"Sequence (biology); Representation (politics); Embedding; Computer science; Protein sequencing; Space (punctuation); Artificial intelligence; Value (mathematics); Machine learning; Peptide sequence; Biology; Genetics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002134606,0.0006550764,0.0008084784,0.001150556,0.0004320135,0.001253764,0.001202148,0.001243263,0.001593587],"category_scores_gemma":[0.008724653,0.0005118823,0.0008715836,0.001083439,0.001026486,0.002199438,0.001846347,0.002005584,0.0007917631],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008233646,"about_ca_system_score_gemma":0.0006647187,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001109511,"about_ca_topic_score_gemma":0.001459921,"domain_scores_codex":[0.9991954,0.0003838952,0.00004071444,0.0001851707,0.0001432435,0.00005152641],"domain_scores_gemma":[0.9977136,0.001281883,0.0001909065,0.0005289533,0.0001953769,0.00008928224],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002025096,0.0001601563,0.004509403,0.0001348413,0.0001234262,0.0001616292,0.0001846999,0.8089552,0.02107856,0.04441112,0.002046389,0.118032],"study_design_scores_gemma":[0.000005439246,0.00002249799,0.00009555178,0.000003922561,0.000003954141,0.0000201435,0.000007881556,0.9829701,0.001976648,0.01452672,0.0003620984,0.000004954035],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1118636,0.0002843008,0.8835008,0.0004731727,0.0000325186,0.00003603142,0.0002920268,0.002049446,0.001468073],"genre_scores_gemma":[0.6623642,0.0002649135,0.3339268,0.0002778608,0.00003991526,0.0001156208,0.001308814,0.0004029803,0.001298923],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002134606,"threshold_uncertainty_score":0.011289,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04581633523021925,"score_gpt":0.2658758574834676,"score_spread":0.2200595222532484,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}