{"id":"W4405033893","doi":"10.52202/079017-3217","title":"Multi-Scale Representation Learning for Protein Fitness Prediction","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Protein Structure and Dynamics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Samsung; National Institutes of Health; Tencent; Canadian Institute for Advanced Research; Natural Sciences and Engineering Research Council of Canada; Institut de Valorisation des Données; Microsoft Research","keywords":"Computer science; Artificial intelligence; Sequence (biology); Representation (politics); Benchmark (surveying); Machine learning; Protein structure prediction; Merge (version control); Protein sequencing; Protein function prediction; Fitness function; Protein function; Protein structure; Peptide sequence; Biology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000887128,0.001214319,0.001110778,0.0009387034,0.0003142039,0.001005145,0.001307347,0.001587751,0.002092621],"category_scores_gemma":[0.002577793,0.0003797793,0.0009541233,0.001013788,0.0008243856,0.001888735,0.001215997,0.001845986,0.0008282942],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001197436,"about_ca_system_score_gemma":0.0005779691,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002241309,"about_ca_topic_score_gemma":0.002388956,"domain_scores_codex":[0.9995969,0.0001118744,0.00001473891,0.0001358649,0.00008909678,0.00005143648],"domain_scores_gemma":[0.999328,0.0003152679,0.00009149391,0.0001192337,0.00008514412,0.00006086711],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001446457,0.0001515808,0.002431131,0.000163057,0.0001016888,0.00009526097,0.00006522779,0.7871923,0.01029354,0.01451435,0.007091503,0.1777558],"study_design_scores_gemma":[0.000004068203,0.00001580288,0.000144828,0.000003977892,0.000004095569,0.00001053219,0.000004883799,0.990103,0.0006687227,0.008765419,0.0002697649,0.000004818218],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.08769041,0.001404445,0.9028955,0.0008424647,0.00007364517,0.00004423611,0.0005621489,0.004449484,0.002037595],"genre_scores_gemma":[0.8496514,0.0007338432,0.1433182,0.0004652487,0.00009974599,0.0001542929,0.001858942,0.0004573117,0.003261062],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002241309,"threshold_uncertainty_score":0.008688092,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01459490691840631,"score_gpt":0.2917391632314136,"score_spread":0.2771442563130073,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}