{"id":"W4407394592","doi":"10.1126/sciadv.adr7338","title":"Self-supervised machine learning methods for protein design improve sampling but not the identification of high-fitness variants","year":2025,"lang":"en","type":"article","venue":"Science Advances","topic":"Evolutionary Algorithms and Applications","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Benchmark (surveying); Machine learning; Identification (biology); Toolbox; Sampling (signal processing); Artificial intelligence; Data mining; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004874969,0.0009299806,0.00105274,0.0008537722,0.0005025786,0.0009615743,0.001324048,0.001082189,0.001274088],"category_scores_gemma":[0.01159129,0.0003671287,0.000825304,0.0005721427,0.0008963314,0.001543258,0.001075161,0.00143195,0.0007615359],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005887226,"about_ca_system_score_gemma":0.0009046571,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005219143,"about_ca_topic_score_gemma":0.001275787,"domain_scores_codex":[0.9977022,0.001348109,0.0001077359,0.000332122,0.0004455634,0.00006435729],"domain_scores_gemma":[0.9911757,0.005660511,0.0006408851,0.00159867,0.0007872639,0.0001369554],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002028284,0.0003628613,0.00913141,0.0003040018,0.0002860482,0.00008609296,0.0001747152,0.7103596,0.01145728,0.01931199,0.002501328,0.2458218],"study_design_scores_gemma":[0.00001627597,0.00007518245,0.0004490229,0.00001313971,0.00001462637,0.00003040944,0.00000796766,0.9873947,0.002650011,0.008487685,0.0008537497,0.000007169763],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1115605,0.0007689142,0.8816593,0.0004538693,0.00006602301,0.00008860506,0.00009700285,0.001893979,0.003411851],"genre_scores_gemma":[0.6626272,0.0003262793,0.3341987,0.000359037,0.00006726442,0.0001717385,0.000338364,0.0003633965,0.001548106],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004874969,"threshold_uncertainty_score":0.02578163,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03000362555580971,"score_gpt":0.3482759882634606,"score_spread":0.3182723627076509,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}