{"id":"W4415046037","doi":"10.1093/bib/bbaf536","title":"Data-efficient protein mutational effect prediction with weak supervision by molecular simulation and protein language models","year":2025,"lang":"en","type":"article","venue":"Briefings in Bioinformatics","topic":"Protein Structure and Dynamics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Institute of Genetics; Japan Society for the Promotion of Science; Cybermedia Center, Osaka University; Japan Association for Chemical Innovation; New Energy and Industrial Technology Development Organization; Japan Agency for Medical Research and Development; Japan Society for the Promotion of Science London; National Institute of Advanced Industrial Science and Technology","keywords":"Benchmark (surveying); Training set; Experimental data; Training (meteorology); Protein engineering; Pathogenicity; Computational model; Mutation","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002490386,0.0001655912,0.0001354767,0.0000792255,0.0000681235,0.000060274,0.000163795,0.0001532171,0.000001241227],"category_scores_gemma":[0.00009004607,0.0001397705,0.00001856409,0.0001555205,0.00006462602,0.00003259414,0.0001799412,0.0001116031,6.142497e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000220347,"about_ca_system_score_gemma":0.00006429361,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004469316,"about_ca_topic_score_gemma":0.00001282102,"domain_scores_codex":[0.9990695,0.00003550295,0.0002842018,0.0002438741,0.0001932986,0.0001736649],"domain_scores_gemma":[0.9994529,0.0000141263,0.00008897225,0.000355332,0.00005172993,0.00003686766],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009362221,0.0001753223,0.0005724253,0.001310542,0.0001601167,0.00001204556,0.0007722591,0.533138,0.4093677,0.00335467,0.0007453958,0.04945531],"study_design_scores_gemma":[0.001137362,0.0002226905,0.0001428179,0.0001715393,0.00001768972,0.000007873729,0.00006854386,0.958468,0.03811301,0.0002582869,0.001221575,0.0001705682],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7201068,0.0003150307,0.2778214,0.00009931095,0.00001328842,0.0009072596,0.0001750472,0.00001934729,0.0005424685],"genre_scores_gemma":[0.9840283,0.000004610192,0.01383942,0.0002212353,0.00001295293,0.0000629467,0.001724225,0.0000130963,0.00009323099],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.42533,"threshold_uncertainty_score":0.5699676,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.004676397131470881,"score_gpt":0.2416921666111629,"score_spread":0.2370157694796921,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}