{"id":"W7110092454","doi":"10.64898/2025.12.05.690496","title":"Predicting and Designing Red Fluorescent Protein Variants Using Sequence-to-Function Machine Learning Models","year":2025,"lang":"","type":"article","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Advanced Fluorescence Microscopy Techniques","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada; University of Toronto","funders":"","keywords":"Benchmark (surveying); Deep learning; Protein engineering; Feature engineering; Training set","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0009478349,0.0007727396,0.0005742047,0.0004029274,0.0008375998,0.0002832255,0.0004388937,0.0006770196,0.000007340689],"category_scores_gemma":[0.0006095797,0.0009323826,0.0001129589,0.001019549,0.0002457667,0.00009892829,0.0006431331,0.0007670253,0.000003030054],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003866277,"about_ca_system_score_gemma":0.0006736632,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00008222166,"about_ca_topic_score_gemma":0.000003217176,"domain_scores_codex":[0.9957489,0.0003686643,0.000821278,0.001703229,0.0003155064,0.001042431],"domain_scores_gemma":[0.9977102,0.00002746747,0.0004227449,0.000899072,0.0005714464,0.0003690329],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0003544259,0.0001009924,0.01144597,0.0003262779,0.0001626476,0.00002062829,0.00001203919,0.001902515,0.9853675,0.0002616398,0.00001117317,0.00003419376],"study_design_scores_gemma":[0.0007523053,0.0003882286,0.001595332,0.001862301,0.0001636228,1.030868e-7,0.00000931871,0.05502407,0.9388071,0.00001005366,0.0006019383,0.0007856145],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5111527,0.002763654,0.4840648,0.00008557492,0.0002779985,0.001389686,0.00005229602,0.0002048108,0.000008525564],"genre_scores_gemma":[0.8512454,0.000506049,0.1475835,0.0001898811,0.0001872709,0.000133393,7.636378e-7,0.0001390182,0.00001471042],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3400927,"threshold_uncertainty_score":0.9993127,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01738817698682955,"score_gpt":0.2448467277726866,"score_spread":0.227458550785857,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}