{"id":"W7110092454","doi":"10.64898/2025.12.05.690496","title":"Predicting and Designing Red Fluorescent Protein Variants Using Sequence-to-Function Machine Learning Models","year":2025,"lang":"","type":"article","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Advanced Fluorescence Microscopy Techniques","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada; University of Toronto","funders":"","keywords":"Benchmark (surveying); Deep learning; Protein engineering; Feature engineering; Training set","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000787714,0.000989616,0.0005284449,0.0004214748,0.0002067502,0.0005675716,0.0004911017,0.0007533371,0.0006535258],"category_scores_gemma":[0.001126092,0.0002602723,0.0007153227,0.0002663242,0.0003252435,0.0003793837,0.0002506937,0.0009259061,0.0002669817],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009718811,"about_ca_system_score_gemma":0.00070658,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002586585,"about_ca_topic_score_gemma":0.002789246,"domain_scores_codex":[0.9997917,0.00007474678,0.00001124055,0.00006108002,0.00003812075,0.00002316539],"domain_scores_gemma":[0.999592,0.0002679177,0.00004472259,0.00002523166,0.00004897848,0.00002118507],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006449931,0.00008770685,0.001701743,0.00004102192,0.00003071357,0.00004793762,0.000009998356,0.9718159,0.01199535,0.001279746,0.0003808497,0.01254464],"study_design_scores_gemma":[0.000002294678,0.00001159475,0.00004527504,0.000001145432,0.000002833738,0.000002781071,0.00000145569,0.9973453,0.002203822,0.0003015299,0.00008057559,0.000001504874],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5877914,0.0008495252,0.4055293,0.0006110963,0.0000602301,0.00007541029,0.0005295603,0.001733607,0.00281983],"genre_scores_gemma":[0.8998734,0.0002703377,0.09801941,0.0001353999,0.00001357322,0.00007855213,0.0006074425,0.00008497854,0.0009169118],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.002586585,"threshold_uncertainty_score":0.007051528,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01738817698682955,"score_gpt":0.2448467277726866,"score_spread":0.227458550785857,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}