{"id":"W4376872947","doi":"10.1186/s12859-023-05327-8","title":"CysPresso: a classification model utilizing deep learning protein representations to predict recombinant expression of cysteine-dense peptides","year":2023,"lang":"en","type":"article","venue":"BMC Bioinformatics","topic":"Biochemical and Structural Characterization","field":"Biochemistry, Genetics and Molecular Biology","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; Critical Systems Labs; Sunnybrook Health Science Centre","funders":"H2020 Marie Skłodowska-Curie Actions; Medical Research Council; Alliance de recherche numérique du Canada; Natural Sciences and Engineering Research Council of Canada; University of Cambridge; UK Research and Innovation","keywords":"Artificial intelligence; Deep learning; Computer science; Computational biology; Machine learning; Concatenation (mathematics); Convolutional neural network; Chemical space; Support vector machine; Drug discovery; Chemistry; Biology; Biochemistry; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000688259,0.001118913,0.0006527748,0.0007700862,0.0002213324,0.0006952395,0.0007334306,0.001037878,0.0009983307],"category_scores_gemma":[0.0009199564,0.0002690079,0.0007337211,0.0003272435,0.0002702722,0.000641727,0.0004446465,0.001096071,0.0003508008],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001100159,"about_ca_system_score_gemma":0.0008935367,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006802286,"about_ca_topic_score_gemma":0.005215005,"domain_scores_codex":[0.9998542,0.00002524996,0.000009163296,0.00005255421,0.00002495564,0.00003388834],"domain_scores_gemma":[0.9996314,0.0001745552,0.00005717398,0.00002386486,0.0000794525,0.00003344802],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006113262,0.0004347228,0.01319222,0.00009334178,0.0001770563,0.000174409,0.00003753952,0.8380473,0.01198978,0.001376799,0.004529438,0.129336],"study_design_scores_gemma":[0.000003591529,0.0000266213,0.0002256432,0.000002329118,0.000004492422,0.000007291711,0.000001829726,0.9984976,0.0008650563,0.000267869,0.00009517594,0.000002556061],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.6105384,0.001746335,0.3765509,0.001020733,0.0001755342,0.0001737123,0.002216783,0.00523537,0.002342256],"genre_scores_gemma":[0.9436466,0.0003755566,0.04948122,0.0002504993,0.00004913314,0.0001756003,0.003052229,0.00009009199,0.00287898],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.006802286,"threshold_uncertainty_score":0.01352537,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03017812936265334,"score_gpt":0.2728277365811366,"score_spread":0.2426496072184833,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}