{"id":"W7162829424","doi":"","title":"Minimal-Data Peptide Design and Accessible Protein Language Models for Protein Engineering","year":2025,"lang":"","type":"dissertation","venue":"TSpace","topic":"vaccines and immunoinformatics approaches","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Protein engineering; Protein sequencing; Synthetic biology; Pipeline (software); Stability (learning theory); Sequence (biology); Divergence (linguistics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.000495441,0.000601479,0.0005120813,0.0001364009,0.0001888556,0.0002611891,0.0008395392,0.0006107715,0.00001071573],"category_scores_gemma":[0.0002826455,0.0006081486,0.0001208555,0.0001548655,0.00001977413,0.00004705268,0.0004526856,0.0002633572,0.000001828424],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002354473,"about_ca_system_score_gemma":0.0004165924,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00008102439,"about_ca_topic_score_gemma":0.00002035168,"domain_scores_codex":[0.9979061,0.00004410241,0.000572401,0.0007913887,0.0001618966,0.0005240735],"domain_scores_gemma":[0.9981781,0.00002778018,0.0003613182,0.001148195,0.0001754909,0.0001091095],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0009474081,0.0001233211,0.00000196544,0.006488315,0.0004610449,0.000002104481,0.003332425,0.004802145,0.9748962,0.0002963086,0.001098012,0.007550738],"study_design_scores_gemma":[0.001438561,0.0004273809,0.00001114925,0.001014716,0.0001924183,0.000004421986,0.00595578,0.4876524,0.4989986,0.00006125266,0.003444761,0.0007985201],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6065142,0.01578606,0.3669431,0.0002428011,0.0001758154,0.008612019,0.0002348321,0.00004602319,0.001445099],"genre_scores_gemma":[0.6441727,0.0009659835,0.2596627,0.00006237489,0.0003681892,0.002653253,0.01322442,0.0002068628,0.07868358],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4828503,"threshold_uncertainty_score":0.999637,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0368453139958262,"score_gpt":0.3076602548904919,"score_spread":0.2708149408946657,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}