{"id":"W4415683348","doi":"10.1093/bioinformatics/btaf582","title":"Endowing protein language models with structural knowledge","year":2025,"lang":"en","type":"article","venue":"Bioinformatics","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Instituto de Ciencias del Mar y Limnología, Universidad Nacional Autónoma de México; Institute for Catastrophic Loss Reduction","keywords":"Code (set theory); Software; Source code; Knowledge representation and reasoning; Natural language; Language model","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001113399,0.0009349791,0.0005044944,0.0006598359,0.0002965901,0.0009826598,0.001511264,0.0008395659,0.003134991],"category_scores_gemma":[0.007263806,0.0005082093,0.0007776124,0.0008015896,0.0006785199,0.003540552,0.002305134,0.002045743,0.002859422],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006605749,"about_ca_system_score_gemma":0.001021342,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002384328,"about_ca_topic_score_gemma":0.004054407,"domain_scores_codex":[0.9994338,0.0001790985,0.00003440996,0.0001557388,0.0001555165,0.00004141358],"domain_scores_gemma":[0.9979042,0.001201716,0.0001607996,0.0003901209,0.0002613119,0.00008175651],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000250552,0.0002321255,0.002434704,0.0004112076,0.0001107916,0.0001968021,0.000201916,0.691763,0.02214897,0.03458147,0.009201369,0.2384671],"study_design_scores_gemma":[0.000007658566,0.00002185337,0.00009505645,0.000009153392,0.000009446844,0.00002588795,0.00001258811,0.9751394,0.003494923,0.0197148,0.001462194,0.000007024665],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03501519,0.0003559481,0.9567283,0.0008369753,0.00005130005,0.00004897158,0.0006458897,0.004429854,0.001887573],"genre_scores_gemma":[0.5200176,0.0009182527,0.4671672,0.0005394681,0.0001259632,0.0002873873,0.004606192,0.0007124434,0.005625514],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003134991,"threshold_uncertainty_score":0.01048762,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01077066642079481,"score_gpt":0.2676589173807811,"score_spread":0.2568882509599862,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}