{"id":"W7108685807","doi":"10.5376/cmb.2025.15.0016","title":"Large Language Models for Biological Knowledge Extraction","year":2025,"lang":"","type":"article","venue":"Computational Molecular Biology","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Process (computing); Information extraction; Knowledge representation and reasoning; Knowledge extraction; Relation (database); Domain (mathematical analysis); Relationship extraction; Event (particle physics); Domain knowledge","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003635586,0.001329913,0.001186043,0.003684975,0.0009695594,0.003342982,0.00213508,0.001471054,0.005251538],"category_scores_gemma":[0.01763855,0.0009084347,0.003439424,0.004021143,0.0008231382,0.004739771,0.002812637,0.002931178,0.004082886],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002076785,"about_ca_system_score_gemma":0.002783729,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008677028,"about_ca_topic_score_gemma":0.01369634,"domain_scores_codex":[0.9971377,0.001267656,0.000308191,0.0004799259,0.0006886748,0.000117809],"domain_scores_gemma":[0.9908591,0.006703653,0.0004916481,0.001024628,0.0008033312,0.0001177333],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002375351,0.0001868657,0.001842234,0.0009440941,0.0004915917,0.0006248371,0.0005346645,0.3277434,0.003221554,0.1985108,0.03762245,0.42804],"study_design_scores_gemma":[0.0000222232,0.00001987707,0.0002083928,0.0000824965,0.00005790349,0.0001264886,0.00006440434,0.7836182,0.001230542,0.1933632,0.02117545,0.00003102521],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00184695,0.001397831,0.9876651,0.001005117,0.00009538102,0.0001260347,0.002200543,0.003728237,0.001934856],"genre_scores_gemma":[0.134087,0.003331469,0.8413804,0.0008909086,0.0002786674,0.001033065,0.01293294,0.0008776066,0.005188025],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008677028,"threshold_uncertainty_score":0.01922703,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02436661784923981,"score_gpt":0.3726838860090235,"score_spread":0.3483172681597836,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}