{"id":"W4413290922","doi":"10.1038/s42256-025-01088-6","title":"Boosting the predictive power of protein representations with a corpus of text annotations","year":2025,"lang":"en","type":"article","venue":"Nature Machine Intelligence","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":4,"is_retracted":false,"has_abstract":false,"ca_institutions":"Vector Institute; University of Toronto","funders":"Vector Institute; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Boosting (machine learning); Predictive power; Computer science; Natural language processing; Artificial intelligence; Physics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003304983,0.00149648,0.001134408,0.004564447,0.0006538872,0.001979073,0.001232629,0.001929639,0.001408353],"category_scores_gemma":[0.01795885,0.0004690292,0.0008991041,0.0036601,0.0007725415,0.004217129,0.001746412,0.002473539,0.001096664],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006163025,"about_ca_system_score_gemma":0.000990389,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004018251,"about_ca_topic_score_gemma":0.004665596,"domain_scores_codex":[0.9988226,0.0003504764,0.00007463802,0.0003177949,0.0003359065,0.00009853394],"domain_scores_gemma":[0.9851772,0.0114742,0.0005609872,0.001181415,0.001363028,0.0002431593],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002808565,0.001906519,0.02901236,0.001031003,0.000761482,0.0008458399,0.0005198515,0.1666456,0.03277401,0.007053256,0.03461297,0.7220286],"study_design_scores_gemma":[0.00007632383,0.0001834878,0.003252394,0.00008686721,0.0002258486,0.0001467137,0.00009354602,0.9710547,0.007517559,0.01379524,0.003535646,0.00003171817],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7157468,0.008351699,0.2443827,0.004515803,0.0009242984,0.0002912194,0.008004234,0.007055377,0.01072779],"genre_scores_gemma":[0.9134581,0.00224902,0.06791436,0.0006650014,0.000693068,0.0001714858,0.01197613,0.0002876754,0.002585141],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004564447,"threshold_uncertainty_score":0.01747864,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.005744969379399131,"score_gpt":0.2958628101583177,"score_spread":0.2901178407789186,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}