{"id":"W4410643788","doi":"10.26434/chemrxiv-2025-lwnjs","title":"Bioactivity prediction with chemical language models trained on labeled molecules","year":2025,"lang":"en","type":"preprint","venue":"ChemRxiv","topic":"Computational Drug Discovery Methods","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"European Research Council; European Commission; European Federation of Pharmaceutical Industries and Associations; McGill University; Diamond Light Source; Innovative Medicines Initiative","keywords":"Natural language processing; Computer science; Artificial intelligence; Chemistry; Computational biology; Biology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009806723,0.0008674581,0.0005842202,0.0006377535,0.0001631528,0.0008287643,0.0009967117,0.001089541,0.001580257],"category_scores_gemma":[0.00340454,0.0003560936,0.001004706,0.0005564035,0.0004251727,0.001393709,0.0006032162,0.001530474,0.0008871466],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008378537,"about_ca_system_score_gemma":0.000904823,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001599863,"about_ca_topic_score_gemma":0.002264937,"domain_scores_codex":[0.9997181,0.0001021946,0.00001847797,0.00007618112,0.00005950853,0.00002553725],"domain_scores_gemma":[0.9985942,0.0008040708,0.0002267021,0.0001579819,0.0001714241,0.00004568235],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002604491,0.0001668202,0.002376506,0.0002350658,0.0001046085,0.0001261414,0.00003235829,0.8926153,0.0131858,0.009066597,0.002597654,0.07923281],"study_design_scores_gemma":[0.000007152099,0.00002131967,0.00004134246,0.000003619796,0.000004116619,0.000005583453,0.000001770553,0.9944897,0.002327233,0.002850621,0.0002446587,0.000002949291],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1687352,0.001464135,0.8144233,0.00125823,0.000169758,0.0001323236,0.002281735,0.005681059,0.005854211],"genre_scores_gemma":[0.8434169,0.0007635955,0.1469216,0.0005746988,0.00009341801,0.0002883074,0.003239817,0.0002189034,0.004482836],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.001599863,"threshold_uncertainty_score":0.006079018,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02472355869778331,"score_gpt":0.2898794624271198,"score_spread":0.2651559037293365,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}