{"id":"W4414846902","doi":"10.1021/acs.jmedchem.5c01972","title":"Enabling Open Machine Learning of Deoxyribonucleic Acid-Encoded Library Selections to Accelerate the Discovery of Small Molecule Protein Binders","year":2025,"lang":"en","type":"article","venue":"Journal of Medicinal Chemistry","topic":"Chemical Synthesis and Analysis","field":"Biochemistry, Genetics and Molecular Biology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; Princess Margaret Cancer Centre; University Health Network; Structural Genomics Consortium; University of Toronto","funders":"Structural Genomics Consortium; Janssen Biotech; Innovative Medicines Initiative; National Institute of General Medical Sciences; Merck KGaA; Genentech; Bayer; Takeda Pharmaceuticals U.S.A.; Pfizer; Bristol-Myers Squibb; Boehringer Ingelheim","keywords":"Virtual screening; Drug discovery; Chemical space; Small molecule; Transparency (behavior); Training set; Open data; Cheminformatics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"bench_or_experimental","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"},{"model":"gpt","categories":[],"domain":null,"study_design":"bench_or_experimental","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"}],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000376461,0.0001306674,0.000330117,0.00005320537,0.00007248187,0.00004216846,0.0006265296,0.0001045092,0.00005281483],"category_scores_gemma":[0.0004884973,0.00008841715,0.0001984106,0.0003674508,0.00008175142,0.00001618002,0.0002730885,0.000305059,1.983089e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001714184,"about_ca_system_score_gemma":0.0002506461,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003248127,"about_ca_topic_score_gemma":0.000004289999,"domain_scores_codex":[0.9989073,0.00005635907,0.0005281176,0.0001658815,0.0001907556,0.0001516219],"domain_scores_gemma":[0.9991338,0.00003954621,0.0004142314,0.0001959327,0.0001354713,0.00008101996],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002764485,0.00008430582,0.001170243,0.00012571,0.0002432255,0.000003755424,0.0000314382,0.000165813,0.9966069,0.000005336148,0.0004319173,0.000854917],"study_design_scores_gemma":[0.0003753394,0.0001187589,0.00009832111,0.0003746719,0.0001259149,0.00001586184,0.0003435256,0.0001252599,0.9964604,0.00004580823,0.00183439,0.00008175304],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9927281,0.001436919,0.002638677,0.001553087,0.00002228393,0.00009848434,0.000006716325,0.000002404442,0.001513296],"genre_scores_gemma":[0.9972157,0.0001870998,0.0005868901,0.0001877312,0.0001220596,0.000005530539,0.00001273663,0.00001271187,0.001669478],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.00448763,"threshold_uncertainty_score":0.3605547,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01111729481901655,"score_gpt":0.2501815033511763,"score_spread":0.2390642085321597,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}