{"id":"W4414846902","doi":"10.1021/acs.jmedchem.5c01972","title":"Enabling Open Machine Learning of Deoxyribonucleic Acid-Encoded Library Selections to Accelerate the Discovery of Small Molecule Protein Binders","year":2025,"lang":"en","type":"article","venue":"Journal of Medicinal Chemistry","topic":"Chemical Synthesis and Analysis","field":"Biochemistry, Genetics and Molecular Biology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; Princess Margaret Cancer Centre; University Health Network; Structural Genomics Consortium; University of Toronto","funders":"Structural Genomics Consortium; Janssen Biotech; Innovative Medicines Initiative; National Institute of General Medical Sciences; Merck KGaA; Genentech; Bayer; Takeda Pharmaceuticals U.S.A.; Pfizer; Bristol-Myers Squibb; Boehringer Ingelheim","keywords":"Virtual screening; Drug discovery; Chemical space; Small molecule; Transparency (behavior); Training set; Open data; Cheminformatics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"bench_or_experimental","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"},{"model":"gpt","categories":[],"domain":null,"study_design":"bench_or_experimental","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"}],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.003836276,0.0008243706,0.001143376,0.001215002,0.0004806549,0.002056334,0.00172097,0.0008731692,0.002171587],"category_scores_gemma":[0.008227436,0.0005482461,0.001012386,0.0009937088,0.001115045,0.001726906,0.003280161,0.001621707,0.001265892],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001118064,"about_ca_system_score_gemma":0.001636341,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001548311,"about_ca_topic_score_gemma":0.001895027,"domain_scores_codex":[0.9976253,0.0007296648,0.0001424099,0.0004667292,0.0008481996,0.0001877324],"domain_scores_gemma":[0.996035,0.002097452,0.0004020259,0.0008356468,0.0004609291,0.0001688742],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001617839,0.001308216,0.01131748,0.0006207661,0.0002887073,0.0004745606,0.0002875033,0.3643877,0.2050243,0.02035687,0.006814913,0.3875012],"study_design_scores_gemma":[0.00006332152,0.0001858294,0.0007406002,0.00002085336,0.0000190126,0.00005914204,0.00003018796,0.9007978,0.08320247,0.009864296,0.004978731,0.00003763157],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1336461,0.0004776222,0.8321512,0.0004194528,0.000059796,0.0002112779,0.001298727,0.02766308,0.004072795],"genre_scores_gemma":[0.5841212,0.0005185381,0.4062542,0.0004329295,0.00005411637,0.0004420422,0.00513745,0.0009396898,0.002099946],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.998279,"threshold_uncertainty_score":0.02028841,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01111729481901655,"score_gpt":0.2501815033511763,"score_spread":0.2390642085321597,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}