{"id":"W4409317481","doi":"10.1126/science.adx0339","title":"AI drug development’s data problem","year":2025,"lang":"en","type":"editorial","venue":"Science","topic":"Computational Drug Discovery Methods","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"Centre for International Governance Innovation","funders":"","keywords":"Computer science; Drug discovery; Field (mathematics); Data science; Drug development; Quality (philosophy); Artificial intelligence; Drug; Medicine; Bioinformatics; Pharmacology; Biology; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication","open_science"],"consensus_categories":["open_science"],"category_scores_codex":[0.005833798,0.0003289184,0.0003293529,0.0006041832,0.000585264,0.0015091,0.01973965,0.0001581456,0.000008636125],"category_scores_gemma":[0.002337652,0.0003150411,0.00004192319,0.002960414,0.0004255621,0.003072003,0.01215274,0.000766271,0.000121482],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003761681,"about_ca_system_score_gemma":0.0249998,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003156053,"about_ca_topic_score_gemma":0.00003437032,"domain_scores_codex":[0.9933235,0.0001619649,0.0004952762,0.002217078,0.00315595,0.0006462503],"domain_scores_gemma":[0.9942603,0.001444698,0.0002357434,0.003096563,0.000782499,0.0001801841],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0000012526,0.00002598114,0.000002209142,0.00005895372,0.000007338086,0.000005881433,0.0002042911,0.0003805742,0.000006973563,0.003916807,0.9556214,0.0397683],"study_design_scores_gemma":[0.000103273,0.000005889434,0.00003996892,0.000171777,0.000005251745,0.000001075423,0.000003773274,0.03102781,0.0001567441,0.008771953,0.9593616,0.0003509063],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"editorial","genre_gemma":"methods","genre_scores_codex":[0.00000317648,0.0001991357,0.3462049,0.0008820485,0.6488602,0.0002043311,0.00005882636,0.0001738088,0.003413518],"genre_scores_gemma":[0.00001940103,0.00002383938,0.7556666,0.0004849155,0.2354185,0.00003384044,0.0002890787,0.00001636294,0.008047564],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.4134417,"threshold_uncertainty_score":0.9999301,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02487313314061693,"score_gpt":0.3604463068627179,"score_spread":0.3355731737221009,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}