{"id":"W4402350721","doi":"10.1039/d4dd00065j","title":"A machine learning approach for the prediction of aqueous solubility of pharmaceuticals: a comparative model and dataset analysis","year":2024,"lang":"en","type":"article","venue":"Digital Discovery","topic":"Computational Drug Discovery Methods","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Solubility; Aqueous solution; Machine learning; Computer science; Artificial intelligence; Chemistry; Biochemical engineering; Engineering; Organic chemistry","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002524015,0.0008406426,0.0008468566,0.001895394,0.000353898,0.0008442384,0.001273801,0.0009792476,0.001145938],"category_scores_gemma":[0.003198904,0.0001659878,0.001316246,0.001700432,0.0002365942,0.001139156,0.000528588,0.0009016474,0.0003586987],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001223002,"about_ca_system_score_gemma":0.001091701,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01554305,"about_ca_topic_score_gemma":0.01363654,"domain_scores_codex":[0.9994599,0.000241045,0.00004689818,0.0001037662,0.0001118692,0.00003650599],"domain_scores_gemma":[0.9981372,0.00137263,0.00007510588,0.0001623516,0.0002183382,0.00003439883],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009910233,0.0008839399,0.01715807,0.0004516013,0.0007229532,0.00008819224,0.00002897414,0.8096126,0.002348663,0.001126038,0.007296634,0.1592913],"study_design_scores_gemma":[0.0000542659,0.0002862678,0.004085083,0.00002035142,0.0001057845,0.0000328766,0.0000195987,0.9920655,0.001598625,0.0008268752,0.0008882539,0.00001659445],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8562763,0.008542518,0.1057143,0.001900414,0.0001918323,0.0003039959,0.02015332,0.002965704,0.003951577],"genre_scores_gemma":[0.9300227,0.001736367,0.04460065,0.0001880274,0.00007001904,0.0002662876,0.02187214,0.00009081874,0.001152862],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01554305,"threshold_uncertainty_score":0.03090513,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09472786522244901,"score_gpt":0.3669694291482491,"score_spread":0.2722415639258001,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}