{"id":"W4394998203","doi":"10.21203/rs.3.rs-4272773/v1","title":"Hybrid Fragment-SMILES Tokenization for ADMET Prediction in Drug Discovery","year":2024,"lang":"en","type":"preprint","venue":"Research Square","topic":"Computational Drug Discovery Methods","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"National Research Council Canada; Brock University","funders":"National Research Council Canada; Natural Sciences and Engineering Research Council of Canada","keywords":"Fragment (logic); Lexical analysis; Computer science; Drug discovery; Computational biology; Natural language processing; Artificial intelligence; Algorithm; Biology; Bioinformatics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001521086,0.0009709525,0.001939466,0.002115056,0.0009234863,0.001168343,0.002413265,0.0009969661,0.01099556],"category_scores_gemma":[0.004641701,0.0004794202,0.001621737,0.002837666,0.0006630041,0.002269367,0.002187541,0.001462706,0.003803101],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007560264,"about_ca_system_score_gemma":0.00209035,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003292887,"about_ca_topic_score_gemma":0.00606131,"domain_scores_codex":[0.9990695,0.0002660828,0.00008377459,0.0001892091,0.000250555,0.0001408728],"domain_scores_gemma":[0.9978041,0.001018339,0.0001550773,0.0005843661,0.0003163955,0.0001216213],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.006398239,0.0004566309,0.006441988,0.0007368992,0.0004972474,0.0005818086,0.0001833556,0.2447265,0.02125663,0.03288475,0.02707191,0.6587641],"study_design_scores_gemma":[0.0001366432,0.0001954977,0.0006654181,0.00002080539,0.00007913238,0.0001435861,0.00004682683,0.9623705,0.01152041,0.01957788,0.005206344,0.00003696906],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.05585583,0.0006623883,0.9181731,0.0002755408,0.0001665221,0.0001756862,0.003687767,0.01850058,0.002502724],"genre_scores_gemma":[0.4717295,0.0005655587,0.5036998,0.0002965494,0.0002245243,0.0004769049,0.01472873,0.002335259,0.005943028],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01099556,"threshold_uncertainty_score":0.03678381,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05776631353166508,"score_gpt":0.4107230850898549,"score_spread":0.3529567715581898,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}