{"id":"W4242311369","doi":"10.26434/chemrxiv-2021-9nhm1","title":"Sanitize It Yourself: human-based sanitization checker against machine-generated chemical structures","year":2021,"lang":"en","type":"preprint","venue":"ChemRxiv","topic":"Computational Drug Discovery Methods","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Chemical space; Space (punctuation); Embedded system; Computer engineering; Artificial intelligence; Drug discovery; Operating system; Bioinformatics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005186255,0.00146569,0.00103135,0.001513337,0.0007185726,0.001468266,0.002466934,0.002155364,0.02192894],"category_scores_gemma":[0.01324753,0.0006365372,0.001307805,0.0006606816,0.001308088,0.001987521,0.00310133,0.001850405,0.006559217],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006801864,"about_ca_system_score_gemma":0.001781248,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001142372,"about_ca_topic_score_gemma":0.002029048,"domain_scores_codex":[0.9979546,0.0005891214,0.0001344937,0.000430873,0.0007123314,0.0001784344],"domain_scores_gemma":[0.9935968,0.003306397,0.0004335901,0.001943668,0.0005377888,0.0001817616],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.004406252,0.001606126,0.01111717,0.001634933,0.0006203003,0.001175551,0.0005791716,0.09962932,0.06210076,0.03755826,0.2381444,0.5414277],"study_design_scores_gemma":[0.0006050397,0.0005165908,0.001152343,0.000147066,0.00009864896,0.0003443483,0.00008775075,0.8191401,0.09138623,0.02469414,0.06173364,0.00009402025],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"software","genre_gemma":"methods","genre_scores_codex":[0.07512029,0.001250279,0.4436609,0.001576399,0.0006364593,0.0005320223,0.004353935,0.4581767,0.01469291],"genre_scores_gemma":[0.4092446,0.0007345694,0.543985,0.001909946,0.0001303275,0.0005992392,0.01065797,0.02084583,0.01189262],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.02192894,"threshold_uncertainty_score":0.07335961,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04264475909364273,"score_gpt":0.3174869362825313,"score_spread":0.2748421771888886,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}