{"id":"W4416176098","doi":"10.1039/d5sc05004a","title":"Grammar-driven SMILES standardization with <i>TokenSMILES</i>","year":2025,"lang":"en","type":"article","venue":"Chemical Science","topic":"Machine Learning in Materials Science","field":"Materials Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"Centro de Investigación y de Estudios Avanzados del Instituto Politécnico Nacional","keywords":"Parsing; Redundancy (engineering); Syntax; Standardization; Scalability; Consistency (knowledge bases); Semantics (computer science); Implementation","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002984036,0.001374736,0.001277585,0.001939349,0.001327905,0.00323313,0.003091417,0.001329771,0.0197085],"category_scores_gemma":[0.01128003,0.001119423,0.002339547,0.001898103,0.002352157,0.003958341,0.007946038,0.003855406,0.02105097],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009033013,"about_ca_system_score_gemma":0.00376873,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00169411,"about_ca_topic_score_gemma":0.004604403,"domain_scores_codex":[0.9960719,0.0008638054,0.0003999427,0.0008448131,0.00140892,0.0004107051],"domain_scores_gemma":[0.9929374,0.001370662,0.0003922636,0.003749259,0.001345641,0.0002048089],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005294657,0.0002645361,0.003164003,0.0006212575,0.0001640814,0.0006653726,0.001094867,0.02555199,0.04678462,0.4704682,0.0800304,0.3706612],"study_design_scores_gemma":[0.00009620851,0.0001770994,0.0006784253,0.0002207315,0.00009291498,0.0005472508,0.0003276292,0.2158869,0.1193094,0.4018547,0.260599,0.0002098996],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.007666642,0.0001107962,0.9252217,0.000386693,0.0004047307,0.0002120578,0.002157564,0.04486147,0.01897829],"genre_scores_gemma":[0.1873747,0.0002679631,0.7535383,0.0009025343,0.0003671579,0.0008312593,0.009999547,0.02670931,0.02000928],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0197085,"threshold_uncertainty_score":0.06593144,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.00302477281127441,"score_gpt":0.2426292945020754,"score_spread":0.239604521690801,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}