{"id":"W4408629203","doi":"10.26434/chemrxiv-2024-zvcb4-v4","title":"3D2SMILES: Translating Physical Molecular Models into Digital DeepSMILES Notations Using Deep Learning","year":2025,"lang":"en","type":"preprint","venue":"ChemRxiv","topic":"Machine Learning in Materials Science","field":"Materials Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Notation; Computer science; Deep learning; Artificial intelligence; Natural language processing; Mathematics; Arithmetic","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0005975023,0.0006176538,0.000754904,0.0002274655,0.0005337251,0.001132554,0.001243431,0.0003139006,0.0001343732],"category_scores_gemma":[0.0005684874,0.0006421655,0.0003064652,0.0003583551,0.0003710387,0.0005792661,0.001168129,0.0009853644,0.00008361732],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001903475,"about_ca_system_score_gemma":0.0003383852,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001859368,"about_ca_topic_score_gemma":0.000007124492,"domain_scores_codex":[0.9963503,0.0002490113,0.0006806261,0.001346238,0.0006916636,0.0006821186],"domain_scores_gemma":[0.9981126,0.000254324,0.0004684319,0.0007649214,0.0002239097,0.0001758007],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000007486779,0.00004352969,0.00009678346,0.0003047676,0.00001249214,0.00001007472,0.003494092,0.5970036,0.39737,0.0007255886,0.000002841509,0.0009287051],"study_design_scores_gemma":[0.0001811465,0.00002095593,0.0000233609,0.000469096,0.00008171409,0.000005614502,0.0003172363,0.8570151,0.1082051,0.03299542,0.00004224776,0.0006430451],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.699246,0.0002031686,0.2963074,0.0001002496,0.0005762115,0.0003056229,0.00001369038,0.000380529,0.002867116],"genre_scores_gemma":[0.9320281,0.000007567724,0.06732696,0.00006609302,0.0002482779,0.00007297946,0.00006952107,0.00006099359,0.0001194935],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.289165,"threshold_uncertainty_score":0.9999044,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02126373074935713,"score_gpt":0.2989196104639774,"score_spread":0.2776558797146202,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}