{"id":"W4414536420","doi":"10.48550/arxiv.2508.13408","title":"NovoMolGen: Rethinking Molecular Language Model Pretraining","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Ministry of Education, India; Indian Institute of Technology Madras; Compute Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Generative grammar; Chemical space; Generative model; Property (philosophy); Scalability; String (physics); Language model; Coherence (philosophical gambling strategy)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001244463,0.0009663427,0.0007386615,0.0006463631,0.0003586113,0.001061265,0.002348884,0.001191186,0.005373001],"category_scores_gemma":[0.005473805,0.0005834528,0.001033561,0.0005875567,0.0005508129,0.002455198,0.0013522,0.002860468,0.002598926],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009347452,"about_ca_system_score_gemma":0.001500827,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002874808,"about_ca_topic_score_gemma":0.008767461,"domain_scores_codex":[0.9996146,0.0001168524,0.00002203956,0.0001168483,0.00009345783,0.000036235],"domain_scores_gemma":[0.9981838,0.001288,0.00007364876,0.0002723517,0.0001203045,0.00006182402],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002641008,0.0002290768,0.003096993,0.0007600934,0.0001877105,0.0002622617,0.0001946827,0.6608542,0.01429501,0.03145353,0.0182245,0.2701779],"study_design_scores_gemma":[0.00002941464,0.00005853544,0.0001046702,0.00002434249,0.00001644176,0.0000443381,0.00002126274,0.9815754,0.004921715,0.009398496,0.003795814,0.00000954066],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07234373,0.001928946,0.8959593,0.001066549,0.0002804588,0.0002010929,0.002046548,0.01861182,0.007561574],"genre_scores_gemma":[0.4887137,0.001505412,0.4919759,0.001202274,0.0001073527,0.000523454,0.008253373,0.002442239,0.005276253],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005373001,"threshold_uncertainty_score":0.0179745,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06047665394454849,"score_gpt":0.2974667084531031,"score_spread":0.2369900545085546,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}