{"id":"W4414536420","doi":"10.48550/arxiv.2508.13408","title":"NovoMolGen: Rethinking Molecular Language Model Pretraining","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Ministry of Education, India; Indian Institute of Technology Madras; Compute Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Generative grammar; Chemical space; Generative model; Property (philosophy); Scalability; String (physics); Language model; Coherence (philosophical gambling strategy)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0006096581,0.0003915262,0.0004201268,0.0002167597,0.0001293835,0.0002000648,0.002628909,0.0004571816,0.00001006206],"category_scores_gemma":[0.0001654405,0.000430527,0.0002261132,0.0002492972,0.00003714181,0.0002082228,0.004001234,0.001335254,0.00003097442],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001351874,"about_ca_system_score_gemma":0.0004782545,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00009102497,"about_ca_topic_score_gemma":0.000009892593,"domain_scores_codex":[0.9971823,0.0001058368,0.0004829853,0.001253354,0.0004637097,0.0005117885],"domain_scores_gemma":[0.9970899,0.00007343236,0.0002177379,0.00239107,0.0001147523,0.0001131288],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001132271,0.0001248343,0.01647702,0.0009703269,0.0003931114,0.0007694805,0.06303704,0.774736,0.01273023,0.08346408,0.0003410037,0.04694551],"study_design_scores_gemma":[0.000156319,0.000008932097,0.000238204,0.0004225335,0.00002755126,0.000005483653,0.00004899356,0.9801658,0.003795839,0.01464074,0.00005006646,0.000439481],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.348123,0.0006617776,0.6417542,0.0005978107,0.0006967323,0.0002097138,0.000006149884,0.0004304509,0.007520156],"genre_scores_gemma":[0.7828938,0.00001723141,0.2144654,0.001310258,0.0001258254,0.00003815125,0.00001233923,0.00002234418,0.00111463],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.4347707,"threshold_uncertainty_score":0.9998146,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06047665394454849,"score_gpt":0.2974667084531031,"score_spread":0.2369900545085546,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}