{"id":"W4411576240","doi":"10.26434/chemrxiv-2025-rdf97","title":"A molecular fragment database generated through simulated sequential single-bond-breaking","year":2025,"lang":"en","type":"preprint","venue":"ChemRxiv","topic":"Computational Drug Discovery Methods","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Trent University","funders":"U.S. Forest Service; U.S. Department of Agriculture","keywords":"Fragment (logic); Computer science; Database; Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001409901,0.001353114,0.001517979,0.001758184,0.0008213756,0.001356351,0.004537221,0.00196653,0.02579367],"category_scores_gemma":[0.004305811,0.0009196466,0.001211549,0.002125563,0.0004028796,0.001618637,0.001276205,0.002002073,0.006313832],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00133025,"about_ca_system_score_gemma":0.001827528,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003721132,"about_ca_topic_score_gemma":0.004432538,"domain_scores_codex":[0.9995072,0.00008568295,0.00003358443,0.0001163443,0.0002049917,0.0000522565],"domain_scores_gemma":[0.9990062,0.0004690513,0.00005401489,0.0002249361,0.0001865344,0.00005925613],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003174816,0.0008201938,0.008236879,0.002517948,0.0007723618,0.001545591,0.0004717109,0.5142717,0.02943978,0.07917454,0.1773936,0.1821809],"study_design_scores_gemma":[0.0006419111,0.0002377036,0.0006267274,0.00005711265,0.00008744276,0.0002897166,0.00005474162,0.8921362,0.01895681,0.01985613,0.06698269,0.00007288196],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"dataset","genre_scores_codex":[0.1228041,0.001698646,0.629653,0.001059042,0.0006833654,0.001341186,0.1353344,0.07898644,0.02843989],"genre_scores_gemma":[0.2855958,0.001044927,0.5528694,0.0003317402,0.00006592439,0.002198748,0.1468123,0.004592794,0.006488386],"genre_candidate":"dataset","genre_consensus":null,"teacher_disagreement_score":0.02579367,"threshold_uncertainty_score":0.08628845,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0609696490752425,"score_gpt":0.3405491300206834,"score_spread":0.2795794809454409,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}