{"id":"W1973328576","doi":"10.1016/j.febslet.2006.02.003","title":"A complete small molecule dataset from the protein data bank","year":2006,"lang":"en","type":"article","venue":"FEBS Letters","topic":"Protein Structure and Dynamics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":47,"is_retracted":false,"has_abstract":true,"ca_institutions":"Lunenfeld-Tanenbaum Research Institute; Mount Sinai Hospital","funders":"","keywords":"Protein Data Bank; PubChem; Molecule; Ligand (biochemistry); Chemistry; Set (abstract data type); Small molecule; Data set; Monomer; Computer science; Aromaticity; Data mining; Crystallography; Protein structure; Artificial intelligence; Biochemistry; Receptor; Organic chemistry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001545425,0.003026176,0.003451457,0.004055424,0.001630702,0.002180218,0.002639104,0.002570131,0.02797393],"category_scores_gemma":[0.004616381,0.0009246036,0.001799568,0.008609383,0.0004722519,0.001374674,0.001692375,0.00273797,0.05186556],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001572512,"about_ca_system_score_gemma":0.003747855,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00581927,"about_ca_topic_score_gemma":0.009520584,"domain_scores_codex":[0.9981148,0.0002660832,0.0002545437,0.0005667851,0.0005934857,0.0002043413],"domain_scores_gemma":[0.9981433,0.000447126,0.0002544049,0.0004318009,0.0004649778,0.0002584067],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001700677,0.0004656673,0.005733408,0.006549497,0.0004348803,0.0006299875,0.00009360642,0.00421324,0.01800491,0.002834405,0.9385495,0.02079022],"study_design_scores_gemma":[0.001037763,0.0002946859,0.01726575,0.0003764577,0.0003402187,0.0006670374,0.0001179583,0.006566747,0.00888533,0.003347609,0.9609757,0.0001246694],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.003719574,0.001384174,0.0008271745,0.0001553709,0.00006036491,0.00007314973,0.9908976,0.001201032,0.001681527],"genre_scores_gemma":[0.001658444,0.0003714002,0.001224752,0.00005822766,0.000007477175,0.0001764332,0.9960237,0.00005109507,0.0004284798],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.02797393,"threshold_uncertainty_score":0.09358209,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01653860743476964,"score_gpt":0.2243197527190075,"score_spread":0.2077811452842378,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}