{"id":"W4380606480","doi":"10.1101/2023.06.14.544984","title":"Generative artificial intelligence GPT-4 accelerates knowledge mining and machine learning for synthetic biology","year":2023,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Microbial Metabolic Engineering and Bioproduction","field":"Biochemistry, Genetics and Molecular Biology","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"Impact","funders":"Advanced Research Projects Agency; U.S. Department of Defense; National Science Foundation","keywords":"Yarrowia; Artificial intelligence; Machine learning; Synthetic biology; Computer science; Information extraction; Generative grammar; Biomanufacturing; Knowledge extraction; Natural language processing; Yeast; Biotechnology; Chemistry; Biology; Computational biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001749732,0.0008210419,0.0005494503,0.002191937,0.0002896161,0.001484573,0.001208762,0.0008367132,0.003902318],"category_scores_gemma":[0.005843608,0.000480912,0.001574347,0.001873462,0.0006425802,0.001263704,0.001593861,0.001462083,0.001735744],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006871694,"about_ca_system_score_gemma":0.00147174,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001611372,"about_ca_topic_score_gemma":0.001823793,"domain_scores_codex":[0.9991754,0.0002772857,0.00005334858,0.0001796628,0.0002592162,0.00005509545],"domain_scores_gemma":[0.9971111,0.001645109,0.000203477,0.0006477415,0.0002912769,0.0001013646],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004934533,0.0004599864,0.01094991,0.0009990338,0.0005707791,0.00051549,0.0003227981,0.3360402,0.03491566,0.03238661,0.04049048,0.5418556],"study_design_scores_gemma":[0.00006149758,0.00009175757,0.0009189147,0.0000282694,0.00004558325,0.00008473096,0.00002414747,0.9452811,0.01574625,0.02471499,0.01298298,0.00001971742],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07731897,0.0005684629,0.8497661,0.001799441,0.0002405447,0.0003451608,0.005047436,0.05781378,0.007100013],"genre_scores_gemma":[0.2664985,0.0004590184,0.7156266,0.000429431,0.000103881,0.0003860664,0.01258236,0.001421952,0.002492226],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003902318,"threshold_uncertainty_score":0.01305455,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03617808356001447,"score_gpt":0.2599991571661904,"score_spread":0.2238210736061759,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}