{"id":"W4380606480","doi":"10.1101/2023.06.14.544984","title":"Generative artificial intelligence GPT-4 accelerates knowledge mining and machine learning for synthetic biology","year":2023,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Microbial Metabolic Engineering and Bioproduction","field":"Biochemistry, Genetics and Molecular Biology","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"Impact","funders":"Advanced Research Projects Agency; U.S. Department of Defense; National Science Foundation","keywords":"Yarrowia; Artificial intelligence; Machine learning; Synthetic biology; Computer science; Information extraction; Generative grammar; Biomanufacturing; Knowledge extraction; Natural language processing; Yeast; Biotechnology; Chemistry; Biology; Computational biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0005867014,0.0004402473,0.0004111154,0.0001779012,0.0002025319,0.0001144154,0.000251392,0.0005435796,0.000004923706],"category_scores_gemma":[0.0006478209,0.0004545867,0.0001128934,0.0001933753,0.0001317914,0.000005582457,0.0004131616,0.000324476,0.00000985223],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003536751,"about_ca_system_score_gemma":0.0001516568,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000186099,"about_ca_topic_score_gemma":0.00000965043,"domain_scores_codex":[0.9979005,0.0001165259,0.0004057414,0.001090562,0.00006182796,0.0004248654],"domain_scores_gemma":[0.9988812,0.00002980399,0.0002019456,0.0004692923,0.0002980194,0.000119793],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00003699432,0.00004103257,0.0002780918,0.0002041667,0.0001278227,0.000001203963,0.00001708833,0.0005747988,0.9982495,0.0002790722,0.00008499645,0.0001052181],"study_design_scores_gemma":[0.00009304666,0.0001508285,0.0004772616,0.0001121636,0.0000783173,5.854345e-8,0.000006785092,0.003670535,0.9854254,0.00001025189,0.009422817,0.0005526051],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9288008,0.006820047,0.060945,0.0002486633,0.002082142,0.0006672418,0.000218262,0.0002157967,0.000002007581],"genre_scores_gemma":[0.9867131,0.001423141,0.009781478,0.00002681797,0.001675173,0.0002106827,0.000009473073,0.0001216725,0.00003848676],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.05791223,"threshold_uncertainty_score":0.9997906,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03617808356001447,"score_gpt":0.2599991571661904,"score_spread":0.2238210736061759,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}