{"id":"W4404620644","doi":"10.1039/d4sc04401k","title":"Assessment of fine-tuned large language models for real-world chemistry and material science applications","year":2024,"lang":"en","type":"article","venue":"Chemical Science","topic":"Machine Learning in Materials Science","field":"Materials Science","cited_by":48,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; University of Waterloo","funders":"NCCR Catalysis; H2020 European Research Council; National Institute of Diabetes and Digestive and Kidney Diseases; H2020 Marie Skłodowska-Curie Actions; Government of the United Kingdom; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; University of Toronto; Agencia Estatal de Investigación; National Center of Competence in Research Materials’ Revolution: Computational Design and Discovery of Novel Materials; Seventh Framework Programme; European Research Council; Consejo Superior de Investigaciones Científicas; Novo Nordisk Fonden; National Institutes of Health; European Regional Development Fund; Ministerio de Ciencia e Innovación; European Commission; UK Research and Innovation; National Science Foundation; Grantham Foundation for the Protection of the Environment; Frances and Augustus Newman Foundation; Novo Nordisk; HORIZON EUROPE Framework Programme; Intramural Research Program; Cambridge Trust; Carl-Zeiss-Stiftung","keywords":"Benchmark (surveying); Computer science; Set (abstract data type); Field (mathematics); Computation; Fine-tuning; Work (physics); Simple (philosophy); Range (aeronautics); Machine learning; Artificial intelligence; Engineering; Algorithm; Epistemology; Programming language","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004991997,0.001188192,0.000837372,0.0009489883,0.0005557137,0.001065761,0.002516222,0.002030317,0.002008322],"category_scores_gemma":[0.01820048,0.0005804526,0.001060284,0.0006649123,0.0007462853,0.002199403,0.00122925,0.002291802,0.0007814628],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001860199,"about_ca_system_score_gemma":0.001407763,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01080802,"about_ca_topic_score_gemma":0.01162398,"domain_scores_codex":[0.9988686,0.0004950615,0.00007353767,0.0002912632,0.0001750623,0.0000963204],"domain_scores_gemma":[0.9903215,0.007018027,0.0003481827,0.001180155,0.0008699183,0.0002622195],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0004259341,0.0004130512,0.00358904,0.0001994319,0.0001527236,0.00007207627,0.00006654203,0.9534816,0.002548803,0.002034547,0.003648267,0.03336806],"study_design_scores_gemma":[0.00003935112,0.00008911097,0.0004341396,0.00001204378,0.0000132078,0.00001330386,0.00002076686,0.9951844,0.00188414,0.001669663,0.0006285227,0.00001142682],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8167421,0.003141209,0.1490664,0.002655832,0.0004565258,0.0004161602,0.004702265,0.01289019,0.009929374],"genre_scores_gemma":[0.9264408,0.0003252571,0.06468187,0.0004930742,0.00006202898,0.000325965,0.005711387,0.0003966891,0.001562954],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01080802,"threshold_uncertainty_score":0.02640051,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01023883856110066,"score_gpt":0.3383258651224839,"score_spread":0.3280870265613833,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}