{"id":"W4410521503","doi":"10.1038/s41557-025-01815-x","title":"A framework for evaluating the chemical knowledge and reasoning abilities of large language models against the expertise of chemists","year":2025,"lang":"en","type":"article","venue":"Nature Chemistry","topic":"Machine Learning in Materials Science","field":"Materials Science","cited_by":55,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Consejo Superior de Investigaciones Científicas; HORIZON EUROPE Framework Programme; Slezská Univerzita v Opavě; Helmholtz Association; Office of Multicultural Interests Department of Local Government and Communities; Ministerio de Economía y Competitividad; Southeastern Ontario Academic Medical Organization; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; European Commission; Ministerio de Ciencia, Innovación y Universidades; Deutsche Forschungsgemeinschaft; UK Research and Innovation; Agencia Estatal de Investigación; U.S. Department of State; US-UK Fulbright Commission; Friedrich-Schiller-Universität Jena; Carl-Zeiss-Stiftung; Government of the United Kingdom; Fulbright Association; National Science Foundation","keywords":"Benchmarking; Harm; Process (computing); Value (mathematics); Chemistry; Management science; Computer science; Cognitive science; Psychology; Engineering; Social psychology; Machine learning; Programming language; Management","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01536345,0.003233883,0.001265198,0.008393092,0.00122499,0.003368037,0.002985871,0.002708938,0.004235483],"category_scores_gemma":[0.07014376,0.0008357188,0.002211925,0.002689451,0.00191308,0.005876827,0.004159079,0.003424321,0.001863427],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002371878,"about_ca_system_score_gemma":0.003584706,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01080282,"about_ca_topic_score_gemma":0.01329371,"domain_scores_codex":[0.9898477,0.006014706,0.0007336191,0.001404931,0.001617009,0.0003819878],"domain_scores_gemma":[0.9559932,0.03296224,0.002878032,0.003570686,0.003643802,0.0009520625],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001118375,0.001791187,0.03142639,0.002409108,0.001333219,0.0002978027,0.001157354,0.5464594,0.01526705,0.04724532,0.03105381,0.3204411],"study_design_scores_gemma":[0.00008730561,0.0004424329,0.002989219,0.0001588363,0.0001389398,0.0001051465,0.0001899187,0.9472905,0.007112898,0.03379605,0.007608376,0.0000804501],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06969289,0.002346056,0.8860986,0.002357746,0.0002001128,0.001046085,0.008422986,0.01714029,0.0126953],"genre_scores_gemma":[0.4512185,0.0005601778,0.5334159,0.0008618308,0.0001323059,0.001365219,0.01021625,0.0005774758,0.001652368],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9846365,"threshold_uncertainty_score":0.08125067,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01131941292029752,"score_gpt":0.3572969330387604,"score_spread":0.3459775201184628,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}