{"id":"W4411120204","doi":"10.18653/v1/2025.findings-naacl.424","title":"Tooling or Not Tooling? The Impact of Tools on Language Agents for Chemistry Problem Solving","year":2025,"lang":"en","type":"article","venue":"","topic":"Semantic Web and Ontologies","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Argonne National Laboratory; U.S. National Library of Medicine; Institute for Catastrophic Loss Reduction; National Science Foundation","keywords":"Computer science; Manufacturing engineering; Software engineering; Process engineering; Engineering","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002799681,0.0001127736,0.000169028,0.00003073158,0.00009677833,0.0001538029,0.0007994508,0.00005211572,0.00003161083],"category_scores_gemma":[0.0003859222,0.00005787345,0.0001363153,0.0001818067,0.00002773264,0.00015375,0.0001856625,0.00007478611,0.000002891675],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003837817,"about_ca_system_score_gemma":0.0001427569,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001259919,"about_ca_topic_score_gemma":0.00001509953,"domain_scores_codex":[0.9991948,0.00001425305,0.0002105113,0.0002236557,0.0001183408,0.0002384297],"domain_scores_gemma":[0.9987757,0.0006068859,0.00007458795,0.0004657919,0.0000553117,0.00002174532],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001170296,0.0008640392,0.006269905,0.001571192,0.001006164,0.00007858661,0.01420302,0.01065396,0.2258235,0.09427497,0.04930639,0.594778],"study_design_scores_gemma":[0.002005485,0.000420088,0.02619101,0.0005523069,0.00005013662,0.00001840661,0.001851037,0.1819485,0.7818515,0.003323626,0.00121109,0.0005768102],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7238315,0.00009939611,0.2491045,0.0007690864,0.0001975438,0.0005696657,0.000008628626,0.0002105535,0.02520912],"genre_scores_gemma":[0.9779671,0.000005474204,0.01759726,0.0004092061,0.00003911015,0.00002049434,0.000001418146,0.000004206348,0.003955734],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5942012,"threshold_uncertainty_score":0.2360011,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04620901255201522,"score_gpt":0.3422979205710734,"score_spread":0.2960889080190582,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}