{"id":"W6979255877","doi":"","title":"Evaluating Multi-Hop Reasoning in Large Language Models: A Chemistry-Centric Case Study","year":2025,"lang":"en","type":"article","venue":"ArXiv.org","topic":"Machine Learning in Materials Science","field":"Materials Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Mitacs","keywords":"Pipeline (software); Context (archaeology); Benchmark (surveying); Automated reasoning; Process (computing); Case-based reasoning; Model-based reasoning","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01138526,0.002164069,0.001000241,0.003476179,0.001399328,0.002748056,0.003765045,0.003353726,0.003327535],"category_scores_gemma":[0.03941348,0.0004635602,0.002082978,0.002920935,0.001615617,0.006058358,0.003043642,0.002803875,0.001677818],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002912151,"about_ca_system_score_gemma":0.002895337,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01251435,"about_ca_topic_score_gemma":0.01935513,"domain_scores_codex":[0.9884449,0.006118198,0.0008530236,0.002206108,0.002047635,0.0003300461],"domain_scores_gemma":[0.963993,0.02789224,0.0009203489,0.004262828,0.002254534,0.0006770755],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001763917,0.004048231,0.02231186,0.005113928,0.001309165,0.002371629,0.001629627,0.5154712,0.02366299,0.03459634,0.0451143,0.3426068],"study_design_scores_gemma":[0.0003126707,0.0006459352,0.003143532,0.0001562239,0.0001946938,0.0006073297,0.0008114833,0.9087859,0.03009844,0.03317449,0.02195899,0.0001102296],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4632861,0.006230546,0.4571261,0.00593891,0.0006388142,0.001860662,0.02624085,0.02439274,0.01428526],"genre_scores_gemma":[0.6059869,0.0008869754,0.3546712,0.001340444,0.0002032167,0.0004986508,0.03329918,0.0007378626,0.00237559],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01251435,"threshold_uncertainty_score":0.06021172,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05388192935518151,"score_gpt":0.3765414905720963,"score_spread":0.3226595612169149,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}