{"id":"W4414587094","doi":"10.1007/s11761-025-00474-7","title":"Benchmarking large language models for supply chain risk identification: an extended evaluation within the LARD-SC framework","year":2025,"lang":"en","type":"article","venue":"Service Oriented Computing and Applications","topic":"Supply Chain Resilience and Risk Management","field":"Business, Management and Accounting","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université du Québec à Montréal","funders":"University of New South Wales","keywords":"Benchmarking; Interpretability; Blueprint; Supply chain; Identification (biology); Supply chain risk management; Risk management; Set (abstract data type)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02188163,0.001465443,0.001010731,0.00294297,0.0005876009,0.002844495,0.002361167,0.001678275,0.001764737],"category_scores_gemma":[0.0615709,0.0003661069,0.001460206,0.001974739,0.001238246,0.002737072,0.002743153,0.002440046,0.0005488848],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002402786,"about_ca_system_score_gemma":0.003300665,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01949654,"about_ca_topic_score_gemma":0.01431735,"domain_scores_codex":[0.9863449,0.008899587,0.0008162816,0.0009579503,0.002612681,0.000368616],"domain_scores_gemma":[0.9508081,0.03572742,0.001485905,0.004744621,0.006267112,0.0009668965],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001120714,0.0008950189,0.024213,0.0006907248,0.0003874604,0.000229285,0.0005352815,0.8680276,0.002352619,0.01731937,0.007153527,0.07707548],"study_design_scores_gemma":[0.00003865169,0.000149995,0.0007638565,0.00003425778,0.00002013122,0.00002340867,0.00009122583,0.9938291,0.0008133614,0.003369364,0.0008484839,0.0000182236],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5938638,0.001884653,0.3760254,0.003012501,0.0003963767,0.0009377543,0.004169027,0.0104481,0.009262349],"genre_scores_gemma":[0.8148027,0.0002830123,0.1784331,0.0003414942,0.00006370768,0.0003011751,0.004561123,0.0003543317,0.0008594351],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02188163,"threshold_uncertainty_score":0.1157225,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01134495648519082,"score_gpt":0.289217485497521,"score_spread":0.2778725290123302,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}