{"id":"W4413149443","doi":"10.1016/j.eng.2025.07.037","title":"Can large language models solve complex engineering issues? Practical applications in reliability systems engineering","year":2025,"lang":"en","type":"article","venue":"Engineering","topic":"Software Reliability and Analysis Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"National Natural Science Foundation of China","keywords":"Computer science; Reliability (semiconductor); Complex system; Reliability engineering; Systems engineering; Software engineering; Engineering; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00344805,0.001007879,0.001351842,0.0007599719,0.001000318,0.003048267,0.002055368,0.002285776,0.01001117],"category_scores_gemma":[0.04312681,0.0008911152,0.001281774,0.001142009,0.001721465,0.01413786,0.001867978,0.003807235,0.001740976],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009874824,"about_ca_system_score_gemma":0.001842451,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004009155,"about_ca_topic_score_gemma":0.005784312,"domain_scores_codex":[0.9979383,0.001231902,0.0000866979,0.0002218719,0.000391686,0.0001294918],"domain_scores_gemma":[0.9697146,0.02579888,0.000849617,0.002072603,0.001218146,0.0003460917],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001359153,0.0002222287,0.001490389,0.0003460802,0.0001077191,0.0002251948,0.000566637,0.227151,0.001822045,0.687916,0.01437203,0.06564484],"study_design_scores_gemma":[0.00001876081,0.00001242666,0.00006649502,0.0000180843,0.00001317638,0.000032932,0.00007742899,0.3583661,0.0003228592,0.6384303,0.002629679,0.00001180238],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02245573,0.0005459772,0.9620466,0.006825741,0.0001909281,0.00004797947,0.0002592259,0.000977456,0.006650417],"genre_scores_gemma":[0.6373466,0.001331712,0.348097,0.00152524,0.0005956689,0.0003427621,0.0007311405,0.0008330306,0.009196837],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01001117,"threshold_uncertainty_score":0.03349072,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01648185438212156,"score_gpt":0.3031423616824004,"score_spread":0.2866605073002789,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}