{"id":"W7125644766","doi":"10.1109/cascon66301.2025.00081","title":"Towards Improving the Reliability of LLMs in Requirements Engineering with Structured Confidence and Tag Governance","year":2025,"lang":"","type":"article","venue":"","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Reliability (semiconductor); Corporate governance; Risk management; Risk assessment; Process (computing); Requirements analysis","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001070455,0.0002538345,0.0003124917,0.00007274914,0.00006596325,0.0001894013,0.0009091672,0.0001057992,0.00001413102],"category_scores_gemma":[0.0008307176,0.0001742888,0.00003577159,0.0008240324,0.0001307099,0.0008632676,0.0005687748,0.0003979592,1.796665e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009983128,"about_ca_system_score_gemma":0.0002276973,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001851479,"about_ca_topic_score_gemma":0.00007652411,"domain_scores_codex":[0.9982353,0.00005981045,0.0004905476,0.0005466488,0.0003480614,0.0003196491],"domain_scores_gemma":[0.9981757,0.000495666,0.0002393289,0.0009079312,0.0001337158,0.0000476299],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003856354,0.0002377001,0.08009133,0.004049202,0.0002150589,0.0000570619,0.004740309,0.0488002,0.01166661,0.4439507,0.0002987809,0.4055074],"study_design_scores_gemma":[0.0007913164,0.0003843547,0.2365387,0.001358779,0.00005285703,0.00002191277,0.00007537615,0.718774,0.03722254,0.0027895,0.001458285,0.0005324352],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1188844,0.001292184,0.8773865,0.001085376,0.0004074225,0.0004948094,0.000004376296,0.0001277455,0.0003171632],"genre_scores_gemma":[0.8902156,0.0002767174,0.1093048,0.00008092378,0.00001237755,0.0000192113,1.72764e-7,0.000008094862,0.00008208634],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7713313,"threshold_uncertainty_score":0.7107292,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.00842075722685158,"score_gpt":0.2434121089369344,"score_spread":0.2349913517100828,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}