{"id":"W4415230413","doi":"10.36227/techrxiv.176053480.01524488/v1","title":"Semantic Pruning of Requirement Specifications: An NLP Framework for Redundancy Detection","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Redundancy (engineering); Word2vec; Software; Pruning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002840942,0.001437947,0.0007412947,0.004504119,0.000755865,0.001307131,0.001261833,0.0009663791,0.002155867],"category_scores_gemma":[0.01331741,0.0005583799,0.001317282,0.002627463,0.0009060432,0.00276919,0.00191962,0.001833151,0.001674204],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009770837,"about_ca_system_score_gemma":0.002883352,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005093715,"about_ca_topic_score_gemma":0.00888647,"domain_scores_codex":[0.996017,0.001275625,0.0003843975,0.0007190647,0.001464472,0.000139404],"domain_scores_gemma":[0.9916358,0.003950372,0.0009874682,0.001101335,0.002206167,0.0001189251],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002323864,0.0002330383,0.00445408,0.001678175,0.0001759338,0.0008479921,0.001487214,0.04013942,0.03623312,0.03290848,0.02768578,0.8539243],"study_design_scores_gemma":[0.00005661227,0.0001996612,0.003290271,0.0002973851,0.0001461997,0.001335803,0.0007747994,0.8306307,0.04365931,0.06475494,0.05476388,0.00009032762],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.009192774,0.0002822416,0.9820185,0.0004167021,0.00005983359,0.0002332454,0.001562422,0.00503175,0.001202566],"genre_scores_gemma":[0.1056743,0.0003496498,0.8847745,0.0002551381,0.0000602477,0.0003407183,0.006206196,0.0005232343,0.001816018],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005093715,"threshold_uncertainty_score":0.01502448,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09976732863886802,"score_gpt":0.354254591948756,"score_spread":0.254487263309888,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}