{"id":"W4413451213","doi":"10.1016/j.jss.2025.112594","title":"Hybrid approach for multilevel multi-class requirement classification: Impact of stop-word removal and data augmentation","year":2025,"lang":"en","type":"article","venue":"Journal of Systems and Software","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Saskatchewan","funders":"Natural Sciences and Engineering Research Council of Canada; University of Saskatchewan","keywords":"Class (philosophy); Word (group theory); Computer science; Artificial intelligence; Data mining; Natural language processing; Engineering; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008252187,0.00008862618,0.0002469983,0.0001163271,0.00007678296,0.0001085073,0.0003989862,0.00003613816,2.757202e-7],"category_scores_gemma":[0.0001179538,0.00006773682,0.00004821064,0.00005984645,0.00002281012,0.0004958434,0.0001408698,0.00007482561,4.475279e-8],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000739082,"about_ca_system_score_gemma":0.0001401744,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004885877,"about_ca_topic_score_gemma":9.303146e-7,"domain_scores_codex":[0.9988992,0.00004944925,0.0005394998,0.0002176105,0.0001908006,0.0001034678],"domain_scores_gemma":[0.9987718,0.00009567932,0.0004460717,0.0003913302,0.000237159,0.00005793057],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003182043,0.0007346595,0.04197823,0.00394665,0.001097153,0.00003854395,0.004217453,0.02089968,0.007283835,0.01353866,0.006775206,0.8991717],"study_design_scores_gemma":[0.001324088,0.00008679483,0.01309173,0.0002376482,0.00002992908,0.0001212594,0.0002505438,0.9842137,0.00003836387,0.0001245211,0.0004047905,0.00007667799],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07824583,0.001219088,0.9198932,0.00008276528,0.0002682097,0.0002460036,0.00002592926,0.000009654394,0.000009319736],"genre_scores_gemma":[0.7268901,0.0000472951,0.2728748,0.00001246605,0.00006127276,0.000004220621,0.000006996761,0.000003654847,0.00009915212],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.963314,"threshold_uncertainty_score":0.2762227,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.133419085261096,"score_gpt":0.3621934879905279,"score_spread":0.2287744027294318,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}