{"id":"W4408162704","doi":"10.1007/978-3-031-73143-3_15","title":"Practical Guidelines for the Selection and Evaluation of Natural Language Processing Techniques in Requirements Engineering","year":2025,"lang":"en","type":"book-chapter","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Selection (genetic algorithm); Computer science; Natural (archaeology); Natural language processing; Software engineering; Linguistics; Systems engineering; Artificial intelligence; Engineering; History; Archaeology; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02251277,0.002426062,0.001490406,0.008207791,0.001896647,0.008188494,0.005358013,0.003719112,0.03026186],"category_scores_gemma":[0.09165264,0.001804433,0.001076628,0.006487836,0.001906431,0.007658022,0.002802279,0.003871711,0.01515036],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001763179,"about_ca_system_score_gemma":0.003270501,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002952378,"about_ca_topic_score_gemma":0.007712747,"domain_scores_codex":[0.9765556,0.01231101,0.002858106,0.0008662878,0.006976555,0.0004324094],"domain_scores_gemma":[0.9169744,0.05301037,0.002463711,0.006810969,0.01984254,0.0008981101],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001768656,0.0006837005,0.0007672631,0.001998913,0.00004364707,0.0005733438,0.001592589,0.00584865,0.01101808,0.1465611,0.1013114,0.7294246],"study_design_scores_gemma":[0.0005211338,0.0005436104,0.002748956,0.00685576,0.0001846521,0.001820878,0.002767659,0.07828555,0.03346095,0.3545016,0.5179192,0.0003900798],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.002628119,0.003649107,0.9273482,0.004288454,0.0003573331,0.002595272,0.001178829,0.007982549,0.04997225],"genre_scores_gemma":[0.005153777,0.00109331,0.9854119,0.0003690654,0.00006130201,0.0009487319,0.0007618867,0.0006731274,0.005526911],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.03026186,"threshold_uncertainty_score":0.1190604,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1088786616259275,"score_gpt":0.4323619416566628,"score_spread":0.3234832800307353,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}