{"id":"W3194114348","doi":"10.1007/s10664-021-09986-0","title":"Empirical evaluation of tools for hairy requirements engineering tasks","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":45,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Waterloo","funders":"Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Empirical research; Systems engineering; Software engineering; Engineering; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01613807,0.0008280085,0.0004813223,0.002902193,0.0007698184,0.001945292,0.001995827,0.001330706,0.003032925],"category_scores_gemma":[0.1774106,0.0004990667,0.0005227111,0.001584877,0.001030508,0.002736174,0.002939257,0.001194169,0.0009403788],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001292672,"about_ca_system_score_gemma":0.001570498,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001922776,"about_ca_topic_score_gemma":0.003154954,"domain_scores_codex":[0.9840821,0.009330714,0.001385141,0.0009633723,0.003660475,0.0005782144],"domain_scores_gemma":[0.6508252,0.3012716,0.01188787,0.01690819,0.0154945,0.003612624],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.01308387,0.0222708,0.09134758,0.004371434,0.000404185,0.001065705,0.02570763,0.020672,0.03525048,0.004789432,0.006594094,0.7744428],"study_design_scores_gemma":[0.008439044,0.06642298,0.5210679,0.002815084,0.001111755,0.001919171,0.02849996,0.2589056,0.06279161,0.00939551,0.03801745,0.0006139723],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9883222,0.0001691467,0.00805306,0.0001109203,0.00002285563,0.0003798335,0.0001571404,0.0004795176,0.002305364],"genre_scores_gemma":[0.9755033,0.0001514634,0.02192347,0.00007815606,0.0000147898,0.0004066977,0.0005767143,0.00014685,0.001198573],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01613807,"threshold_uncertainty_score":0.08534735,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.134283438534654,"score_gpt":0.3755986931044231,"score_spread":0.241315254569769,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}