{"id":"W4411505879","doi":"10.1145/3729176.3729200","title":"A Case Study Investigating the Role of Generative AI in Quality Evaluations of Epics in Agile Software Development","year":2025,"lang":"en","type":"article","venue":"","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"IBM (Canada)","funders":"","keywords":"Agile software development; Computer science; Software quality; Software engineering; Software development; Quality (philosophy); Software; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001672396,0.00006197129,0.0001247269,0.0001211399,0.00004039772,0.00002050838,0.0002866552,0.00002454403,0.00000241929],"category_scores_gemma":[0.0009223577,0.00004691936,0.00001389144,0.0006896212,0.00002007814,0.0001676397,0.0002058813,0.0001212028,1.898045e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004027655,"about_ca_system_score_gemma":0.000243074,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001823759,"about_ca_topic_score_gemma":0.001817811,"domain_scores_codex":[0.9990034,0.0002447375,0.0003847777,0.0001401347,0.0001477969,0.0000792046],"domain_scores_gemma":[0.9986979,0.0008138681,0.00009600284,0.0002783855,0.0001012567,0.00001252361],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.000004717908,0.0007734236,0.6424261,0.00008529389,0.0000658408,0.00005270265,0.08367724,0.01828119,0.001024682,0.04012708,0.00009703211,0.2133847],"study_design_scores_gemma":[0.001505701,0.0003407787,0.4426921,0.0003925249,0.00003301052,0.00005824796,0.02348563,0.3405445,0.1215812,0.0679488,0.0007651517,0.000652294],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6097469,0.00005831016,0.3896451,0.000169075,0.00002208647,0.0002408782,3.739912e-7,0.00004232063,0.00007486184],"genre_scores_gemma":[0.7130198,7.845487e-7,0.286843,0.00007262716,0.000001453391,0.00004988765,2.705754e-7,0.000001335364,0.00001079788],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3222633,"threshold_uncertainty_score":0.2756991,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06305782420443068,"score_gpt":0.3873828332371154,"score_spread":0.3243250090326847,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}