{"id":"W4407375771","doi":"10.1109/tse.2025.3541166","title":"Automated Test Case Repair Using Language Models","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":false,"ca_institutions":"Carleton University; University of Ottawa","funders":"","keywords":"Computer science; Programming language; Test (biology); Software engineering; Model-based testing; Reliability engineering; Test case; Natural language processing; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001408475,0.001332396,0.0009020319,0.001532322,0.000504591,0.001753727,0.002124179,0.001191475,0.004174627],"category_scores_gemma":[0.01053442,0.0008102179,0.001580284,0.0006877905,0.0007569813,0.002699861,0.001536552,0.001229314,0.001145697],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001051797,"about_ca_system_score_gemma":0.002127741,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005527335,"about_ca_topic_score_gemma":0.008360042,"domain_scores_codex":[0.9974821,0.0009828376,0.000154993,0.0003489352,0.00079894,0.0002322165],"domain_scores_gemma":[0.9899734,0.006243499,0.0007479583,0.001821464,0.001064068,0.0001496147],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008766106,0.0006637577,0.005510743,0.0005797607,0.000232261,0.001290413,0.000620761,0.488429,0.03810626,0.03786708,0.01009853,0.4157248],"study_design_scores_gemma":[0.00004460781,0.00005852451,0.0001709583,0.00002826341,0.00004210971,0.00009985192,0.00004486083,0.9765317,0.008725635,0.01257193,0.001663375,0.00001810924],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03533879,0.0001613463,0.9407604,0.00024963,0.00005023406,0.0001479439,0.000249248,0.02088972,0.002152706],"genre_scores_gemma":[0.6549723,0.0001234976,0.340464,0.0001225707,0.0000206672,0.0002041125,0.0006969121,0.001559019,0.001836851],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005527335,"threshold_uncertainty_score":0.01396549,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0169379124492823,"score_gpt":0.264907490693648,"score_spread":0.2479695782443657,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}