{"id":"W4413212461","doi":"10.1109/ms.2025.3597574","title":"When AI-Generated Unit Tests Validate Bugs: The Risk of Faulty Assertions","year":2025,"lang":"en","type":"article","venue":"IEEE Software","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Software bug; Computer science; Unit testing; Software engineering; Programming language; Reliability engineering; Software; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01796497,0.0006537081,0.0005822778,0.00178561,0.0005128158,0.00230796,0.002039561,0.002451472,0.002356551],"category_scores_gemma":[0.2260777,0.000568928,0.000561498,0.0007805277,0.002264295,0.003873175,0.001906459,0.001624306,0.001028268],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009626049,"about_ca_system_score_gemma":0.001160663,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001857859,"about_ca_topic_score_gemma":0.001313898,"domain_scores_codex":[0.9740121,0.01160966,0.001040147,0.002023715,0.01053558,0.0007788102],"domain_scores_gemma":[0.6486692,0.2690359,0.02481758,0.04051402,0.01563566,0.001327582],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002208956,0.0005613162,0.233629,0.00132869,0.0004990993,0.00650636,0.008270966,0.1352718,0.04936937,0.07221989,0.01405336,0.4760812],"study_design_scores_gemma":[0.0003565616,0.002287181,0.02792418,0.001509352,0.0004440497,0.008344737,0.002039568,0.699391,0.1296779,0.09815353,0.02959812,0.0002737783],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5405851,0.002022031,0.4250059,0.005361979,0.0004440391,0.0002798324,0.0004487513,0.01521551,0.01063681],"genre_scores_gemma":[0.9229412,0.0001926798,0.07348812,0.0008663571,0.00006825737,0.00006793318,0.0002854633,0.001080028,0.001010014],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01796497,"threshold_uncertainty_score":0.09500903,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02597969980246511,"score_gpt":0.2993781457658547,"score_spread":0.2733984459633896,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}