{"id":"W2129377409","doi":"10.1109/icsm.2015.7332456","title":"An empirical study of bugs in test code","year":2015,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":108,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Software bug; Computer science; Code (set theory); Test (biology); Root cause; Regression testing; Code coverage; Programming language; Software; Reliability engineering; Software development; Engineering; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01220853,0.0004032945,0.0003227047,0.005050029,0.0006510147,0.001596186,0.001060989,0.0009651897,0.002078116],"category_scores_gemma":[0.2211746,0.0005095843,0.0004023285,0.005326432,0.00215175,0.003564602,0.001414373,0.001256697,0.0004219019],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009727051,"about_ca_system_score_gemma":0.0007795447,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002083359,"about_ca_topic_score_gemma":0.002033554,"domain_scores_codex":[0.9760105,0.01023787,0.002728961,0.002489471,0.007650849,0.0008824648],"domain_scores_gemma":[0.4882021,0.3663717,0.1006088,0.0154447,0.02576442,0.003608339],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0001217925,0.0001715828,0.9819883,0.0001968367,0.00009717107,0.000178888,0.001894228,0.0004947485,0.0003545057,0.0007170698,0.0007480891,0.01303688],"study_design_scores_gemma":[0.00002860365,0.0004602126,0.9859254,0.0002072804,0.00004851873,0.001083233,0.003631251,0.003859547,0.0007349438,0.0008120622,0.003184766,0.00002427866],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9941293,0.0004888722,0.002217236,0.0002874976,0.0000101067,0.00007076171,0.0007957991,0.0000341525,0.001966426],"genre_scores_gemma":[0.9981765,0.0001402245,0.0006911788,0.00004891873,0.00001120415,0.00006577915,0.0006715338,0.00001671928,0.0001779733],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01220853,"threshold_uncertainty_score":0.06456566,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08033739344236439,"score_gpt":0.3817909511334575,"score_spread":0.3014535576910931,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}