{"id":"W4319663674","doi":"10.1109/tse.2023.3243522","title":"Black-Box Testing of Deep Neural Networks through Test Case Diversity","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":86,"is_retracted":false,"has_abstract":true,"ca_institutions":"École de Technologie Supérieure; University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Black box; Artificial neural network; White-box testing; Test (biology); Diversity (politics); Software testing; Artificial intelligence; Machine learning; Software engineering; Software; Programming language; Software development; Software construction","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006519299,0.001438734,0.0008451802,0.002719207,0.0003883181,0.001205352,0.002055702,0.001237703,0.0007586982],"category_scores_gemma":[0.04638109,0.0004723933,0.0007973266,0.001094853,0.00135931,0.003107501,0.001873377,0.001242875,0.0001681316],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001389049,"about_ca_system_score_gemma":0.001031616,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002968069,"about_ca_topic_score_gemma":0.004130304,"domain_scores_codex":[0.9929224,0.002643467,0.0006068602,0.001445215,0.001918922,0.0004630996],"domain_scores_gemma":[0.9288557,0.05498984,0.006372529,0.005075438,0.003581867,0.001124637],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001290684,0.0005188225,0.1095409,0.0003869871,0.0005112847,0.0006558856,0.0005118351,0.6672203,0.01546193,0.005416331,0.002007034,0.196478],"study_design_scores_gemma":[0.00003526336,0.0002902025,0.005691484,0.00003576665,0.00004316562,0.0001522238,0.00007504083,0.9787965,0.009196274,0.005246384,0.0004163244,0.00002144429],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.875333,0.001120376,0.1194857,0.0003360079,0.0000512078,0.00009477836,0.0005241847,0.001524981,0.001529873],"genre_scores_gemma":[0.9716098,0.00009497542,0.02712383,0.00008977945,0.00002074995,0.00008437032,0.0006373148,0.00009197805,0.0002472018],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006519299,"threshold_uncertainty_score":0.03447777,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02156941011280877,"score_gpt":0.236715522135328,"score_spread":0.2151461120225193,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}