{"id":"W4406957873","doi":"10.1145/3715693","title":"Assessing the Robustness of Test Selection Methods for Deep Neural Networks","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Robustness (evolution); Artificial intelligence; Artificial neural network; Machine learning; Robustness testing; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0404065,0.001421634,0.0007739385,0.003063953,0.0006309539,0.001155607,0.001907362,0.001899458,0.0007317656],"category_scores_gemma":[0.1991591,0.0005113108,0.0009153992,0.001314525,0.001898633,0.001729762,0.002000135,0.001960322,0.0003274844],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001309959,"about_ca_system_score_gemma":0.001267258,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002086159,"about_ca_topic_score_gemma":0.001696305,"domain_scores_codex":[0.9754701,0.01387356,0.001920637,0.002141975,0.005849483,0.0007441508],"domain_scores_gemma":[0.7097489,0.2456345,0.01423514,0.01868482,0.0102186,0.001478001],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002546536,0.0005330928,0.1404047,0.0007260109,0.001342235,0.0003760024,0.0004973376,0.5728644,0.01108735,0.01090358,0.004804438,0.2539143],"study_design_scores_gemma":[0.0000992087,0.0007744652,0.01163297,0.0001247513,0.00009790899,0.0002447726,0.0001231781,0.9616997,0.01572752,0.008327031,0.001107535,0.00004095096],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5506164,0.004046886,0.4372019,0.001437171,0.000283984,0.0004383546,0.0007133895,0.002239992,0.003021877],"genre_scores_gemma":[0.9368037,0.0002823818,0.06110666,0.0002197093,0.00007277293,0.0002128059,0.0007267164,0.0001639011,0.0004112702],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0404065,"threshold_uncertainty_score":0.2136925,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05595642973612091,"score_gpt":0.3873494532374797,"score_spread":0.3313930235013587,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}