{"id":"W4387700646","doi":"10.1145/3617946.3617953","title":"Summary of the Fourth International Workshop on Deep Learning for Testing and Testing for Deep Learning (DeepTest 2023)","year":2023,"lang":"en","type":"article","venue":"ACM SIGSOFT Software Engineering Notes","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Deep learning; Dependability; Interpretability; Software engineering; Computer science; Verification and validation; Intersection (aeronautics); Correctness; Software testing; Software system; Artificial intelligence; Software; Engineering; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01076878,0.002431581,0.001635956,0.002607292,0.0008878413,0.004798222,0.002727491,0.002624622,0.0448667],"category_scores_gemma":[0.01028615,0.0009880394,0.001785143,0.001785879,0.0009171867,0.005402865,0.004215044,0.005220601,0.02042399],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002500312,"about_ca_system_score_gemma":0.003486599,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006341319,"about_ca_topic_score_gemma":0.009102645,"domain_scores_codex":[0.9952434,0.001214544,0.0003514959,0.001040876,0.001630015,0.0005196226],"domain_scores_gemma":[0.9908421,0.001619453,0.0001590192,0.001075855,0.004455711,0.001847827],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000403238,0.0003532075,0.0008295973,0.0004191652,0.0001228356,0.0002540967,0.000168585,0.006656603,0.004532142,0.007366902,0.579614,0.3992796],"study_design_scores_gemma":[0.0001833262,0.0005774046,0.002101184,0.0007298396,0.0001488259,0.0005265191,0.0001866862,0.05117589,0.01006445,0.02669337,0.9074755,0.0001370387],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"editorial","genre_scores_codex":[0.0153742,0.07008097,0.6700914,0.05088711,0.07002063,0.001648484,0.009856489,0.01081632,0.1012245],"genre_scores_gemma":[0.1067374,0.04607287,0.2855204,0.01511472,0.02394249,0.001856344,0.05580083,0.007491977,0.457463],"genre_candidate":"editorial","genre_consensus":null,"teacher_disagreement_score":0.0448667,"threshold_uncertainty_score":0.1500941,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02896485804727705,"score_gpt":0.2640472140121904,"score_spread":0.2350823559649133,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}