{"id":"W4384115435","doi":"10.48550/arxiv.2307.05422","title":"Differential Analysis of Triggers and Benign Features for Black-Box DNN Backdoor Detection","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Tamkeen; York University; Army Research Office; New York University Abu Dhabi","keywords":"Backdoor; Novelty; Computer science; Novelty detection; Artificial intelligence; Detector; Metric (unit); Black box; Artificial neural network; Pattern recognition (psychology); Data mining; Deep learning; Machine learning; Intuition; Engineering","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0002533816,0.000301298,0.0006164007,0.001032451,0.0001757712,0.00009904923,0.00105757,0.0003435678,0.0000136991],"category_scores_gemma":[0.0001617548,0.0003412979,0.0004780091,0.001441065,0.0001575159,0.0002218534,0.001424781,0.000499823,0.000003625229],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001357918,"about_ca_system_score_gemma":0.00007789032,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002641176,"about_ca_topic_score_gemma":0.0002377691,"domain_scores_codex":[0.9980429,0.000168346,0.0002465047,0.001119835,0.0001253857,0.0002970343],"domain_scores_gemma":[0.9980097,0.0004703161,0.0004304583,0.0008134783,0.0001645335,0.0001114838],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001249062,0.0000386547,0.002023118,0.000165205,0.001558178,0.00002637598,0.0004383571,0.9701039,0.0002409786,0.02335031,0.00005486679,0.001875125],"study_design_scores_gemma":[0.0006034609,0.00006635634,0.01042371,0.0000410124,0.001574991,5.40411e-7,0.0001027478,0.9762611,0.0005132938,0.01003647,0.00003117917,0.0003451542],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3167835,0.00001278146,0.6822302,0.00004479075,0.000454973,0.0002442725,0.00002389013,0.0001503296,0.00005530984],"genre_scores_gemma":[0.9959146,0.00007905889,0.002883734,0.00001015337,0.00008166809,0.000001733515,0.00003652273,0.00002343141,0.0009691466],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6793464,"threshold_uncertainty_score":0.9999039,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05699921188946427,"score_gpt":0.2144693550188264,"score_spread":0.1574701431293621,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}