{"id":"W4389520058","doi":"10.18653/v1/2023.findings-emnlp.35","title":"Toward Stronger Textual Attack Detectors","year":2023,"lang":"en","type":"article","venue":"","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Grand Équipement National De Calcul Intensif","keywords":"Adversarial system; Computer science; Benchmark (surveying); Hyperparameter; Artificial intelligence; Machine learning; Trustworthiness; Computer security","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00533958,0.002544257,0.001964803,0.001836834,0.001031707,0.003793239,0.001678861,0.004196868,0.0264456],"category_scores_gemma":[0.02899402,0.0008041901,0.001126505,0.0007760404,0.002586276,0.005832438,0.004762026,0.006757812,0.01416934],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000924819,"about_ca_system_score_gemma":0.0006177282,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004885806,"about_ca_topic_score_gemma":0.000482966,"domain_scores_codex":[0.9958693,0.00133839,0.0001633119,0.000933347,0.001469974,0.0002255368],"domain_scores_gemma":[0.9831267,0.009833572,0.0008617599,0.002936088,0.002653203,0.0005886121],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000901766,0.0005628226,0.001540449,0.0009758733,0.0003163476,0.0004649084,0.0002515284,0.1267116,0.05076428,0.1990008,0.1623085,0.456201],"study_design_scores_gemma":[0.00007572866,0.0002758881,0.0004562787,0.000167022,0.00009001993,0.0004431038,0.00004737279,0.8031839,0.02742596,0.1355543,0.03221714,0.00006341266],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.011977,0.002335102,0.9388371,0.01211128,0.002507882,0.0002321499,0.0005266736,0.004798818,0.02667401],"genre_scores_gemma":[0.5675713,0.003906498,0.2943094,0.0152541,0.009528218,0.0004058295,0.003154605,0.001878504,0.1039916],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0264456,"threshold_uncertainty_score":0.08846933,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.051490275223493,"score_gpt":0.3023379201925875,"score_spread":0.2508476449690946,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}