{"id":"W4391157705","doi":"10.48550/arxiv.2401.11373","title":"Finding a Needle in the Adversarial Haystack: A Targeted Paraphrasing Approach For Uncovering Edge Cases with Minimal Distribution Distortion","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Alliance de recherche numérique du Canada","keywords":"Adversarial system; Generalizability theory; Computer science; Artificial intelligence; Exploit; Classifier (UML); Language model; Machine learning; Generator (circuit theory); Natural language processing; Mathematics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002962427,0.001198628,0.001288365,0.001195809,0.0007822191,0.001381459,0.001846474,0.002097845,0.002653887],"category_scores_gemma":[0.01482062,0.0007144633,0.001201123,0.0007282874,0.002396428,0.003177928,0.003391413,0.002938996,0.001264667],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009132457,"about_ca_system_score_gemma":0.001006313,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009706504,"about_ca_topic_score_gemma":0.001341985,"domain_scores_codex":[0.9980558,0.0008695857,0.00009206453,0.0003630185,0.0004598818,0.0001596541],"domain_scores_gemma":[0.9932841,0.004498697,0.0005606916,0.001116288,0.0003285926,0.0002116463],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003563692,0.0002636623,0.003381108,0.0002508461,0.0001531067,0.001034382,0.0005842953,0.6941741,0.01828304,0.09121756,0.00841181,0.1818899],"study_design_scores_gemma":[0.00001232329,0.00004890561,0.0000880783,0.00001528574,0.000008967038,0.0001153037,0.00002421234,0.9629082,0.002459703,0.03350445,0.0008047235,0.000009793891],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02373823,0.000237708,0.9723067,0.000531205,0.00002649311,0.00007785042,0.00007395084,0.001004167,0.002003666],"genre_scores_gemma":[0.7366451,0.0003120005,0.2562844,0.0008974947,0.0001157775,0.0002578919,0.0004118503,0.000467311,0.004608215],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002962427,"threshold_uncertainty_score":0.01566696,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07982276981754907,"score_gpt":0.2024404657202215,"score_spread":0.1226176959026724,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}