{"id":"W4408146479","doi":"10.1109/icairc64177.2024.10900215","title":"Robustness of Large Language Models Against Adversarial Attacks","year":2024,"lang":"en","type":"article","venue":"","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Robustness (evolution); Computer science; Adversarial system; Language model; Computer security; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01414744,0.001814351,0.001264602,0.001196645,0.001130799,0.00205659,0.001719893,0.001914292,0.001842159],"category_scores_gemma":[0.06181259,0.0007388988,0.001289028,0.0005294828,0.003245754,0.003961442,0.004481804,0.004598489,0.001107772],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001872887,"about_ca_system_score_gemma":0.001605253,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003111418,"about_ca_topic_score_gemma":0.002296333,"domain_scores_codex":[0.9926629,0.003862943,0.0003928853,0.001124567,0.001423242,0.0005334811],"domain_scores_gemma":[0.9517787,0.03538892,0.002587931,0.008039885,0.001580377,0.0006241503],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0003693292,0.00009032866,0.004214385,0.0001613245,0.0002089177,0.0001864657,0.0001899691,0.932042,0.004827235,0.01584949,0.002738343,0.03912211],"study_design_scores_gemma":[0.00001205904,0.000110897,0.0003432592,0.00002968592,0.00001808134,0.00007717762,0.00003781857,0.9814343,0.002809319,0.01441172,0.0006942854,0.00002127372],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2268912,0.002106625,0.7504938,0.00486653,0.000449044,0.000503755,0.001285279,0.005532885,0.007870952],"genre_scores_gemma":[0.9471224,0.0005677879,0.04786206,0.0008826142,0.0001507392,0.000239922,0.0008938907,0.0003735271,0.001907069],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01414744,"threshold_uncertainty_score":0.07481974,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01302120973298575,"score_gpt":0.2868284806767559,"score_spread":0.2738072709437701,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}