{"id":"W4385571199","doi":"10.18653/v1/2023.repl4nlp-1.4","title":"A Multilingual Evaluation of NER Robustness to Adversarial Inputs","year":2023,"lang":"en","type":"article","venue":"","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada; University of Ottawa","funders":"","keywords":"Adversarial system; Hindi; Robustness (evolution); Computer science; German; Training set; Artificial intelligence; Natural language processing; Language model; Test data; Focus (optics); Labeled data; Machine learning; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006935846,0.002074731,0.00113673,0.001057054,0.0008355452,0.0011872,0.001025455,0.001344908,0.003270008],"category_scores_gemma":[0.01525036,0.0003263042,0.001031671,0.0006494499,0.001198107,0.001938097,0.002891726,0.002062815,0.001528607],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001016004,"about_ca_system_score_gemma":0.000663858,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004127499,"about_ca_topic_score_gemma":0.005339165,"domain_scores_codex":[0.9950032,0.002111805,0.0003599482,0.001156377,0.0009957629,0.000372894],"domain_scores_gemma":[0.9910323,0.005103202,0.0004379988,0.001723839,0.001354919,0.0003477624],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001928933,0.0008393488,0.01177048,0.0008956422,0.001154917,0.0006768365,0.0003411853,0.7509822,0.05200005,0.004015847,0.01106382,0.1643308],"study_design_scores_gemma":[0.00009642261,0.001958277,0.01136631,0.0001351538,0.0002495611,0.0008072843,0.0002740142,0.8718556,0.09922715,0.004150551,0.00969813,0.0001815031],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7170392,0.00411199,0.2391896,0.001068439,0.00104033,0.0005036335,0.003816781,0.007165557,0.02606457],"genre_scores_gemma":[0.9458589,0.0004652844,0.03974482,0.0003859736,0.0001020554,0.0001491927,0.006237767,0.0006065837,0.006449538],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006935846,"threshold_uncertainty_score":0.0366807,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04854937151821755,"score_gpt":0.3470135699869087,"score_spread":0.2984641984686912,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}