{"id":"W3213820586","doi":"","title":"How to Select One Among All? An Extensive Empirical Study Towards the Robustness of Knowledge Distillation in Natural Language Understanding.","year":2021,"lang":"en","type":"article","venue":"Empirical Methods in Natural Language Processing","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo; Queen's University","funders":"","keywords":"Robustness (evolution); Computer science; Adversarial system; Distillation; Artificial intelligence; Machine learning; Benchmark (surveying); Domain knowledge; Natural language; Artificial neural network; Theoretical computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01273921,0.00147521,0.001116286,0.001803456,0.001022326,0.002132871,0.001763091,0.002141556,0.00312405],"category_scores_gemma":[0.05639349,0.0004408182,0.001116577,0.001440265,0.002997841,0.006609843,0.002055889,0.004038503,0.001127039],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001286654,"about_ca_system_score_gemma":0.001245728,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002915345,"about_ca_topic_score_gemma":0.003289579,"domain_scores_codex":[0.9921998,0.003666449,0.0004425801,0.001934038,0.001517523,0.0002395153],"domain_scores_gemma":[0.9638404,0.02859019,0.001181043,0.004717668,0.001289604,0.0003811133],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007186966,0.0005284288,0.01774077,0.00125212,0.0008522728,0.0001718809,0.0003500491,0.3827002,0.003372689,0.05373875,0.02137095,0.5172032],"study_design_scores_gemma":[0.00006294808,0.0002698683,0.003425262,0.0002683597,0.000113061,0.0004560672,0.0003088182,0.8818595,0.006287686,0.09563284,0.01125111,0.00006451605],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2339944,0.03146292,0.6979976,0.008438332,0.0005060991,0.0004465457,0.001418,0.002171747,0.0235643],"genre_scores_gemma":[0.854183,0.003561956,0.1357512,0.001123835,0.0001572087,0.0001339172,0.001777895,0.0003436557,0.002967439],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01273921,"threshold_uncertainty_score":0.0673722,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1018436876354653,"score_gpt":0.4440969145619636,"score_spread":0.3422532269264983,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}