{"id":"W3213180921","doi":"10.18653/v1/2021.emnlp-main.603","title":"Universal-KD: Attention-based Output-Grounded Intermediate Layer Knowledge Distillation","year":2021,"lang":"en","type":"article","venue":"Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing","topic":"Topic Modeling","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Layer (electronics); Distillation; Matching (statistics); Interpretability; Projection (relational algebra); Base (topology); Space (punctuation); Architecture; Artificial intelligence; Deep learning; Computer architecture; Machine learning; Algorithm; Mathematics; Nanotechnology; Chromatography; Operating system; Materials science; Chemistry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001106557,0.001486948,0.001243675,0.00120745,0.0007569246,0.001530648,0.003529372,0.001490115,0.007275992],"category_scores_gemma":[0.004092461,0.0006629612,0.001347269,0.001467243,0.001089838,0.005337355,0.004388689,0.003035701,0.002226051],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001197233,"about_ca_system_score_gemma":0.002089198,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007131571,"about_ca_topic_score_gemma":0.01052419,"domain_scores_codex":[0.9991096,0.0001402738,0.00006651468,0.0003374857,0.0001781795,0.0001680753],"domain_scores_gemma":[0.9987871,0.0004130235,0.00008296138,0.0004182696,0.0002216544,0.00007696138],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003326776,0.0002347432,0.001060676,0.0003899106,0.0001364953,0.0002207704,0.0004677069,0.1124056,0.01539009,0.02642752,0.01294352,0.8299904],"study_design_scores_gemma":[0.00004073467,0.00008623925,0.0003244895,0.0000388218,0.00006153434,0.00009096831,0.00008379138,0.9381472,0.01706261,0.03694777,0.007083495,0.00003233843],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01735344,0.0005491757,0.9684529,0.0002849874,0.0001598253,0.0001003597,0.0003745075,0.009335288,0.003389508],"genre_scores_gemma":[0.5222674,0.0004082308,0.46205,0.0006167882,0.0001055348,0.0002622218,0.002481685,0.0007841099,0.01102404],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007275992,"threshold_uncertainty_score":0.02434063,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07139261601228206,"score_gpt":0.3974343703737435,"score_spread":0.3260417543614614,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}