{"id":"W3152607317","doi":"10.18653/v1/2021.eacl-main.212","title":"Annealing Knowledge Distillation","year":2021,"lang":"en","type":"article","venue":"","topic":"Advanced Neural Network Applications","field":"Computer Science","cited_by":69,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Distillation; Artificial neural network; Benchmark (surveying); Simulated annealing; Inference; Artificial intelligence; Annealing (glass); Machine learning; Deep learning; Exploit; Materials science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006527534,0.0009487292,0.001252286,0.0005701326,0.0007269419,0.0009814369,0.002146153,0.001360005,0.00549874],"category_scores_gemma":[0.003820817,0.0005997067,0.0009129568,0.0008193287,0.001193597,0.002556711,0.001856367,0.002787428,0.001351583],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009410497,"about_ca_system_score_gemma":0.001439062,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003310764,"about_ca_topic_score_gemma":0.005278829,"domain_scores_codex":[0.9993994,0.0001107164,0.00003852219,0.0002192519,0.0001539362,0.00007818957],"domain_scores_gemma":[0.9987696,0.0005169915,0.00006685422,0.0004180354,0.000183329,0.00004508663],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000356334,0.0001970355,0.0007823567,0.0003604407,0.0001119421,0.0001723267,0.0003138725,0.5611607,0.01936825,0.07199337,0.007760592,0.3374228],"study_design_scores_gemma":[0.00002416185,0.00005201599,0.0001094706,0.00001460614,0.00001689807,0.00004846708,0.00002111667,0.9613504,0.01137991,0.02291406,0.004050341,0.00001856931],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03697602,0.0006240956,0.9490072,0.0004019589,0.0001717854,0.0001241621,0.0002944787,0.003874446,0.00852578],"genre_scores_gemma":[0.5708171,0.000287638,0.417989,0.0004256932,0.00006763716,0.0002821987,0.0009099982,0.000559485,0.008661211],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00549874,"threshold_uncertainty_score":0.01839513,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02531000715964295,"score_gpt":0.2946477081653764,"score_spread":0.2693377010057335,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}