{"id":"W4385573711","doi":"10.18653/v1/2022.findings-emnlp.385","title":"Continuation KD: Improved Knowledge Distillation through the Lens of Continuation Optimization","year":2022,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Continuation; Computer science; Benchmark (surveying); Generalization; Distillation; Artificial intelligence; Limiting; Noise (video); Machine learning; Image (mathematics); Mathematics; Engineering; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001272037,0.001188384,0.001380856,0.0007344923,0.0006937287,0.001175877,0.002064088,0.001902079,0.004285342],"category_scores_gemma":[0.004817343,0.0006647615,0.0008366422,0.0007478512,0.001597667,0.002547973,0.002599642,0.00294459,0.001459429],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000871633,"about_ca_system_score_gemma":0.002394833,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004790321,"about_ca_topic_score_gemma":0.005621854,"domain_scores_codex":[0.9995889,0.0001265511,0.00002253389,0.0001047515,0.0001021635,0.00005512152],"domain_scores_gemma":[0.9987708,0.0006727062,0.00008014395,0.0002214331,0.0001647382,0.00009010138],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.000232759,0.0001328185,0.0008098575,0.0002141568,0.00007953057,0.0001601892,0.0002304832,0.6421025,0.00707173,0.05641742,0.01025407,0.2822945],"study_design_scores_gemma":[0.00001344223,0.00002546581,0.00003622967,0.00001063731,0.000004729435,0.00001540209,0.000008115217,0.9829314,0.0009216633,0.0149328,0.001092756,0.000007426609],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01040448,0.0003652095,0.9844431,0.0003295787,0.00006554159,0.0000402659,0.00008054415,0.00177037,0.002500934],"genre_scores_gemma":[0.388188,0.0004454523,0.6015686,0.0005997515,0.0001478663,0.000295826,0.0005809352,0.0006958921,0.007477689],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004790321,"threshold_uncertainty_score":0.01433587,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01687823670323815,"score_gpt":0.274094663465086,"score_spread":0.2572164267618479,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}