{"id":"W7104817253","doi":"","title":"A Metamorphic Testing Perspective on Knowledge Distillation for Language Models of Code: Does the Student Deeply Mimic the Teacher?","year":2025,"lang":"","type":"article","venue":"ArXiv.org","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Pipeline (software); Set (abstract data type); Perspective (graphical); Fidelity; Inference; Language model; Code (set theory); Software deployment","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004645227,0.001072964,0.000726974,0.0006764577,0.0004958502,0.001329244,0.002337268,0.001649921,0.001602317],"category_scores_gemma":[0.03442739,0.0004963211,0.0007183525,0.0003814349,0.003168484,0.005015707,0.003343398,0.003373299,0.0004398004],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001033664,"about_ca_system_score_gemma":0.002100277,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00277035,"about_ca_topic_score_gemma":0.003444603,"domain_scores_codex":[0.9963587,0.00167297,0.0001682079,0.0006422452,0.0008446064,0.0003133234],"domain_scores_gemma":[0.9827084,0.0111299,0.001164791,0.00374752,0.0008245946,0.0004248473],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001375264,0.0005089643,0.02645871,0.0006008615,0.0001985407,0.0005390363,0.0009394659,0.7415629,0.03327835,0.05702237,0.003260846,0.1342546],"study_design_scores_gemma":[0.00003181723,0.0003322781,0.0007325032,0.00003844455,0.00001680662,0.0001226539,0.00008788127,0.9646838,0.01310199,0.02001027,0.0008183608,0.00002318506],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3630555,0.0006815216,0.6227397,0.002166323,0.0001004221,0.0002003499,0.0005799251,0.00633409,0.004142124],"genre_scores_gemma":[0.9213823,0.0001487194,0.07642177,0.0003712936,0.00002618641,0.00009827574,0.0004598856,0.0002575501,0.0008340589],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004645227,"threshold_uncertainty_score":0.02456659,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05337935037745778,"score_gpt":0.3541483723111379,"score_spread":0.3007690219336802,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}