{"id":"W7105990745","doi":"10.2139/ssrn.5768480","title":"A Metamorphic Testing Perspective on Knowledge Distillation for Language Models of Code: Does the Student Deeply Mimic the Teacher?","year":2025,"lang":"","type":"preprint","venue":"SSRN Electronic Journal","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Pipeline (software); Perspective (graphical); Set (abstract data type); Fidelity; Inference; Language model; Code (set theory); Behavioral modeling","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006330064,0.0008026811,0.001095061,0.001014631,0.001014229,0.002779131,0.00275804,0.003992663,0.006695184],"category_scores_gemma":[0.04224548,0.0006211826,0.001042587,0.0005922122,0.008445974,0.009376246,0.005648244,0.006862111,0.0006828285],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001607391,"about_ca_system_score_gemma":0.001355298,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001288998,"about_ca_topic_score_gemma":0.0008834191,"domain_scores_codex":[0.9951847,0.002668727,0.0001595207,0.0007033698,0.0009255166,0.0003581347],"domain_scores_gemma":[0.9636037,0.0284727,0.001268552,0.00433551,0.001553199,0.0007664314],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001030876,0.00004615398,0.000797164,0.00006951964,0.00003071822,0.0001546666,0.0003309197,0.04767435,0.001102135,0.9305863,0.001417051,0.017688],"study_design_scores_gemma":[0.00001239218,0.00003443993,0.0000987317,0.00002351811,0.000005848747,0.00004241597,0.00004170819,0.2050063,0.001216937,0.7928204,0.0006844707,0.00001281187],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04731305,0.0002932625,0.925213,0.009459404,0.0001087412,0.00005488055,0.0001546468,0.0004777575,0.01692514],"genre_scores_gemma":[0.9180878,0.0001789081,0.07424197,0.0008503236,0.0001648561,0.00008619866,0.0001398059,0.0001872772,0.006062781],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006695184,"threshold_uncertainty_score":0.03347695,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02683330466154883,"score_gpt":0.3398161319375807,"score_spread":0.3129828272760319,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}