{"id":"W7105990745","doi":"10.2139/ssrn.5768480","title":"A Metamorphic Testing Perspective on Knowledge Distillation for Language Models of Code: Does the Student Deeply Mimic the Teacher?","year":2025,"lang":"","type":"preprint","venue":"SSRN Electronic Journal","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Pipeline (software); Perspective (graphical); Set (abstract data type); Fidelity; Inference; Language model; Code (set theory); Behavioral modeling","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.01231069,0.000833242,0.0009515522,0.0003766604,0.002000648,0.0005429074,0.005099148,0.0003264448,0.000008989111],"category_scores_gemma":[0.002259821,0.0004573291,0.0008119317,0.0009362889,0.0003777297,0.000376983,0.001864827,0.01097567,0.000003979816],"about_ca_system_candidate":true,"about_ca_system_consensus":true,"about_ca_system_score_codex":0.005375271,"about_ca_system_score_gemma":0.008926666,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004879673,"about_ca_topic_score_gemma":0.001531082,"domain_scores_codex":[0.9914923,0.00212747,0.001318494,0.001230968,0.00106046,0.002770314],"domain_scores_gemma":[0.990817,0.003942881,0.002243694,0.001482428,0.001411778,0.0001022398],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001475698,0.0002245559,0.0002434437,0.0000621579,0.001100939,0.00000181523,0.02813646,0.4849374,0.0000474125,0.4445185,0.000008106438,0.0405717],"study_design_scores_gemma":[0.0008526559,0.0004462167,0.0004074854,0.000438872,0.000609622,0.00007792216,0.02683255,0.6418194,0.00003634407,0.3280508,0.00004526876,0.0003828691],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03495308,0.01060803,0.9431827,0.004997058,0.001729256,0.002083815,0.0000269522,0.00007432354,0.002344771],"genre_scores_gemma":[0.9916548,0.0008549889,0.003549804,0.00007213796,0.001260268,0.0001246454,0.000004334589,0.0000611517,0.002417921],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9567017,"threshold_uncertainty_score":0.9997879,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02683330466154883,"score_gpt":0.3398161319375807,"score_spread":0.3129828272760319,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}