{"id":"W4417003938","doi":"10.1109/uemcon67449.2025.11267548","title":"Evaluating Source Code Embeddings From LLMS for Educational Downstream Tasks","year":2025,"lang":"","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Guelph","funders":"","keywords":"Source code; Automatic summarization; Code review; Natural language; Code (set theory); Embedding; Downstream (manufacturing); Paragraph","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001884354,0.001900082,0.0004415628,0.001374941,0.0003301833,0.001012712,0.001067588,0.00155379,0.001862345],"category_scores_gemma":[0.0171317,0.0003603375,0.000774427,0.0009126029,0.0004944086,0.002349605,0.001487969,0.001851878,0.001524642],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000888531,"about_ca_system_score_gemma":0.0009276611,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003806703,"about_ca_topic_score_gemma":0.00590794,"domain_scores_codex":[0.9986035,0.0005248282,0.0001299873,0.0004069065,0.000237298,0.00009754297],"domain_scores_gemma":[0.9935134,0.004049148,0.0003446882,0.0008429268,0.0009907754,0.0002591481],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001397145,0.001074055,0.02403157,0.0008684542,0.0003153604,0.0003031138,0.0004189294,0.3675369,0.02131992,0.002372186,0.01520488,0.5651574],"study_design_scores_gemma":[0.00005303268,0.000321206,0.002176187,0.00004735416,0.00004645947,0.00006999259,0.0001578571,0.975802,0.01681112,0.00271703,0.001779143,0.00001862727],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7775912,0.001850742,0.1915765,0.001036059,0.0003630959,0.0003284751,0.003665319,0.01906209,0.004526538],"genre_scores_gemma":[0.8640132,0.0003899239,0.1162978,0.0002389246,0.00005987252,0.0002413751,0.01520469,0.0005872717,0.002966972],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.003806703,"threshold_uncertainty_score":0.009965539,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04525139358938427,"score_gpt":0.3912953139093164,"score_spread":0.3460439203199321,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}