{"id":"W4407771189","doi":"10.1145/3641554.3701917","title":"Quantitative Evaluation of Using Large Language Models and Retrieval-Augmented Generation in Computer Science Education","year":2025,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia, Okanagan Campus; University of British Columbia","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Language model; Information retrieval","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01559865,0.0009113902,0.0005911089,0.001162492,0.0005066898,0.001707999,0.001251606,0.001266052,0.003830833],"category_scores_gemma":[0.09012094,0.0003317027,0.0005103053,0.0009190019,0.0008949029,0.002719821,0.00148307,0.001161206,0.00113167],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001083446,"about_ca_system_score_gemma":0.000790845,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002319461,"about_ca_topic_score_gemma":0.003084522,"domain_scores_codex":[0.9839252,0.0112812,0.0009940963,0.001190351,0.002314067,0.0002951357],"domain_scores_gemma":[0.8346133,0.141843,0.004818716,0.008911294,0.008050614,0.001763138],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.01779877,0.01432178,0.04594129,0.00605343,0.0006258658,0.0002674887,0.005200346,0.07959148,0.05735853,0.003654945,0.01295132,0.7562348],"study_design_scores_gemma":[0.003752138,0.0519982,0.1476642,0.0006777379,0.001482088,0.0008096042,0.00481001,0.6078638,0.1349733,0.008682651,0.03675199,0.000534256],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9483604,0.001299449,0.03595302,0.0005921673,0.0001446935,0.001225788,0.001405198,0.003284556,0.007734646],"genre_scores_gemma":[0.9456531,0.0003358682,0.04785356,0.0001533555,0.00007206915,0.0007201253,0.001981724,0.000356755,0.00287351],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01559865,"threshold_uncertainty_score":0.08249456,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09692513595673274,"score_gpt":0.3935415129644818,"score_spread":0.2966163770077491,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}