{"id":"W4392564834","doi":"10.1145/3626252.3630789","title":"Beyond Traditional Teaching: Large Language Models as Simulated Teaching Assistants in Computer Science","year":2024,"lang":"en","type":"article","venue":"","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":49,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"CLARITY; Python (programming language); Computer science; Implementation; Matching (statistics); Mathematics education; Software engineering; Human–computer interaction; Programming language; Psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008412683,0.0004813773,0.0003324458,0.0004235146,0.0006720946,0.004079126,0.00241555,0.001016083,0.005043267],"category_scores_gemma":[0.03247865,0.0003157881,0.0004337589,0.0003840324,0.001725664,0.004968929,0.003611563,0.001233951,0.001822806],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001060056,"about_ca_system_score_gemma":0.002016838,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008240392,"about_ca_topic_score_gemma":0.001561311,"domain_scores_codex":[0.9922087,0.006313168,0.0001430629,0.0005453983,0.0005142989,0.0002752749],"domain_scores_gemma":[0.9749224,0.01586639,0.001429106,0.004049163,0.001201575,0.002531233],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002319919,0.005328478,0.04306851,0.001757256,0.0001822618,0.0007351998,0.05482448,0.05284995,0.03597989,0.09921216,0.01577712,0.6879648],"study_design_scores_gemma":[0.001008424,0.008223464,0.01346883,0.001245919,0.0003646242,0.001598725,0.02308545,0.4215219,0.05220861,0.1502305,0.3266565,0.0003871886],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4841157,0.0005233949,0.4771724,0.00344185,0.0002367947,0.000521152,0.0001183447,0.00394017,0.02993029],"genre_scores_gemma":[0.8139611,0.0001851935,0.1793622,0.0003796542,0.00003815722,0.0003173337,0.0001260608,0.0002235844,0.005406803],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008412683,"threshold_uncertainty_score":0.04449111,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1310483018404336,"score_gpt":0.4375658554510142,"score_spread":0.3065175536105805,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}