{"id":"W4413199870","doi":"10.3102/ip.25.2197254","title":"Evaluating the Efficacy of Generative AI in Automating Assessment in Introductory Computer Science Courses (Poster 47)","year":2025,"lang":"en","type":"article","venue":"","topic":"Advanced Data Processing Techniques","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Computer science; Generative grammar; Software engineering; Artificial intelligence; Human–computer interaction","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0123438,0.0007654364,0.0005578846,0.0008253684,0.0004925545,0.001898016,0.001779577,0.001589853,0.003842245],"category_scores_gemma":[0.08599351,0.0005043035,0.0005378826,0.0004705777,0.0006612152,0.001495948,0.0014905,0.001695663,0.001248686],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006702284,"about_ca_system_score_gemma":0.001480871,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001953795,"about_ca_topic_score_gemma":0.002020655,"domain_scores_codex":[0.9946912,0.002957382,0.0004245029,0.0007607743,0.0009019125,0.0002642843],"domain_scores_gemma":[0.8637915,0.1185333,0.003912552,0.005126994,0.005017604,0.003618064],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.03601095,0.05202742,0.1061176,0.001205522,0.0005374998,0.0001688903,0.007918484,0.03755612,0.02757616,0.002006652,0.003529707,0.725345],"study_design_scores_gemma":[0.01180923,0.2020388,0.4051654,0.001038058,0.002179947,0.0005483855,0.005132879,0.2748956,0.07555481,0.00931961,0.01173093,0.000586312],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9869152,0.00009029499,0.00812739,0.0002240948,0.00008916101,0.0005330243,0.00006174966,0.0004819231,0.003477237],"genre_scores_gemma":[0.9783494,0.0001115788,0.01866898,0.0001732176,0.00003935709,0.0004803607,0.0001528445,0.00005090743,0.001973311],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0123438,"threshold_uncertainty_score":0.06528097,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02715858326500707,"score_gpt":0.4017002551157903,"score_spread":0.3745416718507832,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}