{"id":"W4391157645","doi":"10.48550/arxiv.2401.11314","title":"CodeAid: Evaluating a Classroom Deployment of an LLM-based Programming Assistant that Balances Student and Educator Needs","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Online Learning and Analytics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Code (set theory); Software deployment; Transparency (behavior); Class (philosophy); Thematic analysis; Mathematics education; Software engineering; Multimedia; Programming language; Psychology; Qualitative research; Artificial intelligence; Sociology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004610428,0.000828482,0.0004356786,0.001010898,0.0008558656,0.001956409,0.002437427,0.001495297,0.003247432],"category_scores_gemma":[0.02982689,0.0003936977,0.0002897904,0.0006613775,0.001033642,0.002729357,0.002917357,0.001270022,0.001668272],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001323119,"about_ca_system_score_gemma":0.002316489,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00354281,"about_ca_topic_score_gemma":0.005222548,"domain_scores_codex":[0.9952343,0.00206529,0.0003791729,0.0008215138,0.001093438,0.0004063598],"domain_scores_gemma":[0.9782442,0.0126358,0.0009976599,0.002049354,0.003697935,0.002374982],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.006111981,0.01305227,0.03243824,0.004917601,0.0001739492,0.001700197,0.04089455,0.01072473,0.1551021,0.004514134,0.03139079,0.6989795],"study_design_scores_gemma":[0.005599739,0.05783159,0.09199281,0.001545134,0.0006649588,0.003295275,0.04166706,0.1818788,0.2068534,0.00969635,0.3980181,0.00095667],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9258005,0.000347069,0.05098105,0.0005931601,0.0001203271,0.002137389,0.0008016754,0.01004736,0.009171562],"genre_scores_gemma":[0.8452612,0.0003663214,0.1400018,0.0005909572,0.0000636514,0.001643331,0.001988052,0.001023749,0.009060893],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004610428,"threshold_uncertainty_score":0.02438259,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0855630401474765,"score_gpt":0.2724213960874086,"score_spread":0.1868583559399321,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}