{"id":"W4404952914","doi":"10.1109/issre62328.2024.00030","title":"Assessing the Performance of AI-Generated Code: A Case Study on GitHub Copilot","year":2024,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"National Natural Science Foundation of China","keywords":"Computer science; Code (set theory); Computer security; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005974474,0.00008968267,0.0000902247,0.0001194124,0.00009450709,0.0005032738,0.0004884657,0.00002002559,0.00001350446],"category_scores_gemma":[0.0000904522,0.00005397856,0.00002272693,0.0008574365,0.00002862551,0.0003792534,0.0001802773,0.0002536525,0.00004048515],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000038583,"about_ca_system_score_gemma":0.00013643,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007964737,"about_ca_topic_score_gemma":0.000008116124,"domain_scores_codex":[0.9989929,0.0000624241,0.0001455946,0.0002502532,0.0003649099,0.0001839335],"domain_scores_gemma":[0.9986987,0.0006516377,0.00001150587,0.0005133249,0.00008374357,0.00004109548],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004593838,0.003488949,0.3859099,0.00145663,0.001151723,0.03454944,0.04320902,0.1290493,0.02819008,0.03010648,0.0393776,0.3034649],"study_design_scores_gemma":[0.0001742642,0.0006146839,0.01305175,0.00007074774,0.00000590059,0.0005403396,0.0003581361,0.9729887,0.01175222,0.000009050586,0.0003032261,0.0001309884],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9335716,0.0000730076,0.06503362,0.0003309477,0.0003137614,0.0001934012,6.575567e-7,0.0003463028,0.0001366434],"genre_scores_gemma":[0.9981974,0.000001463103,0.001355425,0.00006239014,0.0000368554,0.00001882304,2.241587e-7,0.00001037322,0.0003170181],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8439394,"threshold_uncertainty_score":0.4853081,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06393546958244994,"score_gpt":0.3681672486282733,"score_spread":0.3042317790458234,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}