{"id":"W4386982649","doi":"10.1007/s10664-023-10380-1","title":"Is GitHub’s Copilot as bad as humans at introducing vulnerabilities in code?","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":101,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Vulnerability (computing); Code (set theory); Process (computing); Computer security; Perspective (graphical); Secure coding; Software; Software engineering; Software security assurance; Artificial intelligence; Operating system; Information security; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01284018,0.0007556767,0.0005805771,0.002195365,0.001953775,0.004201843,0.00148113,0.003567981,0.007533938],"category_scores_gemma":[0.1135264,0.0005736612,0.0004657946,0.001622883,0.005434384,0.00982682,0.003005204,0.003289641,0.003115741],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001564931,"about_ca_system_score_gemma":0.004348669,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01272423,"about_ca_topic_score_gemma":0.01876717,"domain_scores_codex":[0.9867927,0.005636173,0.0003781918,0.001327821,0.004550868,0.001314382],"domain_scores_gemma":[0.9246243,0.03559589,0.008131279,0.01611131,0.01120339,0.004333802],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.001873805,0.0006240356,0.2007792,0.0009830421,0.0005266014,0.00113212,0.0146606,0.004594635,0.007666979,0.07533218,0.2782953,0.4135316],"study_design_scores_gemma":[0.0004531831,0.00168336,0.1989031,0.002518631,0.0007303943,0.005750274,0.03573762,0.03147067,0.02586375,0.2132635,0.4830087,0.0006168806],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6374408,0.004350051,0.0443414,0.1714395,0.002593075,0.0001521607,0.001369019,0.01016487,0.1281492],"genre_scores_gemma":[0.9504278,0.001140921,0.01825848,0.01746825,0.000322523,0.00005797809,0.0008132141,0.00237447,0.009136404],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01284018,"threshold_uncertainty_score":0.0679062,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03474644801521013,"score_gpt":0.3154596742478399,"score_spread":0.2807132262326297,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}