{"id":"W4388891009","doi":"10.48550/arxiv.2311.11177","title":"Assessing the Security of GitHub Copilot Generated Code -- A Targeted Replication Study","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Massey University; Canadian Institute for Advanced Research","keywords":"Computer science; Python (programming language); Code (set theory); Computer security; Replication (statistics); Software engineering; Programming language","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01512549,0.001152033,0.0007380758,0.003351183,0.0009617176,0.001623807,0.002737989,0.001709427,0.001011215],"category_scores_gemma":[0.1041771,0.000766426,0.001304869,0.001859951,0.002286639,0.003729985,0.002602973,0.003061306,0.0009165613],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002899295,"about_ca_system_score_gemma":0.00195931,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009582656,"about_ca_topic_score_gemma":0.006590889,"domain_scores_codex":[0.9854375,0.004789546,0.0009012631,0.002103005,0.006249968,0.0005187353],"domain_scores_gemma":[0.8499159,0.05538416,0.008945501,0.05046512,0.03365504,0.00163427],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.007179524,0.005157462,0.2486265,0.003703108,0.001588329,0.004459887,0.01886649,0.08569926,0.08826821,0.007746077,0.04115939,0.4875458],"study_design_scores_gemma":[0.001182709,0.01000978,0.2431041,0.0009296993,0.001470952,0.005437028,0.00537844,0.5214686,0.1332741,0.008300374,0.06877241,0.0006717807],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9709266,0.0006227695,0.01580803,0.0008239278,0.0001195347,0.0007623535,0.001145108,0.006417732,0.00337378],"genre_scores_gemma":[0.9600811,0.0003108678,0.02939111,0.0004719477,0.00004746811,0.0007731694,0.003836425,0.002426361,0.002661548],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9848745,"threshold_uncertainty_score":0.07999223,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1660223825657985,"score_gpt":0.2735873181719801,"score_spread":0.1075649356061816,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}