{"id":"W4388954848","doi":"10.1145/3605770.3625215","title":"Distinguishing AI- and Human-Generated Code: A Case Study","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Code (set theory); Task (project management); Code review; Software; Source code; Artificial intelligence; Programming language; Natural language processing; Software engineering; Computer security; Static program analysis; Software development; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006174751,0.000622768,0.0003991598,0.001760637,0.001357224,0.001510742,0.001271492,0.002975075,0.001305545],"category_scores_gemma":[0.0497611,0.0002656415,0.0005160863,0.001576066,0.002133635,0.002432481,0.001833271,0.001513023,0.0008674667],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009020792,"about_ca_system_score_gemma":0.0009540747,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004277987,"about_ca_topic_score_gemma":0.008619157,"domain_scores_codex":[0.9905409,0.004856694,0.001122061,0.001239708,0.001932226,0.0003084159],"domain_scores_gemma":[0.8948419,0.08384417,0.003702148,0.009290084,0.007116524,0.001205241],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.002450646,0.007678123,0.2309566,0.003733139,0.0003673668,0.03334947,0.07114043,0.02552837,0.07237092,0.01321306,0.02391959,0.5152922],"study_design_scores_gemma":[0.001238548,0.006256421,0.2007444,0.001380295,0.000368369,0.06046988,0.04533032,0.2587199,0.2481261,0.03182846,0.1449631,0.0005742897],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9668565,0.0004188417,0.0263772,0.0007710143,0.00006035368,0.0005029826,0.0006180994,0.000796802,0.003598048],"genre_scores_gemma":[0.9491886,0.0002145604,0.04733972,0.0003299984,0.00003070938,0.0002031328,0.000978566,0.0001583467,0.001556386],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006174751,"threshold_uncertainty_score":0.03265566,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06023643670637761,"score_gpt":0.3549796554843336,"score_spread":0.294743218777956,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}