{"id":"W4392619070","doi":"10.1145/3643991.3644926","title":"Whodunit: Classifying Code as Human Authored or GPT-4 Generated - A case study on CodeChef problems","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Classifier (UML); Computer science; Machine learning; Artificial intelligence; Genetic programming; Stylometry; Natural language processing; Code (set theory); Source code; Receiver operating characteristic; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.001811427,0.000719318,0.0006607343,0.000971628,0.0003328838,0.002075874,0.002497119,0.0004832964,0.0001173989],"category_scores_gemma":[0.0007578805,0.0005717059,0.0001654076,0.001360324,0.00006412758,0.0001270094,0.007140184,0.003173608,0.0004626417],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005411343,"about_ca_system_score_gemma":0.0009302715,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001920688,"about_ca_topic_score_gemma":0.000781933,"domain_scores_codex":[0.9945539,0.0003921426,0.0007380847,0.002210766,0.001265624,0.0008395243],"domain_scores_gemma":[0.9954253,0.0008017984,0.0001299851,0.002900773,0.0003461239,0.0003960308],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"case_report","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001843903,0.008054409,0.005543041,0.006878875,0.004901005,0.479149,0.09465684,0.2308601,0.008399462,0.03900019,0.06860453,0.05376816],"study_design_scores_gemma":[0.002584123,0.004141533,0.001285749,0.002152727,0.0002145169,0.007323731,0.00248216,0.9570886,0.00460145,0.01053147,0.00356081,0.004033113],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8772435,0.0001768978,0.10893,0.0007127577,0.002108942,0.003400407,0.00002883833,0.005823319,0.001575329],"genre_scores_gemma":[0.9695528,0.000005641607,0.01401078,0.00008861389,0.0002450423,0.0008020517,0.00002134917,0.0001416002,0.01513217],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7262285,"threshold_uncertainty_score":0.9996734,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1743010673399469,"score_gpt":0.4017712111245808,"score_spread":0.2274701437846338,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}