{"id":"W4389544307","doi":"10.1109/icsme58846.2023.00013","title":"GPTCloneBench: A comprehensive benchmark of semantic clones and cross-language clones using GPT-3 model and SemanticCloneBench","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Saskatchewan","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Benchmark (surveying); Computer science; Natural language processing; Artificial intelligence; Programming language; Computational biology; Biology; Geography; Cartography","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0004101116,0.0002850475,0.0004360687,0.000520441,0.0001885531,0.0002508788,0.0005516195,0.0001260839,0.00001129389],"category_scores_gemma":[0.0002767945,0.0002666988,0.0000657869,0.001164661,0.000277789,0.0005550723,0.001131256,0.0002164528,0.0000111469],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002610556,"about_ca_system_score_gemma":0.0000886161,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004659527,"about_ca_topic_score_gemma":0.00002158487,"domain_scores_codex":[0.9976663,0.00005734072,0.0004046355,0.0006853232,0.0005782358,0.0006081394],"domain_scores_gemma":[0.9978207,0.000971465,0.00008874603,0.0006931478,0.0002230154,0.0002028954],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00008827094,0.0003184432,0.3674258,0.004158797,0.0004820988,0.000951072,0.02656665,0.180978,0.363228,0.03968765,0.001314723,0.01480043],"study_design_scores_gemma":[0.0004407201,0.00006059982,0.1030424,0.00008785193,0.00001236519,0.0001042866,0.0002001986,0.8911213,0.003572116,0.001042575,0.0000117609,0.0003038995],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9133232,0.0008621919,0.08477244,0.0001459318,0.0001463489,0.0002674807,0.000009070689,0.0004101681,0.00006314692],"genre_scores_gemma":[0.9618868,0.0001803918,0.03754676,0.0000406921,0.00003491163,0.00001025857,0.000005272306,0.0000287505,0.0002661185],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7101433,"threshold_uncertainty_score":0.9999785,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03741457056263726,"score_gpt":0.3245508019763634,"score_spread":0.2871362314137262,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}