{"id":"W4386301778","doi":"10.48550/arxiv.2308.13963","title":"GPTCloneBench: A comprehensive benchmark of semantic clones and cross-language clones using GPT-3 model and SemanticCloneBench","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Benchmark (surveying); clone (Java method); Programming language; Java; Artificial intelligence; Natural language processing; Python (programming language); Machine learning; Software engineering","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00364402,0.003074097,0.000949906,0.00432454,0.001160925,0.002177046,0.005659645,0.002451668,0.002276155],"category_scores_gemma":[0.0186548,0.0007848985,0.002268109,0.0052096,0.001618627,0.004087381,0.003325339,0.002592962,0.001937693],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002340877,"about_ca_system_score_gemma":0.003133784,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01640482,"about_ca_topic_score_gemma":0.02036786,"domain_scores_codex":[0.9933234,0.001169949,0.0006390575,0.001881023,0.002499145,0.0004873993],"domain_scores_gemma":[0.9895978,0.003797197,0.000770203,0.002711115,0.002522056,0.0006015686],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002394938,0.002204951,0.0892474,0.005459394,0.001077665,0.002873801,0.001511937,0.2230483,0.03114247,0.01468667,0.3421776,0.2841749],"study_design_scores_gemma":[0.0006949446,0.001717236,0.03341089,0.0004433979,0.0003536023,0.001912918,0.001084627,0.7470306,0.05349272,0.01506149,0.1445486,0.0002489562],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5991944,0.005464626,0.1037576,0.002079162,0.0009680794,0.001389243,0.1239571,0.1363475,0.02684226],"genre_scores_gemma":[0.3701894,0.001221641,0.1526986,0.001165208,0.00008563371,0.001292396,0.4586792,0.008935113,0.005732854],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01640482,"threshold_uncertainty_score":0.03261864,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1163536748243747,"score_gpt":0.2545266693069382,"score_spread":0.1381729944825635,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}