{"id":"W3013132080","doi":"10.1109/iwsc50091.2020.9047643","title":"SemanticCloneBench: A Semantic Code Clone Benchmark using Crowd-Source Knowledge","year":2020,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Computer science; Java; Benchmark (surveying); Programming language; clone (Java method); Python (programming language); Source code; Code (set theory)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003420561,0.0002429975,0.0002905156,0.0001596761,0.0001545436,0.0002868187,0.001426328,0.00009192903,0.0001328991],"category_scores_gemma":[0.0007195817,0.0002353413,0.0001040755,0.001393228,0.00006157804,0.0004248614,0.0009541382,0.0003305417,0.0004754532],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007368592,"about_ca_system_score_gemma":0.000164341,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007085676,"about_ca_topic_score_gemma":0.000008160853,"domain_scores_codex":[0.9977987,0.00007755166,0.0003066377,0.0006834859,0.0004852548,0.0006483347],"domain_scores_gemma":[0.9980915,0.000597753,0.0000466263,0.0007001811,0.0001409251,0.0004229913],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001489826,0.002102652,0.1714595,0.004300596,0.001166652,0.002328685,0.0594382,0.1053964,0.3126212,0.09686638,0.1312711,0.1128997],"study_design_scores_gemma":[0.0003907306,0.0000967851,0.003527264,0.00005326424,0.00001062706,0.00005864066,0.00003541992,0.9798949,0.006662132,0.0001083293,0.008743131,0.0004188189],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1557723,0.0003498218,0.8400677,0.001453836,0.0002460647,0.0001997309,0.000001350944,0.0009515278,0.0009575679],"genre_scores_gemma":[0.9290936,0.000009199915,0.06965571,0.0003779485,0.0002125756,0.000007291931,0.000001770417,0.00003495441,0.0006069532],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8744985,"threshold_uncertainty_score":0.959694,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04975131297904765,"score_gpt":0.2908692215705029,"score_spread":0.2411179085914553,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}