{"id":"W2002559933","doi":"","title":"Shuffling and randomization for scalable source code clone detection","year":2016,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Saskatchewan; Concordia University","funders":"","keywords":"Shuffling; Computer science; Scalability; Code (set theory); Source code; clone (Java method); Machine learning; Theoretical computer science; Programming language; Database","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003554883,0.00004518989,0.0000624462,0.00006548902,0.00006526339,0.00006598608,0.000124822,0.00002879831,0.000004612031],"category_scores_gemma":[0.0007097634,0.00003037162,0.00001542707,0.000117286,0.00001268971,0.0002433217,0.00005989029,0.00002226961,0.000008941057],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002529991,"about_ca_system_score_gemma":0.000010777,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000008118342,"about_ca_topic_score_gemma":0.00000458823,"domain_scores_codex":[0.9994953,0.00001342565,0.00007490779,0.0001712625,0.00009948717,0.0001455624],"domain_scores_gemma":[0.9988267,0.000910349,0.00001331092,0.0001450011,0.00005955,0.00004511219],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001360839,0.00002349888,0.005013686,0.00005074224,0.00002182956,6.979062e-7,0.0001837044,0.001550427,0.1174908,0.003236492,0.0005150323,0.871777],"study_design_scores_gemma":[0.007956138,0.0001339334,0.003222256,0.00004634768,0.000004905346,0.00001613932,0.000005993983,0.7219185,0.257312,0.00163471,0.007507759,0.0002413127],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05423937,0.00003424476,0.9448543,0.000325819,0.00009465958,0.0001662983,3.534814e-7,0.0002603857,0.00002462113],"genre_scores_gemma":[0.9462852,0.0000120167,0.05205074,0.00002173058,0.00004610071,0.00003602198,1.725617e-7,0.000007432287,0.00154055],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8928035,"threshold_uncertainty_score":0.1238519,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01251566139029577,"score_gpt":0.2410320729598843,"score_spread":0.2285164115695885,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}