{"id":"W1993673250","doi":"10.1109/saner.2015.7081830","title":"Threshold-free code clone detection for a large-scale heterogeneous Java repository","year":2015,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"clone (Java method); Java; Granularity; Computer science; Benchmark (surveying); Source code; Code (set theory); Data mining; Software; Variety (cybernetics); Algorithm; Artificial intelligence; Operating system; Programming language; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004643672,0.0001114048,0.0001201309,0.00009853308,0.00009682355,0.0001433469,0.0009291091,0.00007883586,0.000002450738],"category_scores_gemma":[0.0003476002,0.0001040887,0.00006645962,0.0002274659,0.00001529882,0.0002289605,0.0003862088,0.0001096766,0.00003243189],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001165664,"about_ca_system_score_gemma":0.00006786901,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002692645,"about_ca_topic_score_gemma":0.00006742448,"domain_scores_codex":[0.998658,0.0000215012,0.0001669719,0.0003769758,0.0003850227,0.0003915251],"domain_scores_gemma":[0.9983765,0.0001856896,0.00002827132,0.001001883,0.0001960176,0.0002116555],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001253354,0.003798965,0.08001351,0.001089376,0.001001703,0.00146637,0.01992279,0.1182331,0.2398221,0.03038558,0.3738237,0.1291895],"study_design_scores_gemma":[0.001960573,0.0006373555,0.001283036,0.00001545028,0.000007807896,0.0003156688,0.00003716139,0.7201906,0.255293,0.002618067,0.01722929,0.0004120596],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.08170114,0.0001631467,0.9157912,0.0001650256,0.0007687242,0.0002654745,0.000003459408,0.0007213673,0.0004205095],"genre_scores_gemma":[0.9200111,0.000001771522,0.0778994,0.00007868227,0.0001919572,0.0001055565,0.000001385424,0.00002177544,0.001688355],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.83831,"threshold_uncertainty_score":0.4244612,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02652208983493504,"score_gpt":0.2673020073773731,"score_spread":0.240779917542438,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}