{"id":"W2588556346","doi":"10.1142/s0218194016400106","title":"A Machine Learning Based Approach for Evaluating Clone Detection Tools for a Generalized and Accurate Precision","year":2016,"lang":"en","type":"article","venue":"International Journal of Software Engineering and Knowledge Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Generality; clone (Java method); Computer science; Measure (data warehouse); Machine learning; Software; Variety (cybernetics); Data mining; Artificial intelligence; Java; Sample (material); Detector; Algorithm; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01296687,0.002089917,0.002273515,0.01169709,0.001338327,0.003876068,0.002920873,0.003183621,0.001447775],"category_scores_gemma":[0.07361379,0.0006050028,0.001799672,0.007544956,0.001613226,0.004794132,0.002130026,0.002426617,0.0009980486],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002496856,"about_ca_system_score_gemma":0.001888037,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003009715,"about_ca_topic_score_gemma":0.003718981,"domain_scores_codex":[0.9740776,0.004728216,0.002878599,0.004929974,0.01257088,0.0008148107],"domain_scores_gemma":[0.9291034,0.03541596,0.01008667,0.009851192,0.01485709,0.0006857456],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007225028,0.0006419948,0.07109307,0.0007397162,0.00085052,0.0002378539,0.0006499783,0.1627843,0.03171868,0.01279776,0.003854743,0.7139089],"study_design_scores_gemma":[0.00006629855,0.0006987477,0.0221846,0.0001244641,0.0001887311,0.0004795603,0.0001764749,0.9261977,0.02894819,0.01710672,0.003673126,0.000155443],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07149266,0.0005773662,0.9190514,0.0002371869,0.00007359544,0.0004838298,0.0005813406,0.003961427,0.003541091],"genre_scores_gemma":[0.439318,0.0001353912,0.5574523,0.0001696393,0.00006594992,0.0005635606,0.000768599,0.0002213757,0.001305171],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9870331,"threshold_uncertainty_score":0.06857622,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03570845271944944,"score_gpt":0.3056461376792703,"score_spread":0.2699376849598209,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}