{"id":"W2806282111","doi":"10.1002/stvr.1669","title":"MuMonDE: A framework for evaluating model clone detectors using model mutation analysis","year":2018,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Engineering Research","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"clone (Java method); Preprocessor; Mutation; Computer science; Mutation testing; Data mining; Detector; Precision and recall; Software engineering; Artificial intelligence; Genetics; Biology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01781196,0.002256056,0.00143349,0.009331075,0.000896103,0.004315079,0.003838578,0.001443588,0.004084579],"category_scores_gemma":[0.05603437,0.001102012,0.002538988,0.002023933,0.002044979,0.003669194,0.00386271,0.001912684,0.0007796316],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0027105,"about_ca_system_score_gemma":0.002975256,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008010807,"about_ca_topic_score_gemma":0.008931389,"domain_scores_codex":[0.9871268,0.004957865,0.001480051,0.001285557,0.004701756,0.0004480025],"domain_scores_gemma":[0.9662088,0.02012628,0.003339469,0.004341397,0.005399919,0.0005841756],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008035997,0.0009598485,0.02600587,0.001138558,0.0006939824,0.0005026818,0.001260492,0.3942782,0.02520125,0.1241067,0.01302713,0.4120216],"study_design_scores_gemma":[0.00007007772,0.000203224,0.00149971,0.0001375443,0.00006534411,0.0001311487,0.0001117828,0.9632713,0.0111546,0.01731731,0.005958166,0.00007977167],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0122728,0.00009993945,0.9698473,0.0001430051,0.0000215041,0.0003959182,0.0004203571,0.01528071,0.001518459],"genre_scores_gemma":[0.1132699,0.00006171018,0.8836408,0.00007548842,0.00001466486,0.0005418735,0.0009497994,0.000906994,0.0005387362],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.982188,"threshold_uncertainty_score":0.09419984,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1031035430604409,"score_gpt":0.3755641998763948,"score_spread":0.2724606568159539,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}