{"id":"W2474835145","doi":"10.1109/tse.2016.2584050","title":"An Empirical Comparison of Model Validation Techniques for Defect Prediction Models","year":2016,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":566,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada; Japan Society for the Promotion of Science; Japan Society for the Promotion of Science London; Compute Canada","keywords":"Computer science; Variance (accounting); Context (archaeology); Cross-validation; Model validation; Sample (material); Data mining; Predictive modelling; Software bug; Software; Machine learning","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.08804668,0.002012373,0.001253437,0.005456057,0.001089895,0.002407084,0.002122576,0.002590074,0.001157252],"category_scores_gemma":[0.268119,0.0006746958,0.003025672,0.004156694,0.001645671,0.004996179,0.002140429,0.003260061,0.0005979206],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002420923,"about_ca_system_score_gemma":0.001927863,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004775952,"about_ca_topic_score_gemma":0.006035721,"domain_scores_codex":[0.953718,0.03066177,0.002978476,0.004560301,0.007287162,0.0007942178],"domain_scores_gemma":[0.5111275,0.4312624,0.01220054,0.02586107,0.01846498,0.001083364],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002208937,0.000835041,0.2669771,0.002063969,0.004484635,0.0002960939,0.001541923,0.4586297,0.002531802,0.01549599,0.01100171,0.2339331],"study_design_scores_gemma":[0.0002010613,0.001724271,0.08326709,0.0008588482,0.0006721755,0.0006702605,0.0006492389,0.8868243,0.003824642,0.01547331,0.00563853,0.0001963037],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6999274,0.0111486,0.2750353,0.001498331,0.0003550572,0.0005150843,0.003384329,0.002021749,0.00611417],"genre_scores_gemma":[0.9277461,0.001358534,0.06456732,0.000200378,0.00007461249,0.0003637347,0.004694594,0.0003723615,0.000622304],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9119533,"threshold_uncertainty_score":0.465641,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05422237551678902,"score_gpt":0.3303165094466313,"score_spread":0.2760941339298422,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}