{"id":"W2086095861","doi":"10.1109/tr.2012.2183912","title":"Evaluating Stratification Alternatives to Improve Software Defect Prediction","year":2012,"lang":"en","type":"article","venue":"IEEE Transactions on Reliability","topic":"Software Engineering Research","field":"Computer Science","cited_by":59,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Machine learning; Software bug; Skewness; Software; Artificial intelligence; Software quality; Data mining; Sampling (signal processing); Stratification (seeds); Reliability engineering; Statistics; Software development; Engineering; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03492776,0.001384261,0.002125474,0.00215656,0.0007228075,0.001590625,0.0008636687,0.001205149,0.001107036],"category_scores_gemma":[0.09440004,0.0004917706,0.001288429,0.001178168,0.00140464,0.002766914,0.001825143,0.001347379,0.0002610922],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001338074,"about_ca_system_score_gemma":0.001505007,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001544836,"about_ca_topic_score_gemma":0.001500644,"domain_scores_codex":[0.9822818,0.01368908,0.0007133803,0.001223682,0.001610264,0.0004817585],"domain_scores_gemma":[0.8757462,0.1018607,0.006656523,0.007318235,0.006984371,0.001433993],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.007174446,0.002195675,0.17445,0.0004149582,0.001039853,0.000180394,0.001148616,0.3028502,0.01364364,0.0183712,0.002677653,0.4758534],"study_design_scores_gemma":[0.0003586625,0.003414468,0.03082408,0.00008936522,0.0003485269,0.00005528203,0.0002662792,0.9347515,0.007213984,0.0213795,0.001196736,0.0001016618],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6962646,0.001310257,0.2984519,0.0005228806,0.0001306687,0.0005391834,0.0002251817,0.0009155624,0.001639851],"genre_scores_gemma":[0.9313976,0.0001465794,0.06758367,0.0001085466,0.00004807036,0.0001871417,0.0003197802,0.0000294786,0.0001790617],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03492776,"threshold_uncertainty_score":0.1847178,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04061733545592563,"score_gpt":0.3384401637331717,"score_spread":0.2978228282772461,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}