{"id":"W2067490448","doi":"","title":"The Impact of Mislabelling on the Performance and Interpretation of Defect Prediction Models","year":2018,"lang":"en","type":"article","venue":"Institutional Repositories DataBase (IRDB)","topic":"Software Engineering Research","field":"Computer Science","cited_by":89,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Noise (video); Reliability (semiconductor); Interpretation (philosophy); Artificial intelligence; Predictive modelling; Machine learning; Rank (graph theory); Training set; Data modeling; Recall; Data mining; Database; Mathematics; Cognitive psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06652968,0.002952415,0.002886446,0.003697442,0.002576045,0.008095733,0.003360508,0.005262353,0.001439214],"category_scores_gemma":[0.2701415,0.0009846757,0.002016634,0.0025833,0.002218496,0.006279184,0.003598867,0.005343459,0.001448155],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002906199,"about_ca_system_score_gemma":0.003003677,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01524298,"about_ca_topic_score_gemma":0.01657603,"domain_scores_codex":[0.9339353,0.03999642,0.004854595,0.01209227,0.007766059,0.001355263],"domain_scores_gemma":[0.4727199,0.4778389,0.0108071,0.02058521,0.01514712,0.002901655],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01047327,0.001947974,0.286235,0.001532693,0.003265557,0.001182559,0.002326571,0.1633116,0.01224456,0.003339136,0.03528275,0.4788583],"study_design_scores_gemma":[0.0003763012,0.00123287,0.0398522,0.0005927336,0.001456874,0.001306433,0.001508045,0.915522,0.01707825,0.01562442,0.005165423,0.0002843984],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8833206,0.01392232,0.07687397,0.006097048,0.002082954,0.0002374274,0.004001985,0.007510995,0.005952661],"genre_scores_gemma":[0.9622378,0.0006933197,0.02882803,0.001177049,0.0002764061,0.00004801349,0.004494181,0.0008810057,0.001364126],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.06652968,"threshold_uncertainty_score":0.3518468,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01975086518738715,"score_gpt":0.2721726546332785,"score_spread":0.2524217894458913,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}