{"id":"W4321214126","doi":"10.1186/s13040-023-00322-4","title":"The Matthews correlation coefficient (MCC) should replace the ROC AUC as the standard metric for assessing binary classification","year":2023,"lang":"en","type":"article","venue":"BioData Mining","topic":"Imbalanced Data Classification Techniques","field":"Computer Science","cited_by":513,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Receiver operating characteristic; Matthews correlation coefficient; Binary classification; Statistics; Confusion matrix; False positive rate; Artificial intelligence; Correlation; Mathematics; Binary number; Sensitivity (control systems); Metric (unit); Classifier (UML); Computer science; Pattern recognition (psychology); Machine learning; Support vector machine; Arithmetic","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01530556,0.002756914,0.004081455,0.003941244,0.0009771422,0.004764434,0.003884683,0.005842996,0.006873861],"category_scores_gemma":[0.09827743,0.0008002041,0.001730083,0.006211227,0.003612601,0.004704223,0.001725498,0.005976944,0.01251109],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002863693,"about_ca_system_score_gemma":0.003009216,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003513164,"about_ca_topic_score_gemma":0.003343298,"domain_scores_codex":[0.981873,0.005489669,0.002355809,0.002591823,0.007197795,0.0004919061],"domain_scores_gemma":[0.9382277,0.0345214,0.007168982,0.005280702,0.01335584,0.001445459],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0005443954,0.0001262674,0.01733394,0.003481501,0.0008541755,0.0007132969,0.0002993405,0.004302681,0.004502175,0.05068823,0.5676448,0.3495092],"study_design_scores_gemma":[0.0002152485,0.001137028,0.04118559,0.003305807,0.0009035009,0.00621452,0.0005875552,0.04324102,0.02648873,0.1804464,0.6952102,0.001064406],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02861718,0.1832888,0.5222214,0.1014407,0.07616145,0.00126635,0.01073733,0.01671157,0.05955528],"genre_scores_gemma":[0.3390985,0.06127909,0.4569612,0.05931128,0.02679771,0.003372934,0.01172351,0.005896053,0.03555974],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9846944,"threshold_uncertainty_score":0.08094454,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1099873591318314,"score_gpt":0.3627967148767833,"score_spread":0.2528093557449519,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}