{"id":"W6977455728","doi":"10.6084/m9.figshare.c.6581771.v1","title":"The Matthews correlation coefficient (MCC) should replace the ROC AUC as the standard metric for assessing binary classification","year":2023,"lang":"en","type":"other","venue":"Figshare","topic":"Health, Education, and Cultural Studies","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Receiver operating characteristic; Binary classification; Matthews correlation coefficient; Binary number; Metric (unit); Confusion matrix; Correlation; Sensitivity (control systems); False positive rate","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["sts","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.001245448,0.0001874684,0.0001840223,0.00006423423,0.0053386,0.0005130486,0.0005556936,0.0002702277,0.009831108],"category_scores_gemma":[0.005193892,0.00008874614,0.0001212073,0.0009125288,0.00008805683,0.00009698881,0.00003868451,0.0003076288,0.000908786],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003947401,"about_ca_system_score_gemma":0.0009997619,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001044768,"about_ca_topic_score_gemma":0.01389135,"domain_scores_codex":[0.9979022,0.000448276,0.0002749247,0.0003216879,0.0006778802,0.0003749973],"domain_scores_gemma":[0.996209,0.002526364,0.0005096927,0.0003895199,0.0003021664,0.00006320435],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000005292811,0.000009035908,0.00001653552,0.00004144126,0.00002473441,1.200026e-7,0.005781026,0.00001253156,1.678013e-7,0.001239683,0.9845346,0.008334812],"study_design_scores_gemma":[0.00005899609,0.00002012212,0.001595387,0.0004630539,0.00003086059,3.192956e-7,0.02212887,0.0001030756,7.724408e-7,0.0001222248,0.9753473,0.0001290258],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.00003788044,0.01695869,0.00009834698,0.0937336,0.005061302,0.008684956,0.04721014,0.001366189,0.8268489],"genre_scores_gemma":[0.003334981,0.002309346,0.00002513037,0.0006577932,0.00236387,0.001922625,0.008730552,0.0003090777,0.9803466],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.1534977,"threshold_uncertainty_score":0.9998691,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1655370857491855,"score_gpt":0.4201413016417844,"score_spread":0.2546042158925989,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}