{"id":"W6977455728","doi":"10.6084/m9.figshare.c.6581771.v1","title":"The Matthews correlation coefficient (MCC) should replace the ROC AUC as the standard metric for assessing binary classification","year":2023,"lang":"en","type":"other","venue":"Figshare","topic":"Health, Education, and Cultural Studies","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Receiver operating characteristic; Binary classification; Matthews correlation coefficient; Binary number; Metric (unit); Confusion matrix; Correlation; Sensitivity (control systems); False positive rate","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02425335,0.001779913,0.002388952,0.006327324,0.001209917,0.004566547,0.001864848,0.003191105,0.002668416],"category_scores_gemma":[0.1683061,0.0004957892,0.001255713,0.005705014,0.00340782,0.00353255,0.001837113,0.00342027,0.003047693],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001577791,"about_ca_system_score_gemma":0.002279509,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003122068,"about_ca_topic_score_gemma":0.003078266,"domain_scores_codex":[0.9662609,0.01125679,0.004648648,0.004544599,0.01239608,0.0008930899],"domain_scores_gemma":[0.8448676,0.08325186,0.02353695,0.01345689,0.03173129,0.003155334],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001032702,0.0002558304,0.1844263,0.003574504,0.002437492,0.001087888,0.001087282,0.02705279,0.0107044,0.07351132,0.2251153,0.4697141],"study_design_scores_gemma":[0.0001887041,0.002315913,0.2424089,0.003361407,0.001236908,0.00649882,0.00182863,0.2116338,0.03958043,0.2419076,0.2476363,0.001402553],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1744409,0.04060132,0.652667,0.03895144,0.0156698,0.001140274,0.007196893,0.009007873,0.06032456],"genre_scores_gemma":[0.791936,0.004098663,0.1817656,0.006733924,0.004219332,0.0009660579,0.003004844,0.001481236,0.005794264],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9757466,"threshold_uncertainty_score":0.1282655,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1655370857491855,"score_gpt":0.4201413016417844,"score_spread":0.2546042158925989,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}