{"id":"W3172444956","doi":"10.1109/access.2021.3084050","title":"The Matthews Correlation Coefficient (MCC) is More Informative Than Cohen’s Kappa and Brier Score in Binary Classification Assessment","year":2021,"lang":"en","type":"article","venue":"IEEE Access","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":429,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Brier score; Kappa; Cohen's kappa; Correlation; Computer science; Statistics; Artificial intelligence; Binary classification; Metric (unit); Binary number; Odds; Machine learning; Mathematics; Natural language processing; Logistic regression; Support vector machine; Arithmetic","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.04748533,0.002029689,0.002728327,0.01878001,0.002366354,0.004328595,0.002052732,0.003295845,0.00371662],"category_scores_gemma":[0.2309605,0.0005500402,0.001597397,0.0151497,0.004038429,0.005089923,0.003263172,0.002487127,0.002142449],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001600248,"about_ca_system_score_gemma":0.002380316,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003856976,"about_ca_topic_score_gemma":0.005155806,"domain_scores_codex":[0.9303644,0.02466812,0.0103636,0.01046705,0.02281631,0.001320581],"domain_scores_gemma":[0.692037,0.2103221,0.03930378,0.02041128,0.03449426,0.003431656],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001186598,0.0003482088,0.4350144,0.003715152,0.004061579,0.000978463,0.00497271,0.02609793,0.005308702,0.06540132,0.07289478,0.3800202],"study_design_scores_gemma":[0.0002186609,0.00237166,0.465639,0.002114486,0.00198283,0.003927005,0.006080043,0.109368,0.01744583,0.2283192,0.1609033,0.001629871],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3018839,0.02365131,0.5825098,0.006985864,0.004290433,0.001482971,0.009727638,0.003979805,0.06548828],"genre_scores_gemma":[0.8530046,0.002222133,0.1320862,0.001302013,0.001366601,0.001053774,0.003145321,0.0008045346,0.005014942],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9525146,"threshold_uncertainty_score":0.2511294,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1877563290803927,"score_gpt":0.4289589341158113,"score_spread":0.2412026050354187,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}