{"id":"W6920799232","doi":"10.6084/m9.figshare.13719529","title":"Additional file 1 of The Matthews correlation coefficient (MCC) is more reliable than balanced accuracy, bookmaker informedness, and markedness in two-class confusion matrix evaluation","year":2021,"lang":"en","type":"article","venue":"Open MIND","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University Health Network","funders":"","keywords":"Randomness; Confusion; Matrix (chemical analysis); Markedness; Confusion matrix; Correlation coefficient","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.003441568,0.000141024,0.0002689781,0.00008189505,0.0001919351,0.0003181511,0.000577662,0.0000752354,0.7069971],"category_scores_gemma":[0.006270701,0.00008592209,0.00007353393,0.0006897772,0.0001368916,0.0004954182,0.0005072083,0.0001524392,0.0003360993],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001132591,"about_ca_system_score_gemma":0.0006176269,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006190207,"about_ca_topic_score_gemma":0.000556161,"domain_scores_codex":[0.9960695,0.0003109871,0.0008091967,0.000469989,0.00215062,0.0001896912],"domain_scores_gemma":[0.9958243,0.002131358,0.0005275591,0.0006157486,0.0008456361,0.00005543598],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001225433,0.0002073615,0.008851609,0.00001220414,0.00001169287,0.000001806095,0.001519192,0.004123319,0.0003463305,0.00002334539,0.9317983,0.05298226],"study_design_scores_gemma":[0.001566029,0.0000466277,0.1814881,0.0006947957,0.0000276821,0.000009282884,0.004264865,0.1137054,0.001554122,0.00152075,0.6948916,0.0002307692],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8796775,0.00009522809,0.0001328298,0.003942598,0.0009066234,0.003351115,0.0683599,0.000003472788,0.04353067],"genre_scores_gemma":[0.9052484,0.00004310237,0.009117014,0.001332079,0.0001215969,0.0009767066,0.01827611,0.00003435903,0.06485064],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.706661,"threshold_uncertainty_score":0.7507068,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09709000465052663,"score_gpt":0.3954520550697885,"score_spread":0.2983620504192619,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}