{"id":"W6920799232","doi":"10.6084/m9.figshare.13719529","title":"Additional file 1 of The Matthews correlation coefficient (MCC) is more reliable than balanced accuracy, bookmaker informedness, and markedness in two-class confusion matrix evaluation","year":2021,"lang":"en","type":"article","venue":"Open MIND","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University Health Network","funders":"","keywords":"Randomness; Confusion; Matrix (chemical analysis); Markedness; Confusion matrix; Correlation coefficient","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.003334855,0.001392294,0.00106439,0.002248026,0.0008159668,0.001924833,0.00217654,0.001350171,0.8163244],"category_scores_gemma":[0.09301826,0.0006007061,0.0009400379,0.003126793,0.0003224697,0.002460529,0.001072298,0.001167226,0.1841503],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001100998,"about_ca_system_score_gemma":0.001773456,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004742005,"about_ca_topic_score_gemma":0.009421241,"domain_scores_codex":[0.9981802,0.0004261044,0.0002561325,0.000402828,0.0005848315,0.0001498963],"domain_scores_gemma":[0.9095997,0.07231409,0.002868847,0.004584267,0.01001113,0.0006219578],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002512429,0.00008610423,0.002178715,0.0008872554,0.00003324575,0.00003666871,0.00004523155,0.0004419721,0.00007660314,0.0009934264,0.9806966,0.01427291],"study_design_scores_gemma":[0.005746525,0.0005050823,0.04742349,0.003303409,0.0003076989,0.0007052704,0.0006138442,0.01322156,0.004243251,0.04202597,0.8815277,0.0003762178],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.0007608999,0.00002595993,0.002300718,0.0002454037,0.000063186,0.0001217387,0.9922124,0.002057928,0.002211816],"genre_scores_gemma":[0.03335555,0.0001408285,0.01936986,0.0008521111,0.0002747107,0.004395014,0.9145874,0.007170451,0.01985396],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9966651,"threshold_uncertainty_score":0.261991,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09709000465052663,"score_gpt":0.3954520550697885,"score_spread":0.2983620504192619,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}