{"id":"W6920920990","doi":"10.6084/m9.figshare.13719529.v1","title":"Additional file 1 of The Matthews correlation coefficient (MCC) is more reliable than balanced accuracy, bookmaker informedness, and markedness in two-class confusion matrix evaluation","year":2021,"lang":"en","type":"article","venue":"Figshare","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University Health Network","funders":"","keywords":"Randomness; Confusion; Matrix (chemical analysis); Markedness; Confusion matrix; Correlation coefficient","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0009779085,0.0001520792,0.0002355855,0.00009251269,0.0001735878,0.0001504268,0.000363925,0.00009069016,0.9654056],"category_scores_gemma":[0.02461278,0.00009472932,0.0001030012,0.0007439293,0.00003795327,0.000337639,0.0003055456,0.0001799758,0.0007457796],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001201732,"about_ca_system_score_gemma":0.0004999191,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000151113,"about_ca_topic_score_gemma":0.0001695442,"domain_scores_codex":[0.9960198,0.0002357715,0.000708373,0.0004141637,0.002418359,0.0002035013],"domain_scores_gemma":[0.9945693,0.002999922,0.0005214454,0.0005779379,0.001274019,0.0000573816],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00002686973,0.00006303639,0.0008500229,0.00003674929,0.000004536703,9.895222e-7,0.0003264333,0.002279337,0.0000364419,0.00000731888,0.99254,0.003828269],"study_design_scores_gemma":[0.0007229096,0.00002447912,0.1688347,0.00255127,0.00001000994,0.000006250422,0.00101781,0.1197406,0.0003802157,0.0008844217,0.7056516,0.000175796],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.01038267,0.0000772776,0.000004314511,0.0007551431,0.0001898196,0.0008035829,0.9828204,0.00001289782,0.004953918],"genre_scores_gemma":[0.1782199,0.00001094743,0.0004916047,0.001565817,0.000114901,0.001715737,0.8063076,0.00002641527,0.01154702],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.9646598,"threshold_uncertainty_score":0.9836033,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08969401580553116,"score_gpt":0.3639828464937436,"score_spread":0.2742888306882124,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}