{"id":"W6920920990","doi":"10.6084/m9.figshare.13719529.v1","title":"Additional file 1 of The Matthews correlation coefficient (MCC) is more reliable than balanced accuracy, bookmaker informedness, and markedness in two-class confusion matrix evaluation","year":2021,"lang":"en","type":"article","venue":"Figshare","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University Health Network","funders":"","keywords":"Randomness; Confusion; Matrix (chemical analysis); Markedness; Confusion matrix; Correlation coefficient","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.003798118,0.001414984,0.001117817,0.002329032,0.0008832638,0.002159113,0.002350291,0.001415189,0.8265628],"category_scores_gemma":[0.09930911,0.0006685376,0.001032472,0.003325968,0.0003565013,0.002706478,0.001117619,0.001250719,0.208856],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001145657,"about_ca_system_score_gemma":0.00200032,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004440194,"about_ca_topic_score_gemma":0.009545526,"domain_scores_codex":[0.997995,0.0004772514,0.0002775723,0.0004310133,0.0006520986,0.0001670486],"domain_scores_gemma":[0.8989061,0.08164756,0.002866957,0.005345855,0.01056527,0.0006683696],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002413698,0.00007604771,0.002033534,0.0008282141,0.00003477573,0.00003612871,0.00004384663,0.0005605898,0.00007325869,0.001086028,0.9806529,0.01433332],"study_design_scores_gemma":[0.005108899,0.0004850435,0.03963438,0.003126029,0.0003188328,0.0006652076,0.0005478763,0.01486601,0.00420301,0.04636388,0.8842934,0.0003874927],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"other","genre_scores_codex":[0.0007448096,0.00002779103,0.00292974,0.0002723203,0.00006808143,0.0001294712,0.990481,0.002741839,0.002605019],"genre_scores_gemma":[0.03660065,0.0001534953,0.02333228,0.0009313102,0.0002706872,0.004360863,0.9032184,0.009845236,0.02128699],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.9962019,"threshold_uncertainty_score":0.2473871,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08969401580553116,"score_gpt":0.3639828464937436,"score_spread":0.2742888306882124,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}