{"id":"W4288357452","doi":"10.48550/arxiv.1905.04667","title":"Functional Correlations in the Pursuit of Performance Assessment of\\n Classifiers","year":2019,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Multi-Criteria Decision Making","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University","funders":"","keywords":"Confusion; Association (psychology); Correlation; Artificial intelligence; Machine learning; Computer science; Mathematics; Psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0415947,0.002162124,0.00151724,0.007884879,0.001374155,0.003965644,0.00195471,0.00257673,0.0008672374],"category_scores_gemma":[0.1496245,0.0004932365,0.0008618951,0.006170507,0.005214988,0.00506677,0.003941565,0.00213118,0.0003063278],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002058932,"about_ca_system_score_gemma":0.00193823,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002320437,"about_ca_topic_score_gemma":0.002029792,"domain_scores_codex":[0.9591553,0.02704509,0.001983998,0.002763679,0.00818787,0.0008640203],"domain_scores_gemma":[0.8490599,0.1237146,0.01210215,0.004997614,0.00883837,0.001287308],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0006670875,0.0003465501,0.0796769,0.001016232,0.001234461,0.0004418204,0.002265468,0.2568686,0.004551186,0.3315387,0.003215612,0.3181775],"study_design_scores_gemma":[0.00003039548,0.0004765448,0.0165133,0.0002252593,0.0001535748,0.0002460115,0.0005437498,0.6882704,0.004181468,0.286517,0.002663853,0.0001783158],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1727314,0.003398664,0.8126221,0.001690876,0.0001625009,0.0001954573,0.000209722,0.0002143891,0.008774832],"genre_scores_gemma":[0.9042111,0.0006973956,0.09375047,0.0001614555,0.0002487229,0.0002141744,0.0001631146,0.00004967664,0.0005039115],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0415947,"threshold_uncertainty_score":0.2199764,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3522135768303779,"score_gpt":0.3090095257840579,"score_spread":0.04320405104631991,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}