{"id":"W4415517332","doi":"10.1016/j.compeleceng.2025.110743","title":"A comprehensive evaluation of metrics on their ability to capture the degrees of Non-IIDness with label skew in federated learning","year":2025,"lang":"en","type":"article","venue":"Computers & Electrical Engineering","topic":"Privacy-Preserving Technologies in Data","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Skew; Metric (unit); Overhead (engineering); Rank (graph theory); Performance metric; Computation; Learning to rank; Strengths and weaknesses","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03796684,0.002301167,0.002495336,0.007367785,0.001367106,0.003760634,0.002561514,0.003106187,0.000859925],"category_scores_gemma":[0.119057,0.0003431402,0.001099079,0.006280696,0.002159256,0.008008773,0.003885724,0.002291997,0.000397581],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00383301,"about_ca_system_score_gemma":0.002839068,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003481531,"about_ca_topic_score_gemma":0.003641891,"domain_scores_codex":[0.9660202,0.01378913,0.003458126,0.003641878,0.01164998,0.001440755],"domain_scores_gemma":[0.8064837,0.1280975,0.01534032,0.02166373,0.02392885,0.004485811],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003328551,0.001240666,0.1085279,0.001787726,0.00138338,0.0002003515,0.0006032187,0.2638447,0.005294388,0.02016581,0.01247716,0.5811461],"study_design_scores_gemma":[0.0001237552,0.002597398,0.02031997,0.0004571021,0.0003116185,0.0006154678,0.0004644764,0.9274991,0.01030581,0.03184452,0.005288914,0.0001717728],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5238525,0.02308247,0.4296611,0.002376763,0.0009115966,0.0007297542,0.005527262,0.006244136,0.007614329],"genre_scores_gemma":[0.8770339,0.001722006,0.1155692,0.0002504343,0.0001666735,0.000219036,0.003810802,0.0002257334,0.00100213],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9620332,"threshold_uncertainty_score":0.2007902,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02739944413914019,"score_gpt":0.275127713051404,"score_spread":0.2477282689122638,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}