{"id":"W4408280719","doi":"10.1109/access.2025.3549702","title":"Machine Learning Evaluation Metric Discrepancies Across Programming Languages and Their Components in Medical Imaging Domains: Need for Standardization","year":2025,"lang":"en","type":"article","venue":"IEEE Access","topic":"Radiomics and Machine Learning in Medical Imaging","field":"Medicine","cited_by":17,"is_retracted":false,"has_abstract":true,"ca_institutions":"Teck (Canada); University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Standardization; Computer science; Metric (unit); Programming language; Artificial intelligence; Medical imaging; Natural language processing; Machine learning","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1723185,0.00157692,0.001762962,0.00607743,0.001513224,0.009668809,0.003535214,0.00135996,0.000930293],"category_scores_gemma":[0.4702399,0.0007189647,0.001487259,0.005826638,0.003384465,0.008993938,0.005971846,0.004774914,0.0008354027],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003992614,"about_ca_system_score_gemma":0.007919739,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005234102,"about_ca_topic_score_gemma":0.004159427,"domain_scores_codex":[0.8302532,0.08512262,0.02452539,0.01173393,0.04623498,0.002129806],"domain_scores_gemma":[0.4923735,0.2993291,0.02233712,0.05272193,0.1299931,0.003245392],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0007738934,0.0002724702,0.08097325,0.00283642,0.0009846601,0.0001794053,0.004695002,0.01695095,0.005828357,0.02686318,0.0223147,0.8373277],"study_design_scores_gemma":[0.0004580669,0.003935147,0.2420291,0.0155852,0.001734882,0.003072556,0.01295416,0.2209951,0.06014835,0.1977954,0.2399117,0.001380208],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2186514,0.02587481,0.7011389,0.01539198,0.002197952,0.001724559,0.002448024,0.01219161,0.02038078],"genre_scores_gemma":[0.5780881,0.003522854,0.4012063,0.003633759,0.0004550511,0.002263633,0.004117866,0.005153954,0.001558502],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.8276815,"threshold_uncertainty_score":0.9113182,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02084814554002965,"score_gpt":0.4047388944958217,"score_spread":0.3838907489557921,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}