{"id":"W4410542464","doi":"10.1148/radiol.241674","title":"Pitfalls and Best Practices in Evaluation of AI Algorithmic Biases in Radiology","year":2025,"lang":"en","type":"review","venue":"Radiology","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":17,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University","funders":"National Cancer Institute; National Institutes of Health","keywords":"Medicine; MEDLINE; Medical physics; Data science; Artificial intelligence; Radiology; Machine learning; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1538707,0.001510689,0.00396011,0.009917353,0.001188722,0.007480751,0.00573422,0.004759713,0.002986721],"category_scores_gemma":[0.3450989,0.001350933,0.00302579,0.006772667,0.00947115,0.009065136,0.004201968,0.007831427,0.001277874],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004837571,"about_ca_system_score_gemma":0.01416194,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00405056,"about_ca_topic_score_gemma":0.006075745,"domain_scores_codex":[0.8429644,0.1009398,0.0242538,0.005284448,0.02575929,0.0007983019],"domain_scores_gemma":[0.3940805,0.555415,0.0155526,0.008460146,0.02544243,0.001049307],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001174066,0.00003832882,0.001543208,0.0643317,0.001394599,0.0001345919,0.001888396,0.001121346,0.0002370123,0.0524979,0.01821999,0.8584756],"study_design_scores_gemma":[0.0001354301,0.000322882,0.00376671,0.3300619,0.00251405,0.001885549,0.001966305,0.002374443,0.001723184,0.1946331,0.4603073,0.0003092218],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.0003035769,0.9740209,0.009247709,0.01290105,0.001176452,0.0001017982,0.00006853164,0.0000526908,0.002127222],"genre_scores_gemma":[0.0193005,0.915211,0.049448,0.01180242,0.002963396,0.0005217385,0.0001337605,0.00008839838,0.0005308047],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.8461293,"threshold_uncertainty_score":0.8137559,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5651568962970117,"score_gpt":0.6028591880848577,"score_spread":0.03770229178784601,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}