{"id":"W4410542464","doi":"10.1148/radiol.241674","title":"Pitfalls and Best Practices in Evaluation of AI Algorithmic Biases in Radiology","year":2025,"lang":"en","type":"review","venue":"Radiology","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":17,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University","funders":"National Cancer Institute; National Institutes of Health","keywords":"Medicine; MEDLINE; Medical physics; Data science; Artificial intelligence; Radiology; Machine learning; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002272344,0.0001843016,0.001507372,0.000830592,0.00001837065,0.00000347905,0.00008916461,0.0005944281,0.0000602752],"category_scores_gemma":[0.007940438,0.0001583136,0.00007802121,0.0004634913,0.0001448777,0.00005490429,0.00002467373,0.0005591866,0.00001242304],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003245389,"about_ca_system_score_gemma":0.00267659,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002354139,"about_ca_topic_score_gemma":0.001025519,"domain_scores_codex":[0.996932,0.00131155,0.0009947806,0.0003889782,0.0001164962,0.0002561392],"domain_scores_gemma":[0.9962304,0.002698117,0.0005894772,0.0002545385,0.0001747253,0.00005273473],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00002186722,0.0001264953,0.00227097,0.006487443,0.00003485328,0.00001088959,0.0002512849,0.000008433361,0.000001457267,0.0001280385,0.0001668459,0.9904914],"study_design_scores_gemma":[0.0008832896,0.003234851,0.002704171,0.09121854,0.005302243,0.003520787,0.002173002,0.003852715,0.00004297561,0.004608628,0.8814842,0.000974604],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.008855586,0.9881636,0.000008226172,0.0006072821,0.0005794421,0.001426823,0.000008102008,0.000006322764,0.0003446346],"genre_scores_gemma":[0.002269358,0.9967217,0.000123654,0.0001255073,0.0002508709,0.0002549295,0.0001548656,0.00001102588,0.00008812758],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.9895168,"threshold_uncertainty_score":0.9506019,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5651568962970117,"score_gpt":0.6028591880848577,"score_spread":0.03770229178784601,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}