{"id":"W4416329793","doi":"10.1088/3049-477x/ae209d","title":"Integrating explainability and bias detection in binary medical image classification: a systematic review","year":2025,"lang":"en","type":"article","venue":"Machine Learning Health","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Interpretability; Debiasing; Context (archaeology); Counterfactual thinking; Binary classification; Binary number; Feature (linguistics); Adversarial system","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006839518,0.0001626534,0.0005214022,0.000256818,0.0003543079,0.0001149731,0.0004746827,0.00007384148,0.00001248355],"category_scores_gemma":[0.006880408,0.0001392917,0.00004799786,0.001255862,0.00006894595,0.0003435945,0.00023131,0.0007701566,0.00002056774],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003277393,"about_ca_system_score_gemma":0.0003519618,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001507542,"about_ca_topic_score_gemma":0.001086215,"domain_scores_codex":[0.9960713,0.001815811,0.0009166772,0.0005085206,0.0003477791,0.0003399135],"domain_scores_gemma":[0.9982767,0.0007005369,0.0002913228,0.0004878135,0.000103589,0.0001400385],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000299753,0.0004322768,0.01864715,0.6408119,0.00003831532,0.0001013465,0.007178006,0.0001208643,0.0004319769,0.09317634,0.0002213477,0.2388105],"study_design_scores_gemma":[0.0001486356,0.0002347418,0.003191317,0.06325737,0.00001078131,0.0000412431,0.0008345012,0.929685,0.0001295384,0.001950056,0.0003075912,0.0002092083],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0258637,0.0395507,0.8665584,0.06328458,0.00036664,0.00260524,8.255127e-7,0.0005926495,0.00117727],"genre_scores_gemma":[0.9886472,0.002838989,0.004996246,0.003126121,0.00001946427,0.0001934533,0.000004595872,0.00001070984,0.0001632233],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9627835,"threshold_uncertainty_score":0.8236988,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03279360825787257,"score_gpt":0.348339967714137,"score_spread":0.3155463594562644,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}