{"id":"W4400163580","doi":"10.1093/jamia/ocae165","title":"Towards objective and systematic evaluation of bias in artificial intelligence for medical imaging","year":2024,"lang":"en","type":"article","venue":"Journal of the American Medical Informatics Association","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":22,"is_retracted":false,"has_abstract":true,"ca_institutions":"Hotchkiss Brain Institute; Alberta Children's Hospital; University of Calgary","funders":"Alberta Innovates; Alberta Children's Hospital Foundation; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Children's Hospital Foundation","keywords":"Artificial intelligence; Computer science; Medical imaging; Machine learning; Data science","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.2420384,0.001939959,0.001871528,0.004347153,0.000981789,0.005451221,0.002871142,0.002770115,0.001798697],"category_scores_gemma":[0.4775245,0.0008012792,0.00230729,0.002050407,0.005245516,0.005105245,0.00522489,0.003562526,0.0002772747],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003908102,"about_ca_system_score_gemma":0.007985067,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00153375,"about_ca_topic_score_gemma":0.001615873,"domain_scores_codex":[0.7821528,0.1807224,0.008584677,0.005874394,0.02188374,0.0007818965],"domain_scores_gemma":[0.429149,0.4428118,0.04643534,0.04427503,0.03575618,0.001572583],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.002162938,0.001048068,0.113031,0.00864626,0.008252341,0.0003062509,0.002476497,0.1666084,0.01396489,0.1733484,0.008391391,0.5017636],"study_design_scores_gemma":[0.000640786,0.00381229,0.03538132,0.006147894,0.002323845,0.0003644016,0.001355604,0.5073705,0.03861395,0.3823901,0.02119725,0.0004021966],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.05411636,0.00750685,0.9271209,0.005044977,0.0003118781,0.00170439,0.0006154369,0.0004930908,0.003086145],"genre_scores_gemma":[0.4964138,0.00135163,0.4971486,0.001431982,0.0003150495,0.002178048,0.0005585549,0.0001978268,0.0004045524],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.2420384,"threshold_uncertainty_score":0.9347016,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1236259313214339,"score_gpt":0.4627124433982143,"score_spread":0.3390865120767803,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}