{"id":"W4283800998","doi":"10.1609/aaai.v36i11.21452","title":"Evaluating Explainable AI on a Multi-Modal Medical Imaging Task: Can Existing Algorithms Fulfill Clinical Requirements?","year":2022,"lang":"en","type":"article","venue":"","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":61,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia; Simon Fraser University","funders":"BC Cancer Foundation; Compute Canada; Nvidia","keywords":"Computer science; Modality (human–computer interaction); Artificial intelligence; Leverage (statistics); Modal; Machine learning; Metric (unit); Feature (linguistics); Medical imaging; Data mining; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01775237,0.001352589,0.000879745,0.002131971,0.0005338378,0.002942351,0.001690446,0.002110532,0.002497463],"category_scores_gemma":[0.1360576,0.0003271801,0.001196874,0.001052925,0.001169411,0.00263842,0.00176528,0.002021781,0.0004298827],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001874379,"about_ca_system_score_gemma":0.002915463,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002719746,"about_ca_topic_score_gemma":0.004323992,"domain_scores_codex":[0.9865822,0.008968635,0.0009871998,0.001479352,0.001693523,0.0002890741],"domain_scores_gemma":[0.8673817,0.115385,0.004430358,0.005914317,0.005925236,0.0009633814],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003284912,0.0008388507,0.08003683,0.003935446,0.001394392,0.0004526208,0.002594671,0.2028376,0.01145469,0.01125318,0.008746516,0.6731703],"study_design_scores_gemma":[0.0003629663,0.001501124,0.01967242,0.0005446889,0.0005359923,0.0005531578,0.001175217,0.9197669,0.01439117,0.03510605,0.006258972,0.0001313007],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4653212,0.006111626,0.5086732,0.006216265,0.0002553283,0.001439717,0.002635415,0.003754548,0.005592721],"genre_scores_gemma":[0.8131419,0.0005804967,0.183061,0.000528516,0.00004429973,0.0003846804,0.001738415,0.0001404329,0.0003801418],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01775237,"threshold_uncertainty_score":0.09388465,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2237399218700203,"score_gpt":0.4682253697246115,"score_spread":0.2444854478545912,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}