{"id":"W4226193037","doi":"10.1609/aaai.v36i11.21452","title":"Evaluating Explainable AI on a Multi-Modal Medical Imaging Task: Can Existing Algorithms Fulfill Clinical Requirements?","year":2022,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia; Simon Fraser University","funders":"BC Cancer Foundation; Compute Canada; Nvidia","keywords":"Computer science; Modality (human–computer interaction); Artificial intelligence; Modal; Leverage (statistics); Machine learning; Metric (unit); Feature (linguistics); Medical imaging; Data mining; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02091744,0.001405036,0.0009226261,0.002218919,0.0005735975,0.003182898,0.001847792,0.002340822,0.002359941],"category_scores_gemma":[0.1571443,0.0003565698,0.001242536,0.001106835,0.001262296,0.003005504,0.00187848,0.002212861,0.0004466401],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001973319,"about_ca_system_score_gemma":0.003037727,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002883108,"about_ca_topic_score_gemma":0.004735362,"domain_scores_codex":[0.9836806,0.01114526,0.001200382,0.001672584,0.001976974,0.0003241043],"domain_scores_gemma":[0.8394157,0.1402733,0.005039849,0.007148077,0.006980615,0.001142532],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003500968,0.0009158844,0.07999068,0.003680315,0.001463154,0.0004345813,0.002758989,0.1953316,0.01079822,0.01097818,0.008981859,0.6811656],"study_design_scores_gemma":[0.0004156621,0.001613372,0.01921626,0.0005272975,0.0005657573,0.0005347724,0.001238969,0.9198194,0.01439517,0.03519997,0.006334445,0.0001388486],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4883326,0.006264481,0.4847751,0.006902933,0.0002672574,0.001484912,0.002486222,0.00392649,0.005560019],"genre_scores_gemma":[0.8096957,0.0005799989,0.1865595,0.0005827784,0.00004576119,0.0003808356,0.001639418,0.0001477547,0.0003682502],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02091744,"threshold_uncertainty_score":0.1106234,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2772565168370234,"score_gpt":0.4433974170613346,"score_spread":0.1661409002243112,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}