{"id":"W4416079975","doi":"10.1117/12.3095478","title":"Consistency and stability benchmarking of Grad-CAM, SHAP, and LIME for diffuse and focal brain MRI classification","year":2025,"lang":"","type":"article","venue":"","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Benchmarking; Robustness (evolution); Consistency (knowledge bases); Stability (learning theory); Pipeline (software); Pattern recognition (psychology); Neuroimaging; Convolutional neural network","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01134857,0.002116018,0.001028173,0.002769357,0.0006266921,0.002833024,0.002216583,0.00249633,0.002896351],"category_scores_gemma":[0.03739702,0.0005010818,0.001333262,0.0009518869,0.001312312,0.002310399,0.003051342,0.001748526,0.001866655],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001030156,"about_ca_system_score_gemma":0.001239922,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003402834,"about_ca_topic_score_gemma":0.005430783,"domain_scores_codex":[0.9962037,0.001085254,0.0003478051,0.001420722,0.000664726,0.0002777482],"domain_scores_gemma":[0.9899148,0.005386231,0.000608091,0.002208324,0.001532915,0.0003497086],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.006384723,0.0008097962,0.08210809,0.002003175,0.002750798,0.0006701494,0.0008612664,0.2468604,0.02653155,0.006100595,0.0408579,0.5840615],"study_design_scores_gemma":[0.0004067711,0.001349453,0.04345292,0.0003591854,0.0004515509,0.001102638,0.0004775668,0.8791538,0.05002866,0.01369963,0.009336732,0.0001810623],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7350027,0.008135395,0.2015825,0.001822871,0.0008573891,0.0005747084,0.007841026,0.03510395,0.009079486],"genre_scores_gemma":[0.8924092,0.0007153588,0.08343107,0.0004544886,0.0001619761,0.0002188677,0.01787336,0.002074976,0.002660652],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01134857,"threshold_uncertainty_score":0.06001765,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05193027304978776,"score_gpt":0.3038734037802744,"score_spread":0.2519431307304866,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}