{"id":"W4416079975","doi":"10.1117/12.3095478","title":"Consistency and stability benchmarking of Grad-CAM, SHAP, and LIME for diffuse and focal brain MRI classification","year":2025,"lang":"","type":"article","venue":"","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Benchmarking; Robustness (evolution); Consistency (knowledge bases); Stability (learning theory); Pipeline (software); Pattern recognition (psychology); Neuroimaging; Convolutional neural network","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.001749382,0.000343117,0.0005587239,0.0002547305,0.0005475818,0.0004650589,0.0003999444,0.0002273496,0.00002841084],"category_scores_gemma":[0.0008409383,0.0003432716,0.00008689995,0.0006182452,0.001273675,0.0008216153,0.0006330191,0.0002016138,0.000001111274],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000743545,"about_ca_system_score_gemma":0.0002742464,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003346802,"about_ca_topic_score_gemma":0.0006334486,"domain_scores_codex":[0.9966491,0.0002431644,0.00103361,0.00127484,0.000261112,0.0005381317],"domain_scores_gemma":[0.9957893,0.00249937,0.0002870333,0.0007597872,0.0004409724,0.0002235249],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000139175,0.000449318,0.04002833,0.0009039817,0.00007553962,0.00000178563,0.004017706,0.000003145265,0.01518237,0.66778,0.0003710985,0.2710476],"study_design_scores_gemma":[0.0008686897,0.0008348051,0.06006056,0.000290885,0.000101215,0.00001062618,0.003336582,0.8152204,0.02986673,0.08760083,0.001282934,0.0005257684],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4760392,0.001901773,0.5072237,0.01179883,0.000331823,0.001289943,0.00002043705,0.00004608659,0.001348288],"genre_scores_gemma":[0.9810942,0.000527707,0.01754121,0.0004767069,0.00003812523,0.00005326448,0.00000338054,0.00001130954,0.0002540394],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8152172,"threshold_uncertainty_score":0.999902,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05193027304978776,"score_gpt":0.3038734037802744,"score_spread":0.2519431307304866,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}