{"id":"W4388976300","doi":"10.21203/rs.3.rs-3528413/v1","title":"Improving Explainable AI Interpretability: Mathematical Models for Evaluating Explanation Methods.","year":2023,"lang":"en","type":"preprint","venue":"Research Square","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Interpretability; Transparency (behavior); Computer science; Accountability; Artificial intelligence; Correctness; Management science; Machine learning; Algorithm; Computer security","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow","scholarly_communication"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.03068607,0.0004518781,0.000708612,0.001091253,0.0007733828,0.001768631,0.003371307,0.0005291165,0.00004854926],"category_scores_gemma":[0.01523923,0.0004505285,0.0003839033,0.001175168,0.0001894566,0.001435265,0.006572414,0.001995968,0.0002331448],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001062301,"about_ca_system_score_gemma":0.001268693,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006275497,"about_ca_topic_score_gemma":0.00006735371,"domain_scores_codex":[0.9906226,0.002350413,0.001141926,0.002029285,0.002175864,0.001679931],"domain_scores_gemma":[0.9862423,0.007314233,0.0002866182,0.002661856,0.003163485,0.0003315057],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001084951,0.000335943,0.00002265047,0.008435667,0.0001237264,0.00004237694,0.01632801,0.09239598,0.003190781,0.5306395,0.001810842,0.3465661],"study_design_scores_gemma":[0.00006018932,0.0001663974,0.000001913703,0.0004383987,0.000007380968,0.000001874454,0.001323117,0.5296791,0.007336309,0.460685,0.00007515935,0.0002252228],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.001349974,0.0002643589,0.9892733,0.002834494,0.0006338487,0.004062642,0.00003417625,0.0007051985,0.0008420512],"genre_scores_gemma":[0.1647043,0.00004800224,0.826868,0.0000966197,0.0003075553,0.006365649,0.00007631377,0.0001427631,0.001390818],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.4372831,"threshold_uncertainty_score":0.9997947,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3921491170560181,"score_gpt":0.5530217707892984,"score_spread":0.1608726537332803,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}