{"id":"W4297081730","doi":"10.36227/techrxiv.21067438","title":"A Trustworthy View on XAI Method Evaluation","year":2022,"lang":"en","type":"preprint","venue":"","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Concordia University","funders":"","keywords":"Consistency (knowledge bases); Trustworthiness; Computer science; Centroid; Process (computing); Cluster analysis; Data mining; Order (exchange); Feature (linguistics); Artificial intelligence; Business; Computer security","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.1936692,0.001947615,0.002242483,0.006117856,0.003076579,0.01519088,0.005164653,0.004445503,0.00335683],"category_scores_gemma":[0.4903977,0.001787122,0.00195362,0.003046544,0.009267523,0.01319958,0.008826656,0.008948884,0.001337266],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00516337,"about_ca_system_score_gemma":0.007255599,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003242633,"about_ca_topic_score_gemma":0.002402445,"domain_scores_codex":[0.6923021,0.2121615,0.01337529,0.01111619,0.06884434,0.002200555],"domain_scores_gemma":[0.4524859,0.3444721,0.02612646,0.09757851,0.07598284,0.003354165],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0009227765,0.0005389542,0.02895693,0.001739574,0.0008053118,0.00054165,0.009349101,0.06026895,0.01106184,0.3245271,0.01328486,0.5480028],"study_design_scores_gemma":[0.000260773,0.0006923557,0.007373156,0.00181007,0.0001945525,0.000722151,0.002023719,0.5934898,0.02880722,0.3320936,0.03225869,0.0002739583],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01439499,0.001268325,0.9729766,0.003609875,0.0001771351,0.0003851392,0.00009147051,0.001160434,0.005936015],"genre_scores_gemma":[0.2061073,0.0005179927,0.7886385,0.0009962961,0.0001863638,0.0007330822,0.0002549196,0.001023363,0.001542059],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.8063307,"threshold_uncertainty_score":0.9943494,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1202530959946061,"score_gpt":0.4160330938312373,"score_spread":0.2957799978366311,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}