{"id":"W4363678929","doi":"10.1109/mc.2022.3233806","title":"A Trustworthy View on Explainable Artificial Intelligence Method Evaluation","year":2023,"lang":"en","type":"article","venue":"Computer","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"ca_institutions":"Concordia University","funders":"","keywords":"Trustworthiness; Consistency (knowledge bases); Computer science; Process (computing); Artificial intelligence; Computer security","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.2915934,0.001963137,0.003954742,0.01405242,0.005856392,0.02482642,0.006601961,0.01026648,0.003287395],"category_scores_gemma":[0.5918122,0.002565147,0.002988647,0.005163901,0.02175716,0.03094737,0.01475278,0.0141297,0.0008838493],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01022531,"about_ca_system_score_gemma":0.01310451,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004076554,"about_ca_topic_score_gemma":0.002332615,"domain_scores_codex":[0.5458176,0.2674997,0.03320271,0.01492048,0.1347796,0.003779924],"domain_scores_gemma":[0.2466103,0.5363854,0.03876202,0.09134858,0.08290776,0.003986049],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000410257,0.0003378615,0.01212882,0.001463645,0.0005338069,0.0003658755,0.006748167,0.01315001,0.002196213,0.8447938,0.004237879,0.1136337],"study_design_scores_gemma":[0.0003201546,0.0005673547,0.003553811,0.002410345,0.0002853133,0.0005453221,0.002080842,0.1009113,0.009519917,0.8549879,0.02457971,0.0002380603],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"commentary","genre_scores_codex":[0.01550774,0.002759991,0.9491472,0.01731932,0.0003000485,0.0007999978,0.0001285311,0.0006553244,0.01338187],"genre_scores_gemma":[0.304394,0.001232688,0.686883,0.002471122,0.0005724581,0.001524094,0.000331617,0.0006848248,0.001906096],"genre_candidate":"commentary","genre_consensus":null,"teacher_disagreement_score":0.2915934,"threshold_uncertainty_score":0.8735915,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1278275846633145,"score_gpt":0.3804678654842235,"score_spread":0.2526402808209089,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}