{"id":"W4405166943","doi":"10.1016/j.engappai.2024.109678","title":"Quantifying the trustworthiness of explainable artificial intelligence outputs in uncertain decision-making scenarios","year":2024,"lang":"en","type":"article","venue":"Engineering Applications of Artificial Intelligence","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Computer science; Trustworthiness; Artificial intelligence; Machine learning; Computer security","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.002092906,0.0004112502,0.0005126212,0.0009918306,0.0002394649,0.000356491,0.002693305,0.0001878775,0.00004696773],"category_scores_gemma":[0.0008818029,0.0003685695,0.0002289636,0.005409802,0.000279116,0.0008201771,0.0004653336,0.0006326567,0.0001339845],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001847403,"about_ca_system_score_gemma":0.0002743216,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003528756,"about_ca_topic_score_gemma":0.0002536649,"domain_scores_codex":[0.9956081,0.00008744626,0.001935533,0.0009271189,0.0007181959,0.0007236179],"domain_scores_gemma":[0.9950383,0.002663861,0.0002808984,0.001513058,0.0003926666,0.000111271],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001170969,0.00009011629,0.00002125585,0.0001254248,0.00001474923,0.00001188813,0.001652065,0.2786761,0.002274111,0.4552297,0.000006562937,0.2618863],"study_design_scores_gemma":[0.000005109107,0.00005294253,0.00002805364,0.0007179782,0.00001332699,0.00001357951,0.00144844,0.7806717,0.1026472,0.1135944,0.0004832771,0.0003239649],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01398098,0.001627765,0.9817836,0.0006201764,0.0006597498,0.0008991662,0.000009811882,0.0002932576,0.0001255291],"genre_scores_gemma":[0.9190083,0.000107868,0.0803203,0.00002565516,0.0001456153,0.0003320453,0.000003029636,0.00004353375,0.00001365995],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9050273,"threshold_uncertainty_score":0.9998766,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04832276726054405,"score_gpt":0.3265637760144577,"score_spread":0.2782410087539136,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}