{"id":"W3206995700","doi":"10.18653/v1/2022.findings-emnlp.125","title":"Evaluating the Faithfulness of Importance Measures in NLP by Recursively Masking Allegedly Important Tokens and Retraining","year":2022,"lang":"en","type":"preprint","venue":"","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; Polytechnique Montréal; Mila - Quebec Artificial Intelligence Institute","funders":"Compute Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Masking (illustration); Computer science; Retraining; Metric (unit); Artificial intelligence; Natural language processing; Task (project management); Machine learning; Closed captioning; Language model","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01209023,0.001894096,0.001005436,0.00399166,0.0006388279,0.002737623,0.001818216,0.002513702,0.001682295],"category_scores_gemma":[0.06138995,0.00058189,0.001314269,0.002132638,0.001579424,0.00742529,0.002318996,0.00348305,0.0005666683],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001681819,"about_ca_system_score_gemma":0.0009935091,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003908806,"about_ca_topic_score_gemma":0.005674442,"domain_scores_codex":[0.994785,0.002529388,0.000499193,0.001115404,0.0008665314,0.0002045859],"domain_scores_gemma":[0.9488976,0.04142483,0.002037809,0.00511646,0.001917862,0.0006053195],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002357084,0.0006751777,0.04414038,0.001883351,0.001325804,0.000428513,0.001512372,0.3386358,0.02003118,0.0120527,0.007683992,0.5692737],"study_design_scores_gemma":[0.00008629946,0.0006008564,0.01246295,0.000110886,0.0002492109,0.0002556037,0.000273517,0.934846,0.01725019,0.03159336,0.002195378,0.00007572836],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4615981,0.008581776,0.5107225,0.001967511,0.0002887581,0.0002730515,0.001619139,0.008287646,0.006661508],"genre_scores_gemma":[0.9056782,0.0006400388,0.09023299,0.0001915529,0.00008941325,0.00008192605,0.001930823,0.0003523179,0.0008028191],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01209023,"threshold_uncertainty_score":0.06393999,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1136404728347689,"score_gpt":0.3655468697069478,"score_spread":0.2519063968721789,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}