{"id":"W4389988584","doi":"10.1109/scam59687.2023.00020","title":"Explaining Transformer-based Code Models: What Do They Learn? When They Do Not Work?","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University; University of Calgary","funders":"","keywords":"Computer science; KPI-driven code analysis; Source code; Code (set theory); Code generation; Downstream (manufacturing); Code review; Transformer; Security token; Static program analysis; Artificial neural network; Software engineering; Set (abstract data type); Software; Artificial intelligence; Programming language; Machine learning; Software development; Key (lock); Computer security; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.00115474,0.0002410867,0.0002312402,0.0003785993,0.0001709605,0.001129948,0.001642274,0.0001126936,0.00006919534],"category_scores_gemma":[0.000170528,0.0002033608,0.0001248225,0.0006981136,0.00003240973,0.001686753,0.0001739709,0.0003928532,0.0005231521],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008126164,"about_ca_system_score_gemma":0.0001329726,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006258289,"about_ca_topic_score_gemma":0.00003180248,"domain_scores_codex":[0.9973104,0.0000839701,0.0002641669,0.000631911,0.0008996387,0.0008099398],"domain_scores_gemma":[0.9972353,0.001486622,0.00002855284,0.000941321,0.00009460418,0.0002136235],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005927826,0.00008338998,0.001363869,0.00007382695,0.000068713,0.0001101511,0.02097752,0.586968,0.0005875693,0.02564577,0.002988682,0.3610733],"study_design_scores_gemma":[0.001201138,0.0001828166,0.0009332728,0.0004027788,0.000007972724,0.000007796722,0.001338746,0.9711107,0.003863253,0.01675407,0.00340202,0.0007954381],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06008921,0.0002403807,0.9300441,0.003778207,0.0005641849,0.000367534,0.000003454697,0.002990596,0.001922374],"genre_scores_gemma":[0.975721,0.0001695492,0.02273519,0.0002410198,0.00006262823,0.00009046606,0.000004932856,0.00004882259,0.0009263864],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9156318,"threshold_uncertainty_score":0.999907,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06524125755685459,"score_gpt":0.2859045553251602,"score_spread":0.2206632977683056,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}