{"id":"W4416961926","doi":"10.1109/pst65910.2025.11268833","title":"Using Counterfactuals for Explainable Android Malware Detection","year":2025,"lang":"","type":"article","venue":"","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"British Columbia Institute of Technology","funders":"","keywords":"Counterfactual thinking; Malware; Counterfactual conditional; Android (operating system); Android malware; GRASP; Static analysis; Mobile device","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0006169485,0.0004625085,0.0004907195,0.0006689605,0.0009767915,0.0005733945,0.0008469499,0.0003346573,0.00009860996],"category_scores_gemma":[0.0002948611,0.0005127635,0.0002349736,0.001331102,0.0001170109,0.001888594,0.0004961523,0.000283446,0.00001555771],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007291433,"about_ca_system_score_gemma":0.0003095421,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001085158,"about_ca_topic_score_gemma":0.00005621819,"domain_scores_codex":[0.9968696,0.00009636067,0.0007632371,0.001158642,0.0003012062,0.0008109344],"domain_scores_gemma":[0.997475,0.0002993918,0.0002958317,0.00103102,0.0007765898,0.0001221773],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0004175583,0.0002995071,0.00006348812,0.00104655,0.0001998639,0.00001563446,0.0005944185,0.004395062,0.1428762,0.02734252,0.002541256,0.8202079],"study_design_scores_gemma":[0.0005334088,0.0003239163,0.00001468853,0.0001660546,0.00003812018,0.00002925628,0.0001834031,0.3235865,0.6263538,0.01657336,0.03180648,0.0003910474],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002951422,0.0005806904,0.9882992,0.0002591752,0.002355622,0.00182789,0.00001949002,0.001064829,0.002641703],"genre_scores_gemma":[0.7434044,0.00008704387,0.2474147,0.000654885,0.0001064336,0.0002335904,0.000001438237,0.00003356328,0.008063869],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8198168,"threshold_uncertainty_score":0.9997324,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04103805696767971,"score_gpt":0.3439023649920591,"score_spread":0.3028643080243794,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}