@inproceedings{62bafcbeda6842f594174c7e0b559772,
title = "Debugging Malware Classification Models Based on Event Logs with Explainable AI",
abstract = "As machine learning models find broader applications in cybersecurity, the importance of model explainability becomes more evident. In the area of malware detection, where the consequence of misclassification can be severe, explainability becomes crucial. AI explainers not only help understand the reasons behind malware classifications but also assist in finetuning models to improve detection accuracy. Additionally, AI explainers can serve as a valuable tool for error detection, ensuring accountability, and mitigating potential biases. In this paper, we demonstrate how AI explainers can play a vital role in identifying issues in data collection and enhancing our comprehension of the model's classification results. Our analysis of explanation results reveals several issues within the data collection process, including event loss and the presence of environment-specific information. Additionally, we have identified mislabelled samples based on the explanation results and shared lessons learned from our data collection efforts.",
keywords = "ETW, Random Forest, SHAP, TreeSHAP, XAI, explainable AI, malware detection",
author = "Gwak, \{Joon Young\} and Priti Wakodikar and Meng Wang and Guanhua Yan and Xiaokui Shu and Stoller, \{Scott D.\} and Ping Yang",
note = "Publisher Copyright: {\textcopyright} 2023 IEEE.; 23rd IEEE International Conference on Data Mining Workshops, ICDMW 2023 ; Conference date: 01-12-2023 Through 04-12-2023",
year = "2023",
doi = "10.1109/ICDMW60847.2023.00125",
language = "English",
series = "IEEE International Conference on Data Mining Workshops, ICDMW",
publisher = "IEEE Computer Society",
pages = "939--948",
editor = "Jihe Wang and Yi He and Dinh, \{Thang N.\} and Christan Grant and Meikang Qiu and Witold Pedrycz",
booktitle = "Proceedings - 23rd IEEE International Conference on Data Mining Workshops, ICDMW 2023",
}