@inproceedings{f01f504cc79b4b07bd59c77b5077dece,
title = "Tracking System Behavior from Resource Usage Data",
abstract = "Resource usage data, collected using tools such as TACC-Stats, capture the resource utilization by nodes within a high performance computing system. We present methods to analyze the resource usage data to understand the system performance and identify performance anomalies. The core idea is to model the data as a three-way tensor corresponding to the compute nodes, usage metrics, and time. Using the reconstruction error between the original tensor and the tensor reconstructed from a low rank tensor decomposition, as a scalar performance metric, enables us to monitor the performance of the system in an online fashion. This error statistic is then used for anomaly detection that relies on the assumption that the normal/routine behavior of the system can be captured using a low rank approximation of the original tensor. We evaluate the performance of the algorithm using information gathered from system logs and show that the performance anomalies identified by the proposed method correlates with critical errors reported in the system logs. Results are shown for data collected for 2013 from the Lonestar4 system at the Texas Advanced Computing Center (TACC).",
keywords = "Anomaly detection, Feature Extraction, HPC, Performance monitoring, Tensor Analysis",
author = "Niyazi Sorkunlu and Varun Chandola and Abani Patra",
note = "Publisher Copyright: {\textcopyright} 2017 IEEE.; 2017 IEEE International Conference on Cluster Computing, CLUSTER 2017 ; Conference date: 05-09-2017 Through 08-09-2017",
year = "2017",
month = sep,
day = "22",
doi = "10.1109/CLUSTER.2017.70",
language = "English",
series = "Proceedings - IEEE International Conference on Cluster Computing, ICCC",
publisher = "Institute of Electrical and Electronics Engineers Inc.",
pages = "410--418",
booktitle = "Proceedings - 2017 IEEE International Conference on Cluster Computing, CLUSTER 2017",
address = "United States",
}