{"id":{"repo_id":"wfu","oai_identifier":"oai:wakespace.lib.wfu.edu:10339/14824"},"canonical_url":"https://search.dev.ndltd.org/etd/wfu/oai:wakespace.lib.wfu.edu:10339/14824","repository":{"repo_id":"wfu","name":"Wake Forest University","base_url":"https://wakespace.lib.wfu.edu/oai/request"},"display":{"title":"Predicting Hard Drive Failures in Computer Clusters","abstract":"Mitigating the impact of computer failure is possible if accurate failure predictions are provided. Resources, and services can be scheduled around predicted failure and limit the impact. Such strategies are especially important for multi-computer systems, such as compute clusters, that experience a higher rate of failure due to the large number of components. However providing accurate predictions with sufficient lead time remains a challenging problem. This research uses a new spectrum-kernel Support Vector Machine (SVM) ap- proach to predict failure events based on system log files. These files contain mes- sages that represent a change of system state. While a single message in the file may not be sufficient for predicting failure, a sequence or pattern of messages may be. This approach uses a sliding window (sub-sequence) of messages to predict the likelihood of failure. Then, a frequency representation of the message sub-sequences observed are used as input to the SVM. The SVM associates the messages to a class of failed or non-failed system. Experimental results using actual system log files from a Linux-based compute cluster indicate the proposed spectrum-kernel SVM approach can predict hard disk failure with an accuracy of 80% about one day in advance.","abstract_html":"Mitigating the impact of computer failure is possible if accurate failure predictions are provided. Resources, and services can be scheduled around predicted failure and limit the impact. Such strategies are especially important for multi-computer systems, such as compute clusters, that experience a higher rate of failure due to the large number of components. However providing accurate predictions with sufficient lead time remains a challenging problem. This research uses a new spectrum-kernel Support Vector Machine (SVM) ap- proach to predict failure events based on system log files. These files contain mes- sages that represent a change of system state. While a single message in the file may not be sufficient for predicting failure, a sequence or pattern of messages may be. This approach uses a sliding window (sub-sequence) of messages to predict the likelihood of failure. Then, a frequency representation of the message sub-sequences observed are used as input to the SVM. The SVM associates the messages to a class of failed or non-failed system. Experimental results using actual system log files from a Linux-based compute cluster indicate the proposed spectrum-kernel SVM approach can predict hard disk failure with an accuracy of 80% about one day in advance.","abstract_has_math":false,"creators":["Featherstun, Robin Wesley"],"institution":"Wake Forest University","degree_name":null,"degree_level":null,"degree_discipline":null,"degree_department":null,"school":null,"contributors":[],"advisors":[],"committee_chairs":[],"committee_members":[],"year":2010,"date_issued":"2010-05-07T18:30:15Z","date_published":"2010-05-07T18:30:15Z","updated_at":"2026-07-27T22:01:01Z","subjects":["computer science"],"languages":["en_US"],"rights":[],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"http://hdl.handle.net/10339/14824","outbound_label":"Handle","outbound_source":"dc:identifier.uri"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:creator","label":"Author","values":["Featherstun, Robin Wesley"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date.accessioned","label":"Dc Date Accessioned","values":["2010-05-07T18:30:15Z"]},{"key":"dc:date.available","label":"Dc Date Available","values":["2010-05-07T18:30:15Z"]},{"key":"dc:date.issued","label":"Date","values":["2010-05-07T18:30:15Z"]},{"key":"dc:publisher","label":"Institution","values":["Wake Forest University"]},{"key":"dc:type","label":"Dc Type","values":["Thesis"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["computer science"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language.iso","label":"Language (ISO)","values":["en_US"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier.uri","label":"Identifier URI","values":["http://hdl.handle.net/10339/14824"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description.abstract","label":"Abstract","values":["Mitigating the impact of computer failure is possible if accurate failure predictions are provided. Resources, and services can be scheduled around predicted failure and limit the impact. Such strategies are especially important for multi-computer systems, such as compute clusters, that experience a higher rate of failure due to the large number of components. However providing accurate predictions with sufficient lead time remains a challenging problem. This research uses a new spectrum-kernel Support Vector Machine (SVM) ap- proach to predict failure events based on system log files. These files contain mes- sages that represent a change of system state. While a single message in the file may not be sufficient for predicting failure, a sequence or pattern of messages may be. This approach uses a sliding window (sub-sequence) of messages to predict the likelihood of failure. Then, a frequency representation of the message sub-sequences observed are used as input to the SVM. The SVM associates the messages to a class of failed or non-failed system. Experimental results using actual system log files from a Linux-based compute cluster indicate the proposed spectrum-kernel SVM approach can predict hard disk failure with an accuracy of 80% about one day in advance."]},{"key":"dc:title","label":"Title","values":["Predicting Hard Drive Failures in Computer Clusters"]}]}],"canonical_facts":{"dc:creator":["Featherstun, Robin Wesley"],"dc:date.accessioned":["2010-05-07T18:30:15Z"],"dc:date.available":["2010-05-07T18:30:15Z"],"dc:date.issued":["2010-05-07T18:30:15Z"],"dc:description.abstract":["Mitigating the impact of computer failure is possible if accurate failure predictions are provided. Resources, and services can be scheduled around predicted failure and limit the impact. Such strategies are especially important for multi-computer systems, such as compute clusters, that experience a higher rate of failure due to the large number of components. However providing accurate predictions with sufficient lead time remains a challenging problem. This research uses a new spectrum-kernel Support Vector Machine (SVM) ap- proach to predict failure events based on system log files. These files contain mes- sages that represent a change of system state. While a single message in the file may not be sufficient for predicting failure, a sequence or pattern of messages may be. This approach uses a sliding window (sub-sequence) of messages to predict the likelihood of failure. Then, a frequency representation of the message sub-sequences observed are used as input to the SVM. The SVM associates the messages to a class of failed or non-failed system. Experimental results using actual system log files from a Linux-based compute cluster indicate the proposed spectrum-kernel SVM approach can predict hard disk failure with an accuracy of 80% about one day in advance."],"dc:identifier.uri":["http://hdl.handle.net/10339/14824"],"dc:language.iso":["en_US"],"dc:publisher":["Wake Forest University"],"dc:subject":["computer science"],"dc:title":["Predicting Hard Drive Failures in Computer Clusters"],"dc:type":["Thesis"]},"updated_at":"2026-07-27T22:01:01Z"}