<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Comput. Sci.</journal-id>
<journal-title>Frontiers in Computer Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Comput. Sci.</abbrev-journal-title>
<issn pub-type="epub">2624-9898</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fcomp.2024.1400943</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Computer Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A survivability analysis of enterprise hard drives incorporating the impact of workload</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Mallik</surname> <given-names>Aman</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2627018/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Reddy</surname> <given-names>B Ranjith</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2804795/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Sahoo</surname> <given-names>Gadadhar</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2182029/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Computer Science, BITs Pilani Hyderabad Campus</institution>, <addr-line>Hyderabad</addr-line>, <country>India</country></aff>
<aff id="aff2"><sup>2</sup><institution>Msrit (MS Ramaiah Institute of Technology)</institution>, <addr-line>Bangalore</addr-line>, <country>India</country></aff>
<aff id="aff3"><sup>3</sup><institution>Department of Computer Science and Engineering, IIT (ISM) Dhanbad</institution>, <addr-line>Dhanbad</addr-line>, <country>India</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0001">
<p>Edited by: Nicola Zannone, Eindhoven University of Technology, Netherlands</p>
</fn>
<fn fn-type="edited-by" id="fn0002">
<p>Reviewed by: Antoine Rauzy, NTNU, Norway</p>
<p>Krishna Kumar Mohbey, Central University of Rajasthan, India</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Aman Mallik, <email>amanmallik01@gmail.com</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>04</day>
<month>10</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>6</volume>
<elocation-id>1400943</elocation-id>
<history>
<date date-type="received">
<day>14</day>
<month>03</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>09</day>
<month>09</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2024 Mallik, Reddy and Sahoo.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Mallik, Reddy and Sahoo</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec id="sec1001">
<title>Introduction</title>
<p>Hard disk drive (HDD) failure is a significant cause of downtime in enterprise storage systems. Research suggests that data access rates strongly influence the survival probability of HDDs.</p>
</sec>
<sec id="sec2001">
<title>Methods</title>
<p>This paper proposes a model to estimate the probability of HDD failure, using factors such as the total data (TD) read or written and the average access rate (AAR) for a specific drive model. The study utilizes a dataset of HDD failures to analyze the effects of these variables.</p>
</sec>
<sec id="sec3001">
<title>Results</title>
<p>The model was validated using case studies, demonstrating a strong correlation between access rate management and reduced HDD failure risk. The results indicate that managing data access rates through improved throttle commands can significantly enhance drive reliability.</p>
</sec>
<sec id="sec4001">
<title>Discussion</title>
<p>Our approach suggests that optimizing throttle commands at the storage controller level can help mitigate the risk of HDD failure by controlling data access rates, thereby improving system longevity and reducing downtime in enterprise storage systems.</p>
</sec>
</abstract>
<kwd-group>
<kwd>cox proportional hazards model</kwd>
<kwd>storage system risk reduction</kwd>
<kwd>HDD failure prediction</kwd>
<kwd>survivability probability</kwd>
<kwd>enterprise storage systems</kwd>
</kwd-group>
<counts>
<fig-count count="17"/>
<table-count count="3"/>
<equation-count count="10"/>
<ref-count count="22"/>
<page-count count="13"/>
<word-count count="6863"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computer Security</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<label>1</label>
<title>Introduction</title>
<p>The fundamental role of any storage system is to efficiently cater to the diverse needs of various applications while accommodating their specific workloads. In modern storage systems, data reading and writing operations are primarily handled by hard disk drives (HDDs). Any interruption of these basic tasks can have significant repercussions, affecting all facets of storage management. Such disruptions can lead to performance degradation, an increased need for human intervention, a higher risk of service outages, and ultimately, potential data unavailability or loss.</p>
<p>Despite the critical importance of seamless data operations, many storage systems face challenges in maintaining uninterrupted performance. Existing solutions often fall short in addressing the complexity and variability of workloads, reducing the efficiency of data management. Existing research has focused on optimizing the individual components of storage systems, but comprehensive strategies that encompass the entire storage infrastructure are still lacking.</p>
<p>This research aims to bridge these gaps by exploring innovative approaches to enhancing the reliability and efficiency of storage systems. By investigating the underlying causes of data operation interruptions and their impact on overall system performance, this study seeks to develop robust solutions that minimize disruptions, reduce human intervention, and ensure continuous data availability. Through this work, we aim to contribute to the advancement of storage technologies, paving the way for more resilient and efficient data management systems.</p>
<sec id="sec2">
<label>1.1</label>
<title>Motivation</title>
<p>The motivation for this paper stems from the critical issue of HDD failure causing significant downtime in enterprise storage systems. Despite advancements in storage technology, HDD failures remain a persistent challenge, leading to substantial disruptions and operational inefficiencies. The existing literature provides valuable insights into individual factors that contribute to HDD failures, but there is a lack of comprehensive models that combine these factors to predict and mitigate such failures effectively.</p>
<p>In this work, empirical analyses are conducted to explore the relationship between data access rates and HDD survivability. By proposing a predictive model based on the total data (TD) read or written and the average access rate (AAR), this work aims to preemptively address HDD failures. The predictive model leverages real-world data from storage controllers to identify HDDs with high failure probabilities.</p>
<p>To validate the findings, the proposed model is tested using data from actual storage systems. A novel strategy is introduced: reallocating HDDs with high failure probabilities to different redundancy groups. This approach aims to optimize resource allocation, enhance system resilience, and mitigate the risks associated with HDD failures.</p>
<p>By addressing these challenges, this research provides a practical solution for storage system administrators and engineers and a proactive method to improve the reliability and efficiency of enterprise storage systems.</p>
</sec>
<sec id="sec3">
<label>1.2</label>
<title>Previous research</title>
<p>HDDs consist of many complex subcomponents that must work in coordination with each other. Depending on the characteristics of the subcomponents, failures can occur at different stages in the lifetime of a product. The mean time to failure (MTTF) and annualized failure rate (AFR) are two of the current metrics of choice for quantifying the survivability of HDDs (<xref ref-type="bibr" rid="ref2">George, 2013</xref>).</p>
<p>Prior research has shown that HDD failure prediction modeling can provide reasonable failure predictions for different kinds of hard disks with various interfaces, including integrated drive electronics (IDE), fiber channels (FC), small computer system interfaces (SCSI), and serial advanced technology attachments (SATA). Statistical modeling techniques such as logistic regression have been applied using the most relevant self-monitoring, analysis, and reporting technology (SMART) parameters to predict HDD failures with reasonable false alarm rates and accuracy (<xref ref-type="bibr" rid="ref17">Shen et al., 2018</xref>; <xref ref-type="bibr" rid="ref22">Zhang et al., 2023</xref>; <xref ref-type="bibr" rid="ref15">Rinc&#x00F3;n et al., 2017</xref>; <xref ref-type="bibr" rid="ref10">Liu and Xing, 2020</xref>; <xref ref-type="bibr" rid="ref19">Smith and Smith, 2004</xref>; <xref ref-type="bibr" rid="ref18">Smith and Smith, 2001</xref>; <xref ref-type="bibr" rid="ref20">Smith and Smith, 2005</xref>; <xref ref-type="bibr" rid="ref11">Mohanta and Ananthamurthy, 2006</xref>).</p>
<p>There have been attempts to increase the survival probability of solid-state driFDriveves (SSDs) in storage systems by controlling the write amplification factor (WAF) depending on the workload (<xref ref-type="bibr" rid="ref12">Mohanta et al., 2015</xref>). Different machine-learning solutions have also been proposed as alternatives, including support vector machines, nonparametric rank-sum tests, and unsupervised clustering algorithms. These solutions provide improvements over existing threshold-based algorithms for predicting HDD failures (<xref ref-type="bibr" rid="ref13">Murray et al., 2003</xref>; <xref ref-type="bibr" rid="ref3">Hamerly and Elkan, 2001</xref>; <xref ref-type="bibr" rid="ref16">Royston and Sauerbrei, 2008</xref>).</p>
<p>Improvements and new trends in HDD packaging methods have led to the high-density packing of physical materials in the drive cage. This has led to less spacing between the head and the media during read and write operations, which is one of the probable causes of HDD failure with media errors. The MTTF parameter alone is inadequate to describe HDD survivability, and according to one study (<xref ref-type="bibr" rid="ref2">George, 2013</xref>), the total amount of data transferred is a more appropriate parameter of choice.</p>
<p>Many large enterprise storage companies require that drive manufacturers to provide detailed information to be queried from HDDs for further failure analysis. Along with the standard parameters, such as SMART attributes, recoverable errors, and unrecoverable errors parameter values are also made available to query from HDDs. This research employs these data for predictive modelling and analyses.</p>
</sec>
<sec id="sec4">
<label>1.3</label>
<title>Unique contributions</title>
<p>Different kinds of hard disks form components of the storage systems deployed in data centers and cloud environments, including private, public, and hybrid clouds, to provide platform-as-a-service (PaaS) technology. In storage systems, these hard disks are building blocks for offering proper performance and survivability to the hosted application. Disk failure is inevitable, and catastrophic errors threaten both mission-critical data and the performance and survivability of applications.</p>
<p>The novel application of the model proposed in this work is that multiple HDDs in the same storage pool with a high failure probability threshold are reallocated to different storage pools after ensuring proper migration of their data to enhance overall system survivability.</p>
<p>The following are unique contributions of this work that advance the understanding of hard drive failures and their prediction:</p>
<list list-type="bullet">
<list-item><p><bold>Predictive Modeling:</bold> By leveraging the statistical programming language R, the proposed predictive model has the capability to analyze workload patterns and their correlation with hard drive failures. By training models on extensive datasets that incorporate workload parameters, these algorithms can forecast potential failures more accurately.</p></list-item>
<list-item><p><bold>Dynamic Adjustments:</bold> The same model can be pragmatically implemented for developing real-time monitoring tools that continuously assess workload parameters and dynamically adjust hard drive operations. This proactive approach allows systems to mitigate potential stressors or redistribute the workload to ensure optimal drive health.</p></list-item>
<list-item><p><bold>End-to-End Assessment:</bold> This model has the potential to perform a comprehensive lifecycle analysis that considers the cumulative impact of workload variations over the entire lifespan of a hard drive. The appropriately chosen statistical model, namely, the Cox Proportional Hazards Model (CPHM) utilizes censored data to provide with valid estimates of the survival probabilities for potential HHD failures throughout the entire life cycle. This holistic approach helps in identifying critical periods or thresholds beyond which workload intensities significantly affect survivability.</p></list-item>
<list-item><p><bold>Smart Resource Allocation</bold>: This model unveils avenues for ushering in adaptive throttling mechanisms that optimize resource allocation based on workload analysis. By dynamically adjusting read/write operations, applying caching strategies, and making use of data placement, systems can reduce wear and tear, thereby enhancing hard drive longevity.</p></list-item>
<list-item><p><bold>Unified Metrics</bold>: This model facilitates the use of proposing standardized metrics that incorporate workload considerations based on the AAR and TD. By establishing a common framework for evaluating hard drive survivability across diverse workloads, researchers and industry practitioners can compare results more effectively and drive advancements collaboratively.</p></list-item>
</list>
</sec>
<sec id="sec5">
<label>1.4</label>
<title>Mathematical formulations and equations</title>
<p>Storage Array Downtime (SD) is defined as the percentage of time that the storage arrays in the installed base are down and not available to service requests. Measuring the storage array downtime is useful for clearly visualizing different contributions to the total array downtime (such as firmware-related downtime and downtime with other causes). However, the proposed research has adopted another approach is adopted in this work to investigate the impact of workload on HDD failures. As part of this analysis, primary data was collected from real production systems. The total amount of data read from or written to HDDs is a major parameter of this analysis. It is called the total data (TD) and is defined as follows:</p>
<disp-formula id="E1">
<label>(1)</label>
<mml:math id="M1">
<mml:mi>T</mml:mi>
<mml:mi>D</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>D</mml:mi>
<mml:mi>R</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>D</mml:mi>
<mml:mi>W</mml:mi>
</mml:math>
</disp-formula>
<p>where TDR is the total data read and TDW is the total data written in bytes.</p>
<p>Attempts have been made to determine how different kinds of workloads affect the survivability of HDDs. However, information pertaining to the amount of random data or sequential data transferred to the disks is not available in this dataset.</p>
<p>Therefore, another parameter, the AAR, is exploited in this investigation. The values of this parameter May vary over time. There could be significant activity at certain times and a lack of activity at other times, depending on application requirements. Keeping in mind the lack of information on the timelines of these cycles in view, a parameter called the average access rate (AAR) is introduced, which is defined as follows:</p>
<disp-formula id="E2">
<label>(2)</label>
<mml:math id="M2">
<mml:mi>A</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>R</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:math>
</disp-formula>
<p>where, PT is the power-on time in minutes. This parameter intuitively gives a sense of the data transfer rate of a customer application to the HDD. In this analysis, the TD in bytes and the AAR in bytes per minute are used as parameters to measure the survivability of the HDDs.</p>
</sec>
</sec>
<sec id="sec6">
<label>2</label>
<title>Brief overview of survival analysis</title>
<p>Survival analytic models, which form a branch of statistics, share certain similarities with logistic and linear models. In this analysis, survival analytic models are chosen over logistic or linear models because they consider parameters such as the event time and event probability that are not considered by the alternatives (<xref ref-type="bibr" rid="ref5">Kalbfleisch and Prentice, 2011</xref>; <xref ref-type="bibr" rid="ref8">Lambert and Royston, 2009</xref>). The key parameters and associated functions are outlined in this section.</p>
<p>Usually, failed HDDs are returned by different customers at different times. This research takes into account HDDs received over the past 5 years. A detailed analysis was carried out based on the power-on time of the HDDs as well as their workloads. The power-on time is the amount of time for which the HDDs are used in any system. The start time for the analysis on all HDDs is the same, i.e., 0. The end time is the highest value of the power-on time available in the data set. A test suite determines whether one or more events have occurred on an HDD that belongs to a set of failure events. If such an event occurs, the amount of power-on time on that HDD is considered its failure event time.</p>
<p>In survival analysis, the event time distribution is quantified using the following four functions.</p>
<sec id="sec7">
<label>2.1</label>
<title>The cumulative distribution function (CDF)</title>
<p>The CDF of hard disk random failure is expressed using a random variable &#x1D44B; as</p>
<disp-formula id="E3">
<label>(3)</label>
<mml:math id="M3">
<mml:mi>F</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mi>x</mml:mi>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mi>P</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x003C;</mml:mo>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<p>where, <inline-formula>
<mml:math id="M4">
<mml:mi>F</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mi>x</mml:mi>
</mml:mfenced>
</mml:math>
</inline-formula> is the CDF and the right-hand side represents the probability that &#x1D44B; has a value less than or equal to &#x1D465;.</p>
</sec>
<sec id="sec8">
<label>2.2</label>
<title>The probability density function (PDF)</title>
<p>The PDF of a random variable &#x1D44B;, denoted &#x1D453; (&#x1D465;), is defined by:</p>
<disp-formula id="E4">
<label>(4)</label>
<mml:math id="M5">
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mi>x</mml:mi>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mspace width="thickmathspace"/>
<mml:mi>F</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mi>x</mml:mi>
</mml:mfenced>
</mml:math>
</disp-formula>
<p>The PDF is the derivative or slope of the cumulative distribution function (<xref ref-type="bibr" rid="ref22">Zhang et al., 2023</xref>).</p>
</sec>
<sec id="sec9">
<label>2.3</label>
<title>The survival function</title>
<p>The survival function captures the probability that a component or system is alive and functional beyond a defined point on the time axis (<xref ref-type="bibr" rid="ref15">Rinc&#x00F3;n et al., 2017</xref>). Let &#x1D44B; be a continuous random variable with cumulative distribution function &#x1D439;(&#x1D465;) in the interval [0, &#x221E;]. Its survival function is defined as follows:</p>
<disp-formula id="E5">
<label>(5)</label>
<mml:math id="M6">
<mml:mi>S</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mi>x</mml:mi>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mi>P</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mfenced open="{" close="}">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x003E;</mml:mo>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mi>x</mml:mi>
</mml:mfenced>
</mml:math>
</disp-formula>
<p>The above function helps to define the probability of &#x1D44B; being alive just before exceeding duration &#x1D465;, or the probability that the failure event has not occurred at all in the considered &#x1D465; interval. The survival curve describes the relationship between the probability of survival and time.</p>
</sec>
<sec id="sec10">
<label>2.4</label>
<title>Hazard function</title>
<p>The hazard function &#x210E;(&#x1D465;) is given by the following equation:</p>
<disp-formula id="E6">
<label>(6)</label>
<mml:math id="M7">
<mml:mi>h</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mi>x</mml:mi>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mi>x</mml:mi>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mi>x</mml:mi>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
</mml:math>
</disp-formula>
<p>The hazard function is more intuitive in survival analysis than the probability distribution function because it quantifies the instantaneous risk that an event will take place at time &#x1D465; given that the subject survived to time &#x1D465; [6].</p>
<p>In survival or failure test cases, it is essential to determine whether variables are correlated with survival or failure times. However, this correlation analysis is not simple (<xref ref-type="bibr" rid="ref16">Royston and Sauerbrei, 2008</xref>; <xref ref-type="bibr" rid="ref5">Kalbfleisch and Prentice, 2011</xref>; <xref ref-type="bibr" rid="ref8">Lambert and Royston, 2009</xref>). This is because, most likely, the dependent variable of interest does not follow an exponential distribution but instead a normal distribution. Another contributing factor is incomplete datasets generated from research or analytical studies, where the complete output is not as expected or the dataset is censored.</p>
</sec>
<sec id="sec11">
<label>2.5</label>
<title>The cox proportional hazards model (CPHM)</title>
<p>The CPHM relies on variables that are correlated to survival and does not make presumptions about base line hazard rate of each variable. Therefore, the Cox regression method is much more useful than the Kaplan&#x2013;Meier estimator (KME) approach, which involves a lot of explanatory variables (<xref ref-type="bibr" rid="ref7">Kleinbaum and Klein, 1996</xref>; <xref ref-type="bibr" rid="ref4">Hosmer et al., 2008</xref>; <xref ref-type="bibr" rid="ref1">David, 1972</xref>). A brief exposition on KME is given below to validate the choice of the CPHM for the proposed research.</p>
<p>In its generic form, the KME can be expressed as below using an example.</p>
<p>Consider a sample size of population &#x1D441;, with &#x1D461; being the time axis. Assume &#x1D461;1, &#x1D461;2, &#x2026; &#x1D461;i, &#x2026; &#x1D461;&#x1D441; are the observed lifetimes of the sample size &#x1D441;, such that &#x1D461;1 &#x2264; &#x1D461;2 &#x2264; &#x1D461;3&#x2009;&#x2264;&#x2009;&#x2026; &#x2264; &#x1D461;&#x1D441;, with &#x015C;(&#x1D461;) the probability that a member has a lifetime exceeding &#x1D461;. In this scenario, the Kaplan&#x2013;Meier estimator tries to establish a survival function at &#x1D461;i between members who have experienced the event versus members who have not. Let &#x1D451;&#x1D456; be the set of members who have experienced the event and &#x1D45B;&#x1D456; be the members who are yet to experience the event. Then, the KME can be expressed as:</p>
<disp-formula id="E7">
<label>(7)</label>
<mml:math id="M8">
<mml:mi>S</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>&#x03A0;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.1em"/>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<p>Assuming, &#x015C;(&#x1D461;) is the probability that a given member from the sample size has a lifetime exceeding time &#x1D461;.</p>
<p>As no assumptions are made about the nature of the survival distribution, the CPHM model is considered the most generic regression model. Hence, Cox&#x2019;s regression model May be considered to be a &#x201C;<italic>semi-parametric&#x201D;</italic> model (<xref ref-type="bibr" rid="ref5">Kalbfleisch and Prentice, 2011</xref>). The CPHM is represented as follows:</p>
<disp-formula id="E8">
<label>(8)</label>
<mml:math id="M9">
<mml:mi>h</mml:mi>
<mml:mfenced open="(" close=")" separators=",">
<mml:mi>t</mml:mi>
<mml:mi>x</mml:mi>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>exp</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x03A3;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mspace width="0.1em"/>
<mml:msub>
<mml:mi>&#x03B2;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<p>where, &#x1D44B; = (&#x1D44B;1, &#x1D44B;2, &#x2026;, &#x1D44B;&#x1D45D;) are the explanatory/predictor variables, &#x210E;0(&#x1D461;) is the baseline hazard function, and &#x1D6FD;i are the regression coefficients.</p>
<p>To linearize this model, both sides of the equation are divided by &#x210E;0(&#x1D461;) and then the natural logarithm is then taken on both sides. As a result, a fairly &#x201C;simple&#x201D; linear model can be readily estimated. The final result is:</p>
<disp-formula id="E9">
<label>(9)</label>
<mml:math id="M10">
<mml:mo>log</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mfrac>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mfenced open="(" close=")" separators=",">
<mml:mi>t</mml:mi>
<mml:mi>x</mml:mi>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mi>&#x03A3;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mspace width="0.1em"/>
<mml:msub>
<mml:mi>&#x03B2;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:math>
</disp-formula>
<p>The CPHM has been endorsed by many researchers as it is quite robust for computing the associated survival probabilities through balancing potential predominant variables (<xref ref-type="bibr" rid="ref16">Royston and Sauerbrei, 2008</xref>; <xref ref-type="bibr" rid="ref5">Kalbfleisch and Prentice, 2011</xref>; <xref ref-type="bibr" rid="ref8">Lambert and Royston, 2009</xref>; <xref ref-type="bibr" rid="ref7">Kleinbaum and Klein, 1996</xref>; <xref ref-type="bibr" rid="ref4">Hosmer et al., 2008</xref>). In essence, the CPHM is most suitable for representing and interpreting impact of AAR and TD groups on survival probability of HDD. Also statistically R programming language R is also commensurate tool implementing CPHM for the proposed research (<xref ref-type="bibr" rid="ref7">Kleinbaum and Klein, 1996</xref>; <xref ref-type="bibr" rid="ref4">Hosmer et al., 2008</xref>; <xref ref-type="bibr" rid="ref1">David, 1972</xref>; <xref ref-type="bibr" rid="ref6">Kaplan and Meier, 1958</xref>; <xref ref-type="bibr" rid="ref9">Lane et al., 1986</xref>).</p>
</sec>
</sec>
<sec id="sec12">
<label>3</label>
<title>Modeling and discussions</title>
<p>As the HDDs considered here form part of a storage system, software drives the operation of these hard disks from insertion into the storage system until the drives declare themselves to have failed or software identifies them as faulty (<xref ref-type="disp-formula" rid="E1">Equations 1</xref>&#x2013;<xref ref-type="disp-formula" rid="E9">9</xref>).</p>
<sec id="sec13">
<label>3.1</label>
<title>HDD failure</title>
<p>HDD failure occurs when an HDD is no longer capable of performing IO operations. However, the word HDD failure is somewhat vague, and the threshold for failure is different for each subsystem.</p>
<p>SMART data in its intrinsic form seems to be insufficient because many HDDs undergo unexpected hardware failures, such as head crashes or excessive media errors, without any warning from SMART subsystems. Storage controllers May have different thresholds for declaring an HDD as having failed, such as if a drive gives a certain number of unrecovered errors within a certain period, shows too many timeouts during data access, or cannot reassign a logical block address (LBA).</p>
<p>Therefore, the basis of the present analysis is the event of HDD failure in a storage system. Whenever there is an HDD failure event in a customer data center, the failed HDD is sent to a lab for investigation. The test suites used by storage system vendors mostly rely on the log data available on the HDD, which can be queried through Small Computer System Interface (SCSI) commands. These test suites have defined thresholds, and if a drive exceeds the thresholds in any certain area, it is declared a failure. There is a never-ending battle between storage controller providers and HDD manufacturers concerning the definition of failure because the manufacturer has to bear the cost if the HDD is still within a warranty period.</p>
<p>Typically, storage systems support multiple HDD models from different manufacturers. Therefore, the gross or initial drive sample considered for this analysis includes different drive models. Out of the various HDD models present in the gross sample, twenty drive models are chosen for this investigation. Historical data collected for these twenty drive models is analyzed and plotted to determine the failure pattern and its dependence on certain parameters. In-depth results are provided for only two drive models, as the other drive models are similar.</p>
</sec>
<sec id="sec14">
<label>3.2</label>
<title>Data description</title>
<p>Data from two hard disk drive models (the M1-A1 model and the M2-A1 model) were analyzed using R. <xref ref-type="table" rid="tab1">Table 1</xref> represents a sample raw data for model M1-A1(HDD Manufacture&#x2019;s Model Name - NB1000D4450), as given below.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Sample raw data (The highlighted yellow columns represent the data used in the current analysis) of HDD model.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Name</th>
<th align="center" valign="top">HDD Model</th>
<th align="center" valign="top">FW</th>
<th align="center" valign="top">Write Rec Err</th>
<th align="center" valign="top">WriteBytes</th>
<th align="center" valign="top">Read Rec Err</th>
<th align="center" valign="top">ReadBytes</th>
<th align="center" valign="top">Power on time</th>
<th align="center" valign="top">SSUITE status</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">6001r.txt</td>
<td align="center" valign="middle">NB1000D4450</td>
<td align="center" valign="middle">XX04</td>
<td align="center" valign="middle">0&#x00D7;0000</td>
<td align="center" valign="middle">0x0f60381db400</td>
<td align="center" valign="middle">0x685eba9c</td>
<td align="center" valign="middle">0x2e739518ec00</td>
<td align="center" valign="middle">640,975</td>
<td align="center" valign="middle">0</td>
</tr>
<tr>
<td align="left" valign="middle">6002r.txt</td>
<td align="center" valign="middle">NB1000D4450</td>
<td align="center" valign="middle">XX06</td>
<td align="center" valign="middle">0&#x00D7;0000</td>
<td align="center" valign="middle">0x00ced22e8800</td>
<td align="center" valign="middle">0x36257e</td>
<td align="center" valign="middle">0x0061ba630800</td>
<td align="center" valign="middle">10,030</td>
<td align="center" valign="middle">0</td>
</tr>
<tr>
<td align="left" valign="middle">6005r.txt</td>
<td align="center" valign="middle">NB1000D4450</td>
<td align="center" valign="middle">XX06</td>
<td align="center" valign="middle">0&#x00D7;0000</td>
<td align="center" valign="middle">0x04d085cf5200</td>
<td align="center" valign="middle">0x82bbf6</td>
<td align="center" valign="middle">0x004f4faa9c00</td>
<td align="center" valign="middle">626,385</td>
<td align="center" valign="middle">0</td>
</tr>
<tr>
<td align="left" valign="middle">6007r.txt</td>
<td align="center" valign="middle">NB1000D4450</td>
<td align="center" valign="middle">XX06</td>
<td align="center" valign="middle">0&#x00D7;0000</td>
<td align="center" valign="middle">0&#x00D7;000032800</td>
<td align="center" valign="middle">0&#x00D7;0035</td>
<td align="center" valign="middle">0x0000203e00</td>
<td align="center" valign="middle">11</td>
<td align="center" valign="middle">1</td>
</tr>
<tr>
<td align="left" valign="middle">6007r.txt</td>
<td align="center" valign="middle">NB1000D4450</td>
<td align="center" valign="middle">XX06</td>
<td align="center" valign="middle">0&#x00D7;0000</td>
<td align="center" valign="middle">0x01bd86913c00</td>
<td align="center" valign="middle">0&#x00D7;2284569</td>
<td align="center" valign="middle">0x00682858ca00</td>
<td align="center" valign="middle">403,928</td>
<td align="center" valign="middle">0</td>
</tr>
<tr>
<td align="left" valign="middle">6009r.txt</td>
<td align="center" valign="middle">NB1000D4450</td>
<td align="center" valign="middle">XX06</td>
<td align="center" valign="middle">0&#x00D7;0000</td>
<td align="center" valign="middle">0x02b9bb3ab000</td>
<td align="center" valign="middle">0x171ce42f</td>
<td align="center" valign="middle">0x11ce58d46400</td>
<td align="center" valign="middle">225,990</td>
<td align="center" valign="middle">0</td>
</tr>
<tr>
<td align="left" valign="middle">6010r.txt</td>
<td align="center" valign="middle">NB1000D4450</td>
<td align="center" valign="middle">XX06</td>
<td align="center" valign="middle">0&#x00D7;0000</td>
<td align="center" valign="middle">0x05415ac38800</td>
<td align="center" valign="middle">0x3e9a2c69</td>
<td align="center" valign="middle">0x0930f8101800</td>
<td align="center" valign="middle">233,520</td>
<td align="center" valign="middle">0</td>
</tr>
<tr>
<td align="left" valign="middle">6010r.txt</td>
<td align="center" valign="middle">NB1000D4450</td>
<td align="center" valign="middle">XX05</td>
<td align="center" valign="middle">0&#x00D7;0000</td>
<td align="center" valign="middle">0x03b3d4f68e00</td>
<td align="center" valign="middle">0x96ba9570</td>
<td align="center" valign="middle">0x0f8ca67b10</td>
<td align="center" valign="middle">677,146</td>
<td align="center" valign="middle">0</td>
</tr>
<tr>
<td align="left" valign="middle">6012r.txt</td>
<td align="center" valign="middle">NB1000D4450</td>
<td align="center" valign="middle">XX06</td>
<td align="center" valign="middle">0&#x00D7;0000</td>
<td align="center" valign="middle">0&#x00D7;000029000</td>
<td align="center" valign="middle">0&#x00D7;0007</td>
<td align="center" valign="middle">0x000010bc00</td>
<td align="center" valign="middle">4</td>
<td align="center" valign="middle">1</td>
</tr>
<tr>
<td align="left" valign="middle">6013r.txt</td>
<td align="center" valign="middle">NB1000D4450</td>
<td align="center" valign="middle">XX04</td>
<td align="center" valign="middle">0&#x00D7;0000</td>
<td align="center" valign="middle">0x12a77d1a0c00</td>
<td align="center" valign="middle">0x95e8008</td>
<td align="center" valign="middle">0x2765ae202e00</td>
<td align="center" valign="middle">359,596</td>
<td align="center" valign="middle">0</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The graphical representation of the output using the power-on time (in minutes) on the x-axis and the failure count on the y-axis were useful for assessing efficacy of the proposed prediction model. To analyze how the workload affects the survivability of an HDD, two attributes are considered: the TD and the AAR. HDDs of the same model were grouped by both TD as well as AAR.</p>
<p>These g number of groups were labeled as TDi where <italic>i</italic>&#x2009;=&#x2009;1, 2, &#x2026;,g. The HDDs were also grouped by their average access rates (AAR) into <italic>m</italic> number of groups called AAR<sub>j</sub> where j&#x2009;=&#x2009;1, 2,&#x2026;.,<italic>m</italic>.</p>
<p>The ranges for these groups were determined automatically in R such that each group had an approximately equal number of drives. A particular drive could belong to any one of the TD groups and any one of the AAR groups, so a drive with low total data could have a high access rate and a drive with high total data could have a low access rate.</p>
<p>To create the total data (TD) accessed till the drive was in operational state and average access rate (AAR) groups for the same period is as in <xref ref-type="fig" rid="fig1">Figure 1</xref>.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>The flowchart to form the dynamic cluster of TD groups and ARR groups from the HDD datasets.</p>
</caption>
<graphic xlink:href="fcomp-06-1400943-g001.tif"/>
</fig>
<p>The flowchart shown above for formation of TD_groups (denoted as &#x002A;&#x002A;&#x002A; inside <xref ref-type="fig" rid="fig1">Figure 1</xref>) utilizes the <bold>cut2</bold> function in R to divide the variable AAR into groups or intervals. The components of the algorithm are as follows:</p>
<list list-type="bullet">
<list-item><p><bold>AAR</bold>: This is the variable contains the AAR data. It represents the ratio of the total data accessed to the power-on time of the hard disk drive.</p></list-item>
<list-item><p><bold>cut2()</bold>: This is a function in R, contained in <bold>Hmisc</bold> package, is used to divide a continuous variable into intervals or groups. Unlike the base R <bold>cut()</bold> function, <bold>cut2()</bold> ensures that each group has approximately the same number of observations.</p></list-item>
<list-item><p><bold>g&#x2009;=&#x2009;n_groups</bold>: This argument specifies the desired number of groups into which the variable AAR is to be divided and n_groups represent the value assigned to the variable indicating the number of groups to be created.</p></list-item>
<list-item><p><bold>levels.mean&#x2009;=&#x2009;TRUE</bold>: This argument indicates that the labels for the intervals or groups should represent the mean value of the observations within each group. This helps with interpreting the intervals more intuitively.</p></list-item>
<list-item><p>Overall, the <bold>cut2()</bold> function(denoted as ### inside <xref ref-type="fig" rid="fig1">Figure 1</xref>) with the provided arguments divides the AAR variable into n_groups intervals, ensuring that each interval contains approximately the same number of observations, and assigns labels representing the mean value of the observations within each interval.</p></list-item>
</list>
<p>From the data and plotted graphs (<xref ref-type="fig" rid="fig2">Figures 2A</xref>,<xref ref-type="fig" rid="fig2">B</xref>), it is observed that there are initially a large number of failures, but later, the rate of failures is constant for some time and then increases again. This behavior seems to correspond to a &#x201C;bathtub&#x201D; -type curve.</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p><bold>(A)</bold> The failure counts vs. power-on time for the M1-A1 hard drive model. <bold>(B)</bold> The failure counts vs. power-on time for the M2-A1 hard drive model.</p>
</caption>
<graphic xlink:href="fcomp-06-1400943-g002.tif"/>
</fig>
</sec>
<sec id="sec15">
<label>3.3</label>
<title>Procedure for replicating the model analysis</title>
<p>i. Setting Up the Environment.</p>
<p>R is installed on the system along with necessary libraries.</p>
<p>ii. Reading and Formatting Raw Data.</p>
<p>The CSV file containing the dataset is read ensuring appropriate file path.</p>
<p>The SSUITEstatus column is formatted in order to convert it to numeric.</p>
<p>iii. Grouping Data Based on Total Data and Data Rate.</p>
<p>Total Data (TD) and Average Access Rate (AAR) are calculated. TD and AAR are grouped into specified number of groups.</p>
<p>iv. Converting Data frame to Data Table.</p>
<p>The dataframe is converted to a data.table for efficient data manipulation:</p>
<p>v. Calculating Cumulative Sum and Percentage Failures.</p>
<p>Cumulative sums and percentage failures by AAR_groups are calculated.</p>
<p>vi. Splitting Data into Passed and Failed Subsets.</p>
<p>Subsets of the data for passed and failed tests are created. Cumulative sums and percentages for both subsets are calculated.</p>
<p>vii. Creation of a Survival Object.</p>
<p>A survival object using PowerOnTime and SSUITEstatus is created.</p>
<p>viii. Creation of Cox Proportional Hazards Model.</p>
<p>The Cox model is created using AAR_groups and TD_groups.</p>
<p>ix. Checking of Proportional Hazards Assumption.</p>
<p>cox.zph is used to check the proportional hazards assumption.</p>
<p>x. Predictions Based on the created Cox Model.</p>
<p>The output is predicted based given inputs using Cox model.</p>
<p>xi. Display of the Output.</p>
<p>The predicted pass or fail value is printed.</p>
</sec>
</sec>
<sec id="sec16">
<label>4</label>
<title>Case studies and comparison of results</title>
<p>Previous research examined the susceptibility to failure of the new generation of hard disks because of a reduction in the spacing between the heads and media. The research also examined the effect of the amount of data written to or read from the drive. The current research attempts to evaluate the effect of the data access rate on the survivability of the HDDs using the Cox model.</p>
<sec id="sec17">
<label>4.1</label>
<title>SMART data based model for internal test suit</title>
<p>SMART is a standard technology embedded in most modern hard disk drives (HDDs) to monitor various indicators of drive health. These indicators include attributes, like read error rates, spin-up time, temperature, and reallocated sectors, count. The primary goal of SMART is to predict drive failures before they occur, allowing for preventive measures such as data backup or drive replacement. The SMART system has been reported in literature (<xref ref-type="bibr" rid="ref21">Villalobos, 2020</xref>; <xref ref-type="bibr" rid="ref14">Rajashekarappa and Sunjiv Soyjaudah, 2011</xref>) for HDD failure prediction. The internal analysis for the SMART based data has employed Kaplan-Meir approach. The impact of TD and AAR are exhibited in the following <xref ref-type="fig" rid="fig3">Figures 3</xref>, <xref ref-type="fig" rid="fig4">4</xref>.</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>Impact of TD on Survival Probability on KME approach.</p>
</caption>
<graphic xlink:href="fcomp-06-1400943-g003.tif"/>
</fig>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption>
<p>Impact of AAR on Survival Probability on KME approach.</p>
</caption>
<graphic xlink:href="fcomp-06-1400943-g004.tif"/>
</fig>
<p>SMART is widely used and supported across the industry, making it a reliable standard for monitoring HDD health (<xref ref-type="bibr" rid="ref21">Villalobos, 2020</xref>; <xref ref-type="bibr" rid="ref14">Rajashekarappa and Sunjiv Soyjaudah, 2011</xref>). It provides early warnings based on predefined thresholds for various health indicators, enabling proactive maintenance. However, it is limited by the predefined thresholds and May not account for all failure modes; sometimes produces false positives or misses failures. It is evident from the above figures that KME is well suited for incorporating one impact of TD on survival probability. The impact of AAR is also reflected in survivability in <xref ref-type="fig" rid="fig4">Figure 4</xref>. However, the simultaneous two impact of both TD and ARR is not possible to capture using KME approach.</p>
<p><xref ref-type="fig" rid="fig5">Figures 5A</xref>,<xref ref-type="fig" rid="fig5">B</xref>, <xref ref-type="fig" rid="fig6">6A,B</xref> present the data for HDDs with model parameters of interest in this analysis, namely, TD and AAR. The color code indicates which groups the hard drives belong to and the total amount of data read by or written to them. <xref ref-type="fig" rid="fig5">Figures 5</xref>, <xref ref-type="fig" rid="fig6">6</xref> show AAR and TD distributions of the drive population. The dispersion of the data used for the M1-A1 model is given below in <xref ref-type="table" rid="tab2">Table 2</xref> (for AAR_groups) and <xref ref-type="table" rid="tab3">Table 3</xref> (for TD_groups).</p>
<fig position="float" id="fig5">
<label>Figure 5</label>
<caption>
<p><bold>(A)</bold> The M1-A1 model HDDs declared &#x201C;passed&#x201D; by internal test suite. <bold>(B)</bold> The M1-A1 model HDDs declared &#x201C;failed&#x201D; by internal test suite.</p>
</caption>
<graphic xlink:href="fcomp-06-1400943-g005.tif"/>
</fig>
<fig position="float" id="fig6">
<label>Figure 6</label>
<caption>
<p><bold>(A)</bold> The M2-A1 model HDDs declared &#x201C;passed&#x201D; by internal test suite. <bold>(B)</bold> The M2-A1 model HDDs declared &#x201C;failed&#x201D; by internal test suite.</p>
</caption>
<graphic xlink:href="fcomp-06-1400943-g006.tif"/>
</fig>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Dispersion of AAR_groups the dataset for M1-A1 used by the internal test suite.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Data Sets</th>
<th align="center" valign="top">Count</th>
<th align="center" valign="top">AAR groups</th>
<th align="center" valign="top">Mean</th>
<th align="center" valign="top">Median</th>
<th align="center" valign="top">Minimum</th>
<th align="center" valign="top">Maximum</th>
<th align="center" valign="top">Standard deviation</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top" rowspan="5">Total dataset</td>
<td align="center" valign="top">7,903</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">3,102,686</td>
<td align="center" valign="top">469603.1</td>
<td align="center" valign="top">5.35E-02</td>
<td align="center" valign="top">1.24E+07</td>
<td align="center" valign="top">3.92E+06</td>
</tr>
<tr>
<td align="center" valign="top">7,903</td>
<td align="center" valign="top">2</td>
<td align="center" valign="top">26,100,943</td>
<td align="center" valign="top">25884602.9</td>
<td align="center" valign="top">1.24E+07</td>
<td align="center" valign="top">4.04E+07</td>
<td align="center" valign="top">7.99E+06</td>
</tr>
<tr>
<td align="center" valign="top">7,903</td>
<td align="center" valign="top">3</td>
<td align="center" valign="top">58,562,114</td>
<td align="center" valign="top">57,867,575</td>
<td align="center" valign="top">4.04E+07</td>
<td align="center" valign="top">7.90E+07</td>
<td align="center" valign="top">1.12E+07</td>
</tr>
<tr>
<td align="center" valign="top">7,903</td>
<td align="center" valign="top">4</td>
<td align="center" valign="top">104,903,206</td>
<td align="center" valign="top">103477796.4</td>
<td align="center" valign="top">7.90E+07</td>
<td align="center" valign="top">1.36E+08</td>
<td align="center" valign="top">1.67E+07</td>
</tr>
<tr>
<td align="center" valign="top">7,902</td>
<td align="center" valign="top">5</td>
<td align="center" valign="top">1.4613E+10</td>
<td align="center" valign="top">188,703,828</td>
<td align="center" valign="top">1.36E+08</td>
<td align="center" valign="top">9.34E+13</td>
<td align="center" valign="top">1.07E+12</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="5">Dataset declared as Failed by internal test suite</td>
<td align="center" valign="top">4,864</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">2,402,446</td>
<td align="center" valign="top">228,134</td>
<td align="center" valign="top">5.35E-02</td>
<td align="center" valign="top">1.24E+07</td>
<td align="center" valign="top">3.59E+06</td>
</tr>
<tr>
<td align="center" valign="top">3,020</td>
<td align="center" valign="top">2</td>
<td align="center" valign="top">25,880,789</td>
<td align="center" valign="top">25,160,218</td>
<td align="center" valign="top">1.24E+07</td>
<td align="center" valign="top">4.04E+07</td>
<td align="center" valign="top">7.99E+06</td>
</tr>
<tr>
<td align="center" valign="top">2,932</td>
<td align="center" valign="top">3</td>
<td align="center" valign="top">58,695,593</td>
<td align="center" valign="top">58,137,170</td>
<td align="center" valign="top">4.04E+07</td>
<td align="center" valign="top">7.90E+07</td>
<td align="center" valign="top">1.13E+07</td>
</tr>
<tr>
<td align="center" valign="top">2,761</td>
<td align="center" valign="top">4</td>
<td align="center" valign="top">104,873,553</td>
<td align="center" valign="top">103,724,219</td>
<td align="center" valign="top">7.90E+07</td>
<td align="center" valign="top">1.36E+08</td>
<td align="center" valign="top">1.66E+07</td>
</tr>
<tr>
<td align="center" valign="top">2,862</td>
<td align="center" valign="top">5</td>
<td align="center" valign="top">3.4415E+10</td>
<td align="center" valign="top">189,678,100</td>
<td align="center" valign="top">1.36E+08</td>
<td align="center" valign="top">9.34E+13</td>
<td align="center" valign="top">1.75E+12</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="5">Dataset declared as passed by Internal test suite</td>
<td align="center" valign="top">3,039</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">4,223,439</td>
<td align="center" valign="top">2,980,341</td>
<td align="center" valign="top">5.66E-02</td>
<td align="center" valign="top">1.24E+07</td>
<td align="center" valign="top">4,155,452</td>
</tr>
<tr>
<td align="center" valign="top">4,883</td>
<td align="center" valign="top">2</td>
<td align="center" valign="top">26,237,101</td>
<td align="center" valign="top">26,245,474</td>
<td align="center" valign="top">1.24E+07</td>
<td align="center" valign="top">4.04E+07</td>
<td align="center" valign="top">7,985,086</td>
</tr>
<tr>
<td align="center" valign="top">4,971</td>
<td align="center" valign="top">3</td>
<td align="center" valign="top">58,483,386</td>
<td align="center" valign="top">57,727,494</td>
<td align="center" valign="top">4.04E+07</td>
<td align="center" valign="top">7.90E+07</td>
<td align="center" valign="top">11,098,270</td>
</tr>
<tr>
<td align="center" valign="top">5,142</td>
<td align="center" valign="top">4</td>
<td align="center" valign="top">104,919,129</td>
<td align="center" valign="top">103,317,505</td>
<td align="center" valign="top">7.90E+07</td>
<td align="center" valign="top">1.36E+08</td>
<td align="center" valign="top">16,738,192</td>
</tr>
<tr>
<td align="center" valign="top">5,040</td>
<td align="center" valign="top">5</td>
<td align="center" valign="top">3,368,262,156</td>
<td align="center" valign="top">188,156,025</td>
<td align="center" valign="top">1.36E+08</td>
<td align="center" valign="top">1.56E+13</td>
<td align="center" valign="top">2.1924E+11</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption>
<p>Dispersion of TD_groups the dataset for M1-A1 used by the internal test suite.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Data Sets</th>
<th align="center" valign="top">Counts</th>
<th align="center" valign="top">TD groups</th>
<th align="center" valign="top">Mean</th>
<th align="center" valign="top">Median</th>
<th align="center" valign="top">Minimum</th>
<th align="center" valign="top">Maximum</th>
<th align="center" valign="top">Standard Deviation</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top" rowspan="5">Total dataset</td>
<td align="center" valign="top">7,903</td>
<td align="center" valign="bottom">1</td>
<td align="center" valign="bottom">8.80E+13</td>
<td align="center" valign="bottom">4.67E+13</td>
<td align="center" valign="bottom">123,392</td>
<td align="center" valign="bottom">1.97E+15</td>
<td align="center" valign="bottom">1.12E+14</td>
</tr>
<tr>
<td align="center" valign="top">7,903</td>
<td align="center" valign="bottom">2</td>
<td align="center" valign="bottom">8.80E+13</td>
<td align="center" valign="bottom">4.67E+13</td>
<td align="center" valign="bottom">123,392</td>
<td align="center" valign="bottom">1.97E+15</td>
<td align="center" valign="bottom">1.12E+14</td>
</tr>
<tr>
<td align="center" valign="top">7,903</td>
<td align="center" valign="bottom">3</td>
<td align="center" valign="bottom">8.80E+13</td>
<td align="center" valign="bottom">4.67E+13</td>
<td align="center" valign="bottom">123,392</td>
<td align="center" valign="bottom">1.97E+15</td>
<td align="center" valign="bottom">1.12E+14</td>
</tr>
<tr>
<td align="center" valign="top">7,903</td>
<td align="center" valign="bottom">4</td>
<td align="center" valign="bottom">8.80E+13</td>
<td align="center" valign="bottom">4.67E+13</td>
<td align="center" valign="bottom">123,392</td>
<td align="center" valign="bottom">1.97E+15</td>
<td align="center" valign="bottom">1.12E+14</td>
</tr>
<tr>
<td align="center" valign="top">7,902</td>
<td align="center" valign="bottom">5</td>
<td align="center" valign="bottom">8.80E+13</td>
<td align="center" valign="bottom">4.67E+13</td>
<td align="center" valign="bottom">123,392</td>
<td align="center" valign="bottom">1.97E+15</td>
<td align="center" valign="bottom">1.12E+14</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="5">Dataset declared as failed by internal test suite</td>
<td align="center" valign="top">5,211</td>
<td align="center" valign="bottom">1</td>
<td align="center" valign="bottom">8.80E+13</td>
<td align="center" valign="bottom">4.67E+13</td>
<td align="center" valign="bottom">1.23E+05</td>
<td align="center" valign="bottom">1.97E+15</td>
<td align="center" valign="bottom">1.12E+14</td>
</tr>
<tr>
<td align="center" valign="top">3,337</td>
<td align="center" valign="bottom">2</td>
<td align="center" valign="bottom">8.80E+13</td>
<td align="center" valign="bottom">4.67E+13</td>
<td align="center" valign="bottom">1.23E+05</td>
<td align="center" valign="bottom">1.97E+15</td>
<td align="center" valign="bottom">1.12E+14</td>
</tr>
<tr>
<td align="center" valign="top">2,927</td>
<td align="center" valign="bottom">3</td>
<td align="center" valign="bottom">8.80E+13</td>
<td align="center" valign="bottom">4.67E+13</td>
<td align="center" valign="bottom">1.23E+05</td>
<td align="center" valign="bottom">1.97E+15</td>
<td align="center" valign="bottom">1.12E+14</td>
</tr>
<tr>
<td align="center" valign="top">2,777</td>
<td align="center" valign="bottom">4</td>
<td align="center" valign="bottom">8.80E+13</td>
<td align="center" valign="bottom">4.67E+13</td>
<td align="center" valign="bottom">1.23E+05</td>
<td align="center" valign="bottom">1.97E+15</td>
<td align="center" valign="bottom">1.12E+14</td>
</tr>
<tr>
<td align="center" valign="top">2,187</td>
<td align="center" valign="bottom">5</td>
<td align="center" valign="bottom">8.80E+13</td>
<td align="center" valign="bottom">4.67E+13</td>
<td align="center" valign="bottom">123,392</td>
<td align="center" valign="bottom">1.97E+15</td>
<td align="center" valign="bottom">1.12E+14</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="5">Dataset declared as passed by internal test suite</td>
<td align="center" valign="top">2,692</td>
<td align="center" valign="bottom">1</td>
<td align="center" valign="bottom">8.80E+13</td>
<td align="center" valign="bottom">4.67E+13</td>
<td align="center" valign="bottom">1.23E+05</td>
<td align="center" valign="bottom">1.97E+15</td>
<td align="center" valign="bottom">1.12E+14</td>
</tr>
<tr>
<td align="center" valign="top">4,566</td>
<td align="center" valign="bottom">2</td>
<td align="center" valign="bottom">8.80E+13</td>
<td align="center" valign="bottom">4.67E+13</td>
<td align="center" valign="bottom">1.23E+05</td>
<td align="center" valign="bottom">1.97E+15</td>
<td align="center" valign="bottom">1.12E+14</td>
</tr>
<tr>
<td align="center" valign="top">4,976</td>
<td align="center" valign="bottom">3</td>
<td align="center" valign="bottom">8.80E+13</td>
<td align="center" valign="bottom">4.67E+13</td>
<td align="center" valign="bottom">1.23E+05</td>
<td align="center" valign="bottom">1.97E+15</td>
<td align="center" valign="bottom">1.12E+14</td>
</tr>
<tr>
<td align="center" valign="top">5,126</td>
<td align="center" valign="bottom">4</td>
<td align="center" valign="bottom">8.80E+13</td>
<td align="center" valign="bottom">4.67E+13</td>
<td align="center" valign="bottom">1.23E+05</td>
<td align="center" valign="bottom">1.97E+15</td>
<td align="center" valign="bottom">1.12E+14</td>
</tr>
<tr>
<td align="center" valign="top">5,715</td>
<td align="center" valign="bottom">5</td>
<td align="center" valign="bottom">8.80E+13</td>
<td align="center" valign="bottom">4.67E+13</td>
<td align="center" valign="bottom">1.23E+05</td>
<td align="center" valign="bottom">1.97E+15</td>
<td align="center" valign="bottom">1.12E+14</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Some interesting inferences can be made from the plots shown in <xref ref-type="fig" rid="fig5">Figures 5</xref>, <xref ref-type="fig" rid="fig6">6</xref>. <xref ref-type="fig" rid="fig5">Figure 5A</xref> shows the data for M1-A1 HDDs that failed in the customer environment as soon as a certain rate of access was reached but passed the internal test suite run in the lab. The drives were declared failed either by the storage controllers that were using them or by the test suites run on them later. This shows that each drive has an access rate threshold point, after which the probability of failure increases drastically.</p>
<p><xref ref-type="fig" rid="fig5">Figure 5B</xref> presents the data for M1-A1 HDDs that failed in the customer environment and failed in the internal suite. <xref ref-type="fig" rid="fig6">Figure 6A</xref> shows the data for M2-A1 HDDs that failed in the customer environment as soon as a certain rate of access was reached but passed the internal test suite run in the lab. <xref ref-type="fig" rid="fig6">Figure 6B</xref> shows the data for the M2-A1 HDDs that failed in the customer environment and failed in the internal suite. These figures show that the failures are similar between groups with different amounts of data. After a certain rate of access, the majority of the drives failed regardless of the amount of data accessed.</p>
</sec>
<sec id="sec18">
<label>4.2</label>
<title>Survivability analysis using cox proportionality model</title>
<p>The proposed model is a predictive model based on Total Data (TD) read/written and Average Access Rate (AAR). It introduces a novel strategy of reallocating HDDs with high failure probabilities to different redundancy groups.</p>
<p>It provides with a proactive approach to HDD management by predicting failures based on data access patterns. It enhances system resilience by dynamically reallocating resources based on failure probabilities. Also, it utilizes real-world data from storage controllers for validation, ensuring practical applicability. However, it May need customization for different storage environments and workloads.</p>
<p>The Cox Model uses the parameters of the AAR groups and TD groups to generate the survival model. The equation for the Cox model using R is (<xref ref-type="disp-formula" rid="E10">Equation 10</xref>):</p>
<disp-formula id="E10">
<label>(10)</label>
<mml:math id="M11">
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mi>M</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>A</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>_</mml:mo>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>x</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="italic">coxph</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>M</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>A</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>_</mml:mo>
<mml:mi mathvariant="italic">surv</mml:mi>
<mml:mspace width="0.25em"/>
<mml:mo>~</mml:mo>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>.</mml:mo>
<mml:mspace width="0.25em"/>
<mml:mi mathvariant="italic">factor</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>R</mml:mi>
<mml:mo>_</mml:mo>
<mml:mi mathvariant="italic">groups</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mspace width="0.25em"/>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>+</mml:mo>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>.</mml:mo>
<mml:mspace width="0.25em"/>
<mml:mi mathvariant="italic">factor</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>D</mml:mi>
<mml:mo>_</mml:mo>
<mml:mi mathvariant="italic">groups</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mspace width="0.25em"/>
<mml:mi mathvariant="italic">data</mml:mi>
<mml:mspace width="0.25em"/>
<mml:mo>=</mml:mo>
<mml:mi>M</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>_</mml:mo>
<mml:mi>A</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>The Cox model equation fits a Cox proportional hazards regression model (<bold>coxph()</bold>) to the survival data (<bold>M1A1_surv</bold>). It includes the average access rate (<bold>AAR_groups</bold>) and the total data groups (<bold>TD_groups</bold>) as covariates. The model aims to understand how these factors influence the hazard rate or survival probability over time. The output, <bold>M1A1_cox</bold>, is a Cox model object containing the results of the regression analysis.</p>
<p>The effect of the AAR on each of these TD groups was analyzed. For a given AAR, the survival data were generated for each of the TD groups using the above Cox model, whose algorithm is outlined in <xref ref-type="fig" rid="fig7">Figure 7</xref>.</p>
<fig position="float" id="fig7">
<label>Figure 7</label>
<caption>
<p>The flowchart for the validation of the survivability-based predictive data model.</p>
</caption>
<graphic xlink:href="fcomp-06-1400943-g007.tif"/>
</fig>
<p>For each value of AAR, survival data were generated for the different data groups, using the algorithm given below in <xref ref-type="fig" rid="fig7">Figure 7</xref>.</p>
<p>The algorithm in <xref ref-type="fig" rid="fig7">Figure 7</xref>, as demonstrated using ARR1 values for different TD groups (TD1-TD5), was applied in a similar way to generate the survival graphs for the other AAR groups, AAR2&#x2013;AAR5. It utilizes cut2() for formation of TD_groups (denoted as &#x002A;&#x002A;&#x002A; inside <xref ref-type="fig" rid="fig7">Figure 7</xref>) and also for AAR_groups (denoted as ### inside <xref ref-type="fig" rid="fig7">Figure 7</xref>). In the algorithms, the HDD model M1-A1 was used as an example. The same algorithms can be used for M2-A2 model drives to generate survival data and survival graphs. The unit of the PowerOnTime (X-axis) in <xref ref-type="fig" rid="fig8">Figures 8</xref>&#x2013;<xref ref-type="fig" rid="fig17">17</xref> is in minutes. The effect of workload on the survivability of a hard disk (M1-A1 model).</p>
<fig position="float" id="fig8">
<label>Figure 8</label>
<caption>
<p>The M1-A1 model hard disk survival probability graph (AAR1 and TD1:5).</p>
</caption>
<graphic xlink:href="fcomp-06-1400943-g008.tif"/>
</fig>
<fig position="float" id="fig9">
<label>Figure 9</label>
<caption>
<p>The M1-A1 model hard disk survival probability graph (AAR2 and TD1:5).</p>
</caption>
<graphic xlink:href="fcomp-06-1400943-g009.tif"/>
</fig>
<fig position="float" id="fig10">
<label>Figure 10</label>
<caption>
<p>The M1-A1 model hard disk survival probability graph (AAR3 and TD1:5).</p>
</caption>
<graphic xlink:href="fcomp-06-1400943-g010.tif"/>
</fig>
<fig position="float" id="fig11">
<label>Figure 11</label>
<caption>
<p>The M1-A1 model hard disk survival probability graph (AAR4 and TD1:5).</p>
</caption>
<graphic xlink:href="fcomp-06-1400943-g011.tif"/>
</fig>
<fig position="float" id="fig12">
<label>Figure 12</label>
<caption>
<p>The M1-A1 model hard disk survival probability graph (AAR5 and TD1:5).</p>
</caption>
<graphic xlink:href="fcomp-06-1400943-g012.tif"/>
</fig>
<fig position="float" id="fig13">
<label>Figure 13</label>
<caption>
<p>The M2-A1 model hard disk survival probability graph (AAR1 and TD1:5).</p>
</caption>
<graphic xlink:href="fcomp-06-1400943-g013.tif"/>
</fig>
<fig position="float" id="fig14">
<label>Figure 14</label>
<caption>
<p>The M2-A1 model hard disk survival probability graph (AAR2 and TD1:5).</p>
</caption>
<graphic xlink:href="fcomp-06-1400943-g014.tif"/>
</fig>
<fig position="float" id="fig15">
<label>Figure 15</label>
<caption>
<p>The M2-A1 model hard disk survival probability graph (AAR3 and TD1:5).</p>
</caption>
<graphic xlink:href="fcomp-06-1400943-g015.tif"/>
</fig>
<fig position="float" id="fig16">
<label>Figure 16</label>
<caption>
<p>The M2-A1 model hard disk survival probability graph (AAR4 and TD1:5).</p>
</caption>
<graphic xlink:href="fcomp-06-1400943-g016.tif"/>
</fig>
<fig position="float" id="fig17">
<label>Figure 17</label>
<caption>
<p>The M2-A1 model hard disk survival probability graph (AAR5 and TD1:5).</p>
</caption>
<graphic xlink:href="fcomp-06-1400943-g017.tif"/>
</fig>
<p>In summary, the proposed method stands out by integrating workload-specific metrics (TD and AAR) into the predictive model, which offers a more targeted approach to identifying potential HDD failures compared to traditional SMART models. By focusing on empirical data from actual storage systems and introducing dynamic reallocation strategies, this method aims to provide a more practical and effective solution for mitigating HDD failures in enterprise environments.</p>
<p>This comparison highlights the advantages of the proposed method in enhancing system resilience and optimizing resource allocation, while also acknowledging the challenges and areas for further improvement. The authors emphasize that this comprehensive approach can significantly contribute to the advancement of storage system reliability and efficiency.</p>
<p>The graphs shown in <xref ref-type="fig" rid="fig8">Figures 8</xref>&#x2013;<xref ref-type="fig" rid="fig12">12</xref> were plotted using data for the HDD model M1-A1. They show that for a given data volume (TD), as the access rate is changed from AAR1 to AAR5, the survival probabilities are reduced. They also show that the access rate has a significant effect on HDD failures.</p>
<p>The graphs shown in <xref ref-type="fig" rid="fig8">Figures 8</xref>&#x2013;<xref ref-type="fig" rid="fig12">12</xref> also confirm that for a fixed amount of data, if the data access rate is increased, the survival probability of the drives decreases. They also show that the rate of access has a significant impact on survivability apart from just the total data written or read.</p>
<p>The graphs shown in <xref ref-type="fig" rid="fig13">Figures 13</xref>&#x2013;<xref ref-type="fig" rid="fig17">17</xref> were plotted using data from the HDD model M2-A1. They show that for a given amount of data (TD), as the access rate changes from AAR1 to AAR5, the survival probabilities are reduced. They also show that the access rate has a significant effect on HDD failures. The graphs shown in <xref ref-type="fig" rid="fig12">Figures 12</xref>&#x2013;<xref ref-type="fig" rid="fig16">16</xref> confirm that, for a fixed amount of data, if the data access rate is increased, the survival probability of the drives decreases. They also show that the rate of access has a significant impact on survivability apart from just the total data written or read.</p>
</sec>
</sec>
<sec sec-type="conclusions" id="sec19">
<label>5</label>
<title>Conclusion</title>
<p>In the modern digital world, data is one of the critical assets of any business, and data availability and data access are important metrics for predicting the failure of HDDs. The method for storage system controller firmware proposed in this work could be used to manage failure-predicted disk drives efficiently and intelligently, in order to provide greater data availability and survivability to customers. This approach allows one to apply better throttling mechanisms and the preemptive migration of data from HDDs that are predicted to fail. The proposed model paves the way for facilitating collaborations between storage experts, workload analysts, and system architects to merge domain-specific insights. This interdisciplinary approach could foster innovation by integrating expertise from various fields, leading to more robust and comprehensive survivability analyses. Thus, this work should make significant contributions toward improving data availability in companies.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="sec20">
<title>Data availability statement</title>
<p>The datasets presented in this article are not readily available because these datasets are vendor specific. Requests to access the datasets should be directed to <email>amanmallik01@gmail.com</email>.</p>
</sec>
<sec sec-type="author-contributions" id="sec21">
<title>Author contributions</title>
<p>AM: Conceptualization, Methodology, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing, Data curation, Formal analysis, Investigation, Software, Validation, Visualization. BR: Formal analysis, Methodology, Writing &#x2013; original draft, Investigation, Project administration, Resources. GS: Formal analysis, Investigation, Supervision, Writing &#x2013; review &#x0026; editing, Validation.</p>
</sec>
<sec sec-type="funding-information" id="sec22">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research, authorship, and/or publication of this article.</p>
</sec>
<sec sec-type="COI-statement" id="sec23">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="sec24">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>David</surname> <given-names>C. R.</given-names></name></person-group> (<year>1972</year>). <article-title>Regression models and life tables (with discussion)</article-title>. <source>J. R. Stat. Soc.</source> <volume>34</volume>, <fpage>187</fpage>&#x2013;<lpage>202</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.2517-6161.1972.tb00899.x</pub-id></citation></ref>
<ref id="ref2"><citation citation-type="book"><person-group person-group-type="author"><name><surname>George</surname> <given-names>T.</given-names></name></person-group> (<year>2013</year>). <source>Why specify workload?. <italic>Western Digital technologies, Inc.,</italic> 3355 Michelson drive. Suite 100 Irvine</source>. <publisher-loc>California</publisher-loc>.</citation></ref>
<ref id="ref3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hamerly</surname> <given-names>G.</given-names></name> <name><surname>Elkan</surname> <given-names>C.</given-names></name></person-group> (<year>2001</year>). <article-title>Bayesian approaches to failure prediction for disk drives</article-title>. <source>ICML</source> <volume>1</volume>, <fpage>202</fpage>&#x2013;<lpage>209</lpage>.</citation></ref>
<ref id="ref4"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Hosmer</surname> <given-names>D. W.</given-names></name> <name><surname>Lemeshow</surname> <given-names>S.</given-names></name> <name><surname>May</surname> <given-names>S.</given-names></name></person-group> (<year>2008</year>). <source>Applied survival analysis: Regression modeling of time-to-event data</source>. <publisher-loc>Hoboken, New Jersey, USA</publisher-loc>: <publisher-name>John Wiley &#x0026; Sons</publisher-name>.</citation></ref>
<ref id="ref5"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Kalbfleisch</surname> <given-names>J. D.</given-names></name> <name><surname>Prentice</surname> <given-names>R. L.</given-names></name></person-group> (<year>2011</year>). <source>The statistical analysis of failure time data</source>. <publisher-loc>Hoboken, New Jersey, USA</publisher-loc>: <publisher-name>John Wiley &#x0026; Sons</publisher-name>.</citation></ref>
<ref id="ref6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kaplan</surname> <given-names>E. L.</given-names></name> <name><surname>Meier</surname> <given-names>P.</given-names></name></person-group> (<year>1958</year>). <article-title>Nonparametric estimation from incomplete observations</article-title>. <source>J. Am. Stat. Assoc.</source> <volume>53</volume>, <fpage>457</fpage>&#x2013;<lpage>481</lpage>. doi: <pub-id pub-id-type="doi">10.1080/01621459.1958.10501452</pub-id></citation></ref>
<ref id="ref7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kleinbaum</surname> <given-names>D. G.</given-names></name> <name><surname>Klein</surname> <given-names>M.</given-names></name></person-group> (<year>1996</year>). <article-title>Survival analysis a self-learning text</article-title>. <source>Springer</source> <volume>52</volume>:<fpage>1528</fpage>. doi: <pub-id pub-id-type="doi">10.2307/2532873</pub-id></citation></ref>
<ref id="ref8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lambert</surname> <given-names>P. C.</given-names></name> <name><surname>Royston</surname> <given-names>P.</given-names></name></person-group> (<year>2009</year>). <article-title>Further development of flexible parametric models for survival analysis</article-title>. <source>Stata J.</source> <volume>9</volume>, <fpage>265</fpage>&#x2013;<lpage>290</lpage>. doi: <pub-id pub-id-type="doi">10.1177/1536867X0900900206</pub-id></citation></ref>
<ref id="ref9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lane</surname> <given-names>W. R.</given-names></name> <name><surname>Looney</surname> <given-names>S. W.</given-names></name> <name><surname>Wansley</surname> <given-names>J. W.</given-names></name></person-group> (<year>1986</year>). <article-title>An application of the cox proportional hazards model to bank failure</article-title>. <source>J. Bank. Financ.</source> <volume>10</volume>, <fpage>511</fpage>&#x2013;<lpage>531</lpage>. doi: <pub-id pub-id-type="doi">10.1016/S0378-4266(86)80003-6</pub-id></citation></ref>
<ref id="ref10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Q.</given-names></name> <name><surname>Xing</surname> <given-names>L.</given-names></name></person-group> (<year>2020</year>). <article-title>Survivability and vulnerability analysis of cloud RAID systems under disk faults and attacks</article-title>. <source>Int. J. Math. Eng. Manag. Sci.</source> <volume>6</volume>, <fpage>15</fpage>&#x2013;<lpage>29</lpage>. doi: <pub-id pub-id-type="doi">10.33889/ijmems.2021.6.1.003</pub-id></citation></ref>
<ref id="ref11"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Mohanta</surname> <given-names>T.</given-names></name> <name><surname>Ananthamurthy</surname> <given-names>S.</given-names></name></person-group> (<year>2006</year>). <article-title>A method to predict hard disk failures using SMART monitored parameters</article-title>. <conf-name>Recent developments in National Seminar on devices, circuits and communication (NASDEC2-06)</conf-name>. <publisher-loc>Ranchi. India</publisher-loc> <fpage>243</fpage>&#x2013;<lpage>246</lpage>.</citation></ref>
<ref id="ref12"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Mohanta</surname> <given-names>T.</given-names></name> <name><surname>Basireddy</surname> <given-names>R. R.</given-names></name> <name><surname>Nayaka</surname> <given-names>B. G.</given-names></name> <name><surname>Chirumamilla</surname> <given-names>N.</given-names></name></person-group> (<year>2015</year>). <article-title>Method for higher availability of the SSD storage system</article-title>. In <conf-name>2015 international conference on computational intelligence and networks</conf-name> (pp. <fpage>130</fpage>&#x2013;<lpage>135</lpage>). <publisher-name>IEEE</publisher-name>.</citation></ref>
<ref id="ref13"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Murray</surname> <given-names>J. F.</given-names></name> <name><surname>Hughes</surname> <given-names>G. F.</given-names></name> <name><surname>Kreutz-Delgado</surname> <given-names>K.</given-names></name></person-group> (eds.) (<year>2003</year>). <source>Hard drive failure prediction using non-parametric statistical methods</source>.</citation></ref>
<ref id="ref14"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Rajashekarappa</surname> <given-names>S</given-names></name> <name><surname>Sunjiv Soyjaudah</surname> <given-names>K. M.</given-names></name></person-group> (<year>2011</year>). <article-title>Self monitoring analysis and reporting technology (SMART) Copyback</article-title>. In <conf-name>International conference on information processing</conf-name> (pp. <fpage>463</fpage>&#x2013;<lpage>469</lpage>). <publisher-loc>Berlin, Heidelberg</publisher-loc>: <publisher-name>springer Berlin Heidelberg</publisher-name>.</citation></ref>
<ref id="ref15"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Rinc&#x00F3;n</surname> <given-names>C. A.</given-names></name> <name><surname>P&#x00E2;ris</surname> <given-names>J. F.</given-names></name> <name><surname>Vilalta</surname> <given-names>R.</given-names></name> <name><surname>Cheng</surname> <given-names>A. M.</given-names></name> <name><surname>Long</surname> <given-names>D. D.</given-names></name></person-group> (<year>2017</year>) <article-title>Disk failure prediction in heterogeneous environments</article-title>. <conf-name>In 2017 international symposium on performance evaluation of computer and telecommunication systems (SPECTS)</conf-name> (pp. <fpage>1</fpage>&#x2013;<lpage>7</lpage>). <publisher-name>IEEE</publisher-name>.</citation></ref>
<ref id="ref16"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Royston</surname> <given-names>P.</given-names></name> <name><surname>Sauerbrei</surname> <given-names>W.</given-names></name></person-group> (<year>2008</year>). <source>Multivariable model-building: A pragmatic approach to regression anaylsis based on fractional polynomials for modelling continuous variables</source>. <publisher-loc>Hoboken, New Jersey, USA</publisher-loc>: <publisher-name>John Wiley &#x0026; Sons</publisher-name>.</citation></ref>
<ref id="ref17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shen</surname> <given-names>J.</given-names></name> <name><surname>Wan</surname> <given-names>J.</given-names></name> <name><surname>Lim</surname> <given-names>S. J.</given-names></name> <name><surname>Yu</surname> <given-names>L.</given-names></name></person-group> (<year>2018</year>). <article-title>Random-forest-based failure prediction for hard disk drives</article-title>. <source>Int. J. Distrib. Sensor Netw.</source> <volume>14</volume>:<fpage>155014771880648</fpage>. doi: <pub-id pub-id-type="doi">10.1177/1550147718806480</pub-id></citation></ref>
<ref id="ref18"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Smith</surname> <given-names>T.</given-names></name> <name><surname>Smith</surname> <given-names>B.</given-names></name></person-group> (<year>2001</year>). <article-title>Survival analysis and the application of Cox&#x2019;s proportional hazards modeling using SAS</article-title>. <conf-name>SAS conference proceedings:. SAS Users Group International</conf-name>.</citation></ref>
<ref id="ref19"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Smith</surname> <given-names>T.</given-names></name> <name><surname>Smith</surname> <given-names>B.</given-names></name></person-group>, (<year>2004</year>).<article-title>Kaplan Meier and cox proportional hazards modeling: hands on survival analysis</article-title>. <conf-name>SAS conference proceedings: Western users of SAS software 2004 Pasadena, California</conf-name>.</citation></ref>
<ref id="ref20"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Smith</surname> <given-names>T.</given-names></name> <name><surname>Smith</surname> <given-names>B.</given-names></name></person-group> (<year>2005</year>). <article-title>Graphing the probability of event as a function of time using survivor function estimates and the SAS&#x00AE; System's PROC PHREG</article-title>. In <conf-name>SAS Conference Proceedings: Western Users of SAS Software</conf-name> (pp. <fpage>21</fpage>&#x2013;<lpage>23</lpage>).</citation></ref>
<ref id="ref21"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Villalobos</surname> <given-names>C. C.</given-names></name></person-group>, (<year>2020</year>) Survival analysis: hard drive reliability sample. Available at: <ext-link xlink:href="https://github.com/cconejov/Surv_Analysis_HDD" ext-link-type="uri">https://github.com/cconejov/Surv_Analysis_HDD</ext-link> (Accessed November 11, 2023).</citation></ref>
<ref id="ref22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>M.</given-names></name> <name><surname>Ge</surname> <given-names>W.</given-names></name> <name><surname>Tang</surname> <given-names>R.</given-names></name> <name><surname>Liu</surname> <given-names>P.</given-names></name></person-group> (<year>2023</year>). <article-title>Hard disk failure prediction based on blending ensemble learning</article-title>. <source>Appl. Sci.</source> <volume>13</volume>:<fpage>3288</fpage>. doi: <pub-id pub-id-type="doi">10.3390/app13053288</pub-id></citation></ref>
</ref-list>
</back>
</article>