<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<?covid-19-tdm?>
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Public Health</journal-id>
<journal-title>Frontiers in Public Health</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Public Health</abbrev-journal-title>
<issn pub-type="epub">2296-2565</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpubh.2021.741030</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Public Health</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A Novel Matrix Profile-Guided Attention LSTM Model for Forecasting COVID-19 Cases in USA</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Liu</surname> <given-names>Qian</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1420550/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Fung</surname> <given-names>Daryl L. X.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1453161/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Lac</surname> <given-names>Leann</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1420547/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Hu</surname> <given-names>Pingzhao</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/277268/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Biochemistry and Medical Genetics, University of Manitoba</institution>, <addr-line>Winnipeg, MB</addr-line>, <country>Canada</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Computer Science, University of Manitoba</institution>, <addr-line>Winnipeg, MB</addr-line>, <country>Canada</country></aff>
<aff id="aff3"><sup>3</sup><institution>Department of Statistics, University of Manitoba</institution>, <addr-line>Winnipeg, MB</addr-line>, <country>Canada</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Reza Lashgari, Shahid Beheshti University, Iran</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Tarek Mohamed Abd El-Aziz, The University of Texas Health Science Center at San Antonio, United States; Arko Barman, Rice University, United States</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Pingzhao Hu <email>pingzhao.hu&#x00040;umanitoba.ca</email></corresp>
<fn fn-type="other" id="fn001"><p>This article was submitted to Infectious Diseases - Surveillance, Prevention and Treatment, a section of the journal Frontiers in Public Health</p></fn>
<fn fn-type="equal" id="fn002"><p>&#x02020;These authors share first authorship</p></fn></author-notes>
<pub-date pub-type="epub">
<day>07</day>
<month>10</month>
<year>2021</year>
</pub-date>
<pub-date pub-type="collection">
<year>2021</year>
</pub-date>
<volume>9</volume>
<elocation-id>741030</elocation-id>
<history>
<date date-type="received">
<day>26</day>
<month>07</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>02</day>
<month>09</month>
<year>2021</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2021 Liu, Fung, Lac and Hu.</copyright-statement>
<copyright-year>2021</copyright-year>
<copyright-holder>Liu, Fung, Lac and Hu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license> </permissions>
<abstract><p><bold>Background:</bold> The outbreak of the novel coronavirus disease 2019 (COVID-19) has been raging around the world for more than 1 year. Analysis of previous COVID-19 data is useful to explore its epidemic patterns. Utilizing data mining and machine learning methods for COVID-19 forecasting might provide a better insight into the trends of COVID-19 cases. This study aims to model the COVID-19 cases and perform forecasting of three important indicators of COVID-19 in the United States of America (USA), which are the adjusted percentage of daily admitted hospitalized COVID-19 cases (<italic>hospital admission</italic>), the number of daily confirmed COVID-19 cases (<italic>confirmed cases</italic>), and the number of daily death cases caused by COVID-19 (<italic>death cases</italic>).</p>
<p><bold>Materials and Methods:</bold> The actual COVID-19 data from March 1, 2020 to August 5, 2021 were obtained from Carnegie Mellon University Delphi Research Group. A novel forecasting algorithm was proposed to model and predict the three indicators. This algorithm is a hybrid of an unsupervised time series anomaly detection technique called matrix profile and an attention-based long short-term memory (LSTM) model. Several classic statistical models and the baseline recurrent neural network (RNN) models were used as the baseline models. All models were evaluated using a repeated holdout training and test strategy.</p>
<p><bold>Results:</bold> The proposed matrix profile-assisted attention-based LSTM model performed the best among all the compared models, which has the root mean square error (RMSE) = 1.23, 31612.81, 467.17, mean absolute error (MAE) = 0.95, 26259.55, 364.02, and mean absolute percentage error (MAPE) = 0.25, 1.06, 0.55, for <italic>hospital admission, confirmed cases</italic>, and <italic>death cases</italic>, respectively.</p>
<p><bold>Conclusion:</bold> The proposed model is more powerful in forecasting COVID-19 cases. It can potentially aid policymakers in making prevention plans and guide health care managers to allocate health care resources reasonably.</p></abstract>
<kwd-group>
<kwd>COVID-19 forecasting</kwd>
<kwd>LSTM models</kwd>
<kwd>matrix profile</kwd>
<kwd>attention mechanism</kwd>
<kwd>epidemiological indicators</kwd>
</kwd-group>
<contract-sponsor id="cn001">Natural Sciences and Engineering Research Council of Canada<named-content content-type="fundref-id">10.13039/501100000038</named-content></contract-sponsor>
<counts>
<fig-count count="4"/>
<table-count count="2"/>
<equation-count count="24"/>
<ref-count count="40"/>
<page-count count="13"/>
<word-count count="7172"/>
</counts>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Background</title>
<p>It has been more than 1 year since the first case of the novel coronavirus disease (COVID-19) came to light in December 2019 (<xref ref-type="bibr" rid="B1">1</xref>). According to the interactive COVID-19 dashboard created and maintained by Johns Hopkins Center for Systems Science and Engineering (JHU-CSSE), COVID-19 has spread to 191 counties and caused 4,370,447 global deaths out of more than 207 million diagnosed cases by August 16, 2021 (<xref ref-type="bibr" rid="B2">2</xref>). COVID-19 was confirmed to be caused by severe acute respiratory syndrome coronavirus 2 (SARS-CoV-2) as defined by the International Committee on Taxonomy of Viruses (ICTV) (<xref ref-type="bibr" rid="B3">3</xref>). SARS-CoV-2 coronavirus is a type of &#x003B2;-coronavirus with many potential hosts, leading to difficulties in prevention and treatment (<xref ref-type="bibr" rid="B4">4</xref>, <xref ref-type="bibr" rid="B5">5</xref>).</p>
<p>As COVID-19 is rapidly spreading and putting the world under a very distressing situation, the WHO declared COVID-19 as a global pandemic in March 2020 (<xref ref-type="bibr" rid="B6">6</xref>). Since a whole year&#x00027;s data are now available, some epidemic patterns of COVID-19 have been observed. COVID-19 follows the dynamic transmission of an epidemic, with different magnitudes in terms of time, region, season, and weather, and exhibited as a non-linear relationship. Since new case prevention and healthcare resource management have become critical for every country, good time series forecasting tools for COVID-19 are extremely important and necessary for estimating the number of cases in the coming days.</p>
<p>There is a classic time series forecasting algorithm called autoregressive integrated moving average (ARIMA) (<xref ref-type="bibr" rid="B7">7</xref>), which is widely applied for infectious disease prediction in public health (<xref ref-type="bibr" rid="B8">8</xref>, <xref ref-type="bibr" rid="B9">9</xref>). ARIMA has been applied to COVID-19 forecasting as early as February 2020 (<xref ref-type="bibr" rid="B10">10</xref>). Ceylan et al. used ARIMA to predict the prevalence of COVID-19 for confirmed and deceased cases in Italy, Spain, and France from February 21, 2020 to April 15, 2020 (<xref ref-type="bibr" rid="B11">11</xref>). Chintalapudi et al. have forecasted the number of registered and recovered cases after a 60-day lockdown in Italy by ARIMA with an accuracy rate of more than 80% (<xref ref-type="bibr" rid="B12">12</xref>). Researchers have also widely applied ARIMA in comparison with other approaches for COVID-19 forecasting (<xref ref-type="bibr" rid="B13">13</xref>&#x02013;<xref ref-type="bibr" rid="B18">18</xref>). Since the trend of COVID-19 cases follows a seasonal pattern, and ARIMA is not able to capture seasonal patterns well, an improved variant of the ARIMA called Seasonal ARIMA (SARIMA) (<xref ref-type="bibr" rid="B19">19</xref>) was proposed to model the seasonality of time series data.</p>
<p>However, SARIMA is still considered to be too simple to recognize complex patterns in the data. In principle, more complex models, which could include other significant observed or hidden variables/factors in disease prevalence, could be considered when we design the forecasting framework. For example, unsupervised data-driven time series anomaly detection algorithms could find significant abnormal patterns within the time series data (<xref ref-type="bibr" rid="B20">20</xref>). If we could incorporate the anomaly information into the forecasting models, the performance may be increased. Matrix profile is one of such algorithms proposed by Keogh et al. (<xref ref-type="bibr" rid="B19">19</xref>). A matrix profile consists of two components: a distance vector and a profile index vector. The distance vector contains the minimum Euclidean distances among the patterns within the time series data. The indexes of the nearest neighbors are stored in the profile index vector. The idea is that if a part of the time series data is far different from its nearest neighbors, then it is likely an anomaly. Keogh and his team further developed a series of algorithms to calculate the matrix profile to express the abnormal patterns within time-series data (<xref ref-type="bibr" rid="B21">21</xref>&#x02013;<xref ref-type="bibr" rid="B25">25</xref>).</p>
<p>Some machine learning algorithms, such as echo state network (ESN) (<xref ref-type="bibr" rid="B26">26</xref>), gated recurrent unit (GRU) (<xref ref-type="bibr" rid="B27">27</xref>), and long short-term memory (LSTM) (<xref ref-type="bibr" rid="B28">28</xref>), have also been widely applied in time series forecasting. They all belong to a recurrent neural network (RNN), which is a family of neural network technologies with internal memory (state) to process sequences of inputs. With the memory mechanism in the RNN, the standard RNN can handle the time series data very well. However, if a time series is very long, it will be difficult to pass information from the earlier timesteps to the later ones. This problem is called the vanishing gradient problem (<xref ref-type="bibr" rid="B29">29</xref>). The ESN does not suffer from this vanishing gradient problem because the hidden neurons in an ESN are very sparsely connected to form a network reservoir. The weights of the reservoir are randomly assigned and not trainable. The information of the earlier time points is randomly passed to the last points. Due to the untrainable random hidden state, the ESN has high computational efficiency, but the untrainable random hidden state reduces the complexity of the model thus reduces the power (<xref ref-type="bibr" rid="B30">30</xref>). While the GRU and the LSTM reduce the vanishing gradient problem by making it easier to pass previous information throughout the state sequences. They all use gates to regulate the information flow (<xref ref-type="bibr" rid="B28">28</xref>). The difference is that the LSTM has three gates and a cell state while the GRU has only two gates. Therefore, the LSTM may have more flexible control of the information flow. Previous studies have tested the performance of GRU, LSTM, and several variants of LSTM models for predicting COVID-19 cases across different counties and confirmed their accuracy and robustness (<xref ref-type="bibr" rid="B31">31</xref>&#x02013;<xref ref-type="bibr" rid="B34">34</xref>).</p>
<p>However, these models cannot detect which time point is the important one for future prediction. Recently, the attention mechanism in machine learning was developed to overcome this limitation (<xref ref-type="bibr" rid="B35">35</xref>). This is achieved by keeping the intermediate information from the LSTM units, training the model to pay selective attention to the inputs, and relating them to the items in the output time series (<xref ref-type="bibr" rid="B36">36</xref>). The attention mechanism increases the computational burden but results in a more targeted model with better performance. In addition, the model is also able to show how attention is paid to the input time series when predicting the output. It can increase the explainability of the LSTM model, which is an essential characteristic of gaining trust from end users.</p>
<p>Although great advancements have been made in the theories and applications of both the matrix profile and the LSTM, limited efforts have been made to investigate the combination of the two approaches and explore the applications of this combined approach to time series data forecasting (such as COVID-19 cases). In this study, we aim to propose a novel framework, which is a hybrid of the unsupervised matrix profile to detect a potential time series anomaly and an attention-based LSTM model, to model and forecast COVID-19 cases in the United States of America (USA). We aim to achieve a more accurate COVID-19 forecasting model to support decision-making and guide future advanced model building.</p>
</sec> 
<sec sec-type="materials and methods" id="s2">
<title>Materials and Methods</title>
<sec>
<title>Data Source</title>
<p>We considered the USA COVID-19 data from March 1, 2020 to August 5, 2021, which were obtained from the website of Carnegie Mellon University Delphi Research Group (<xref ref-type="bibr" rid="B37">37</xref>). We focused on three indicators: the adjusted percentage of daily admitted hospitalized COVID-19 cases (<italic>hospital admission</italic>), the number of daily confirmed COVID-19 cases (<italic>confirmed cases</italic>), and the number of daily death cases caused by COVID-19 (<italic>death cases</italic>). The <italic>hospital admission</italic> is the estimated percentage of new hospital admissions with COVID-19. It is based on insurance claims data from health system partners and smoothed using a Gaussian linear smoother.</p>
</sec> 
<sec>
<title>Methods</title>
<sec>
<title>Data Pre-processing and Remapping</title>
<p>To improve model performance and consider the consistency in evaluating model performance between statistical and machine learning approaches, pre-processing raw data by normalization is necessary. We applied <italic>z</italic>-normalization or standardization to the data. The formulation to transform the observed raw data into <italic>z</italic>-score is <italic>Z</italic><sub>i</sub> = (Y<sub>i</sub>-&#x00232;)/s, where &#x00232; and s are the sample mean and standard deviation. <italic>Y</italic><sub>i</sub> is the observed raw data at time point <italic>i</italic>. After building a forecasting model and obtaining the predicted value <italic>Z</italic><sub>j</sub>, we remapped these values to the observed raw data scale by applying the formula: <italic>Y</italic><sub>j &#x0003D;</sub> s<sup>&#x0002A;</sup><italic>Z</italic><sub>j</sub> &#x0002B; &#x00232;. Here, <italic>Y</italic><sub>j</sub> is the predicted data with raw data scale at time point <italic>j</italic>.</p>
</sec> 
<sec>
<title>Matrix Profile for Time Series Data Analysis</title>
<p>Matrix profile compares snippets of the time series by computing the distance between each pair of snippets. A matrix profile consists of two components: a distance profile and a profile index vector. The distance profile contains the minimum Euclidean distances among the sub-snippets within the time series. If the minimum distance of a certain sub-snippet is very large, it is probably that this sub-snippet is an anomaly because it is very different from its nearest neighbor. The indexes of the nearest neighbors are stored in the profile index vector. Matrix profiles of the three COVID-19 indicators were calculated using the Python package &#x0201C;matrixprofile&#x0201D; (<xref ref-type="bibr" rid="B38">38</xref>). Several algorithms are provided by the package for computing the matrix profile, such as Scalable Time series Anytime Matrix Profile (STAMP) and Scalable Time series Ordered-search Matrix Profile (STOMP). We selected the STOMP function since it is faster. The window size was set as 7 (weekly anomaly). After the matrix profiles were calculated, top 10 discords (the top 10 sub-snippets with larger Euclidean distances with their nearest neighbors) were highlighted for the visualization. In addition to the distance profile, we also considered the index profile vector. The index profile vector stores the global index of the closest neighbor of each snippet. For instance, if the most similar snippet of the current snippet is at the 15th location, the global position of the current snippet will be 15. The relative position is the relative index of the closest neighbor of the current snippet. It can be calculated using the index profile vector. If the current snippet is at the 10th location and the nearest neighbor of the current snippet is at the 15th location, then the relative position will have a value of &#x0002B;5. In the remaining parts of the report, we mainly focus on the distance profile and the relative position, which are passed together with the normalized observed raw data into the LSTM for forecasting the COVID-19 cases.</p>
</sec> 
<sec>
<title>Baseline Models</title>
<sec>
<title>ARIMA Model for Seasonal Data (SARIMA)</title>
<p>Non-seasonal ARIMA is a generalized form of the autoregressive moving average (ARMA) model. The ARMA is a combination of the auto regression (AR) model of order <italic>p</italic>, and moving average (MA) model of order <italic>q</italic>.</p>
<p>Let <italic>y</italic><sub>t</sub> denote the <italic>d</italic>th difference of <italic>Y</italic><sub>t</sub>, and <italic>Y</italic><sub>t</sub> refer to the observation at time <italic>t</italic>, the general equation of the ARIMA (<italic>p, d, q</italic>) model is as follows.</p>
<disp-formula id="E2"><label>(1)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mtext>t</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>c</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mtext>p</mml:mtext></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mi>p</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B5;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B5;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B5;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>&#x003B5;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mi>q</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003D5; = [&#x003D5;<sub>1</sub>, &#x003D5;<sub>2</sub>, &#x02026;, &#x003D5;<sub>p</sub>] and &#x003B8; = [&#x003B8;<sub>1</sub>, &#x003B8;<sub>2</sub>, &#x02026;&#x003B8;<sub>q</sub>] are coefficients of AR and MA parts of the model, respectively. Here, c is a constant and &#x003B5;<sub>t</sub> is the residual assumed to be uncorrelated in the final selected ARIMA model.</p>
<p>The ARIMA is not capable of modeling seasonal data. Therefore, the SARIMA model includes the additional seasonal term (<italic>p, d, q</italic>)<sub>m</sub> where m is the number of observations per year. To estimate the coefficients in the SARIMA, the maximum likelihood estimation (MLE) and the least square estimation (LSE) were used (<xref ref-type="bibr" rid="B7">7</xref>). R function auto.arima() in the &#x0201C;forecast&#x0201D; package (<xref ref-type="bibr" rid="B39">39</xref>) was utilized to execute SARIMA.</p>
</sec> 
<sec>
<title>Standard RNN Model</title>
<p>Vanilla RNN has backward-linking connections. It can be computed as follows:</p>
<disp-formula id="E3"><label>(2)</label><mml:math id="M3"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x000A0;</mml:mtext><mml:mo>&#x0002B;</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mtext>&#x000A0;</mml:mtext><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>which can also be described as: at a given timestep <italic>t</italic>, each recurrent layer receives the input <italic>x</italic><sub><italic>t</italic></sub> and the output from the previous timestep <italic>y</italic><sub><italic>t</italic>&#x02212;1</sub>, then outputs the non-linearly processed <italic>y</italic><sub><italic>t</italic></sub>. The non-linearity comes from the activation function &#x003C3;. <italic>W</italic><sub><italic>x</italic></sub> and <italic>W</italic><sub><italic>y</italic></sub> are the input weights and output weights. <italic>b</italic><sub><italic>t</italic></sub> is the bias.</p>
</sec> 
<sec>
<title>ESN Model</title>
<p>The simple ESN model can be computed as follows:</p>
<disp-formula id="E4"><label>(3)</label><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x000A0;</mml:mtext><mml:mo>&#x0002B;</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where <italic>x</italic><sub><italic>t</italic></sub>, <italic>h</italic><sub><italic>t</italic></sub>, <italic>y</italic><sub><italic>t</italic></sub> are the input, hidden state, and output at time point <italic>t</italic>, respectively. &#x003C3; is the activation function. <italic>W</italic><sub><italic>x</italic></sub> and <italic>W</italic><sub><italic>r</italic></sub> weight matrices are randomly initialized and fixed in the training step, and only the output weight <italic>W</italic><sub><italic>y</italic></sub> is trainable.</p>
</sec> 
<sec>
<title>GRU Model</title>
<p>The GRU model has a reset gate (<italic>r</italic><sub><italic>t</italic></sub>), a cell state (<italic>h</italic><sub><italic>t</italic></sub>), and an output gate (<italic>y</italic><sub><italic>t</italic></sub>) to control the information flow.</p>
<disp-formula id="E5"><label>(4)</label><mml:math id="M5"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E6"><label>(5)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E7"><label>(6)</label><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mtext>&#x000A0;</mml:mtext><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mtext>&#x000A0;</mml:mtext><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002A;</mml:mo><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002A;</mml:mo><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>h</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>W</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mtext>&#x000A0;</mml:mtext><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002A;</mml:mo><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>C</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where [<italic>h</italic><sub><italic>t</italic>&#x02212;1</sub>, <italic>x</italic><sub><italic>t</italic></sub>] is the concatenation between the hidden state of the previous timestep, <italic>h</italic><sub><italic>t</italic>&#x02212;1</sub>, and the input of the current timestep, <italic>x</italic><sub><italic>t</italic></sub>. <italic>b</italic><sub><italic>o</italic></sub>, <italic>b</italic><sub><italic>t</italic></sub>, <italic>b</italic><sub><italic>C</italic></sub> are the bias of each gate.</p>
</sec>
</sec>
<sec>
<title>Matrix Profile-Guided Attention LSTM Models</title>
<sec>
<title>LSTM Without Attention</title>
<p>The LSTM model contains a forget gate (<italic>f</italic><sub><italic>t</italic></sub>), an input gate (<italic>i</italic><sub><italic>t</italic></sub>), and an output gate (<italic>y</italic><sub><italic>t</italic></sub>). The forget gate receives the hidden state from the previous timestep (<italic>h</italic><sub><italic>t</italic>&#x02212;1</sub>), and concatenates them with the input in the current timestep, then passes them into a linear layer with a sigmoid activation.</p>
<disp-formula id="E8"><label>(7)</label><mml:math id="M8"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>f</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p><italic>W</italic><sub><italic>f</italic></sub> is the weight of the forget gate.</p>
<p>The input gate controls how much new information will be passed into the current timestep, which can be formulated as follows:</p>
<disp-formula id="E9"><label>(8)</label><mml:math id="M9"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E10"><label>(9)</label><mml:math id="M10"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mo>&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>h</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>C</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>C</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p><inline-formula><mml:math id="M11"><mml:mover accent="true"><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mo>&#x0007E;</mml:mo></mml:mover></mml:math></inline-formula> is used to update the cell state (<italic>C</italic><sub><italic>t</italic></sub>) of the current timestep.</p>
<disp-formula id="E11"><label>(10)</label><mml:math id="M12"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x000A0;</mml:mtext><mml:mo>&#x0002A;</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x000A0;</mml:mtext><mml:mo>&#x0002A;</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mo>&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The output gate controls and filters the information from the cell state. The equation is:</p>
<disp-formula id="E12"><label>(11)</label><mml:math id="M13"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E13"><label>(12)</label><mml:math id="M14"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x000A0;</mml:mtext><mml:mo>&#x0002A;</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>h</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec> 
<sec>
<title>Convolutional Neural Network LSTM</title>
<p>The convolutional neural network LSTM (CNN-LSTM) is a model that uses a combination of CNN (<xref ref-type="bibr" rid="B40">40</xref>) to extract the features of the input and pass the extracted features as an input to the LSTM model. A CNN extracts features from a group of inputs based on the kernel size. The equation for convolutional operations is:</p>
<disp-formula id="E14"><label>(13)</label><mml:math id="M15"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi><mml:mi>p</mml:mi><mml:mi>u</mml:mi><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mo>,</mml:mo><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msubsup><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msubsup><mml:mtext>&#x000A0;</mml:mtext><mml:mo>&#x0002A;</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:msubsup><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>e</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:msubsup><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mo>,</mml:mo><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msubsup></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where <inline-formula><mml:math id="M16"><mml:msubsup><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> is the weights in layer <italic>l</italic> at row i and column j of the kernel, <italic>l</italic><sup><italic>l</italic>&#x02212;1</sup> is the output from the previous layer, and <italic>b</italic><sup><italic>l</italic></sup> is the bias in layer <italic>l</italic>. We kept the kernel size 2. With kernel size 2, the sequence length of the input will be reduced by 1 after every layer. In order to maintain the sequence length for the LSTM, we used a transposed CNN to expand the sequence length to the original length. The output of the CNN is fed as input to the LSTM.</p>
</sec> 
<sec>
<title>LSTM With Attention</title>
<p>We added an attention mechanism to our LSTM model. The proposed overall workflow can be found in <xref ref-type="fig" rid="F1">Figure 1</xref>. As there are outliers in the forecasted values, the attention mechanism can help LSTM models to focus on important parts of the time series to prevent getting skewed values. We utilized multiheaded attention where there are several heads and each head contains a query, a key, and a value. Multiheaded attention has been shown to be beneficial with different learned linear projections (<xref ref-type="bibr" rid="B25">25</xref>). The equation for the attention is:</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>The overall workflow of the proposed novel matrix profile-guided attention LSTM algorithm. Matrix profile feature could be the distance profile (the LSTM-MatAtt model) or the relative position profile (the LSTM-RelAtt). The matrix profile feature concatenated with the normalized observed data of the COVID-19 indicators are first input into the LSTM unit, then passed to the later steps of the encoder-decoder attention process. The final output is the predicted future values of the COVID-19 indicators.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpubh-09-741030-g0001.tif"/>
</fig>
<disp-formula id="E15"><label>(14)</label><mml:math id="M17"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>A</mml:mi><mml:mi>t</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>Q</mml:mi><mml:mo>,</mml:mo><mml:mi>K</mml:mi><mml:mo>,</mml:mo><mml:mi>V</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>s</mml:mi><mml:mi>o</mml:mi><mml:mi>f</mml:mi><mml:mi>t</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mi>Q</mml:mi><mml:msup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mi>V</mml:mi><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where Q is the query, K is the key, and V is the value. <italic>d</italic><sub><italic>k</italic></sub> is the dimension of the current layer. Q = [<italic>Q</italic><sub><italic>t</italic>&#x02212;<italic>p</italic></sub>, <italic>Q</italic><sub><italic>t</italic>&#x02212;<italic>p</italic>&#x0002B;1</sub>, &#x02026;, <italic>Q</italic><sub><italic>t</italic></sub>], K = [<italic>K</italic><sub><italic>t</italic>&#x02212;<italic>p</italic></sub>, <italic>K</italic><sub><italic>t</italic>&#x02212;<italic>p</italic>&#x0002B;1</sub>, &#x02026;, <italic>K</italic><sub><italic>t</italic></sub>], V = [<italic>V</italic><sub><italic>t</italic>&#x02212;<italic>p</italic></sub>, <italic>V</italic><sub><italic>t</italic>&#x02212;<italic>p</italic>&#x0002B;1</sub>, &#x02026;, <italic>V</italic><sub><italic>t</italic></sub>]. <italic>p</italic> is the attention span. The attention span would look at the previous <italic>p</italic> timesteps, so the model attends to them. <italic>t</italic> is the current timestep. Each head contains the attention equation:</p>
<disp-formula id="E16"><label>(15)</label><mml:math id="M18"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>h</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>A</mml:mi><mml:mi>t</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>Q</mml:mi><mml:msubsup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>Q</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mi>K</mml:mi><mml:msubsup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mi>V</mml:mi><mml:msubsup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>V</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p><italic>Q</italic><sub><italic>t</italic></sub>, <italic>K</italic><sub><italic>t</italic></sub>, <italic>V</italic><sub><italic>t</italic></sub> are obtained by passing the hidden outputs from the LSTM into a linear layer:</p>
<disp-formula id="E17"><label>(16)</label><mml:math id="M19"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>Q</mml:mi></mml:mrow></mml:msup></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E18"><label>(17)</label><mml:math id="M20"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:msup></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E19"><label>(18)</label><mml:math id="M21"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>V</mml:mi></mml:mrow></mml:msup></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The multiheaded attention concatenates the attentions of the heads and outputs the combination of the attentions of the head, which can be formulated as:</p>
<disp-formula id="E21"><label>(19)</label><mml:math id="M23"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>M</mml:mi><mml:mi>u</mml:mi><mml:mi>l</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>H</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi><mml:mi>d</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>Q</mml:mi><mml:mo>,</mml:mo><mml:mi>K</mml:mi><mml:mo>,</mml:mo><mml:mi>V</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>=</mml:mo><mml:mi>C</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>h</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mi>h</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mi>h</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>O</mml:mi></mml:mrow></mml:msup></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Furthermore, we incorporate an encoder-decoder architecture for the LSTM with an attention model (<xref ref-type="fig" rid="F1">Figure 1</xref>). The output of the LSTM is fed into the encoder. The attention architecture in the encoder undergoes multiheaded attention on its own input to determine which timestep the model should focus more on. It then passes its learned features to the decoder. The decoder receives the input from the output of the encoder. To aid in more guidance for the attention stage in the decoder, the matrix profile input is fed into the attention stage in the decoder in addition to the hidden features of the decoder. The decoder outputs its learned features and passes them into a linear layer to output the predicted value. The overall procedure of the matrix profile<bold>-</bold>guided attention LSTM model is summarized in <xref ref-type="table" rid="T3">Algorithm 1</xref>.</p>
<table-wrap position="float" id="T3">
<label>Algorithm 1</label>
<caption><p>LSTM with attention</p></caption>
<table frame="hsides" rules="groups">
<tbody>
<tr>
<td valign="top" align="left"><bold>Procedure</bold> Training with LSTM Attention (D, MP, Y) // D = data, MP = matrix profile, Y = targets</td>
</tr>
<tr>
<td valign="top" align="left">X &#x0003C; - Concat([D, MP])</td>
</tr>
<tr>
<td valign="top" align="left"><bold>For</bold> all <italic>X, Y</italic> <bold>do //</bold> X = inputs, Y = targets</td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0;&#x000A0;//Pass inputs into LSTM to get LSTM output: <italic>O</italic><sub><italic>L</italic></sub> &#x0003C; - LSTM(X) // encoder part</td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0;&#x000A0;<italic>A</italic><sub><italic>E</italic></sub> &#x0003C; - MultiHeadAtt(<italic>O</italic><sub><italic>L</italic></sub>)</td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0;&#x000A0;<italic>N</italic><sub><italic>E</italic></sub> &#x0003C; - LayerNorm(<italic>O</italic><sub><italic>L</italic></sub> &#x0002B; <italic>A</italic><sub><italic>E</italic></sub><bold>)</bold></td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0;&#x000A0;<italic>A</italic><sub><italic>O</italic></sub> &#x0003C; - <italic>ReLU</italic>(<italic>N</italic><sub><italic>E</italic></sub><italic>W</italic><sub><italic>N</italic><sub><italic>E</italic></sub></sub> &#x0002B; <italic>B</italic><sub><italic>N</italic><sub><italic>E</italic></sub></sub>)</td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0;&#x000A0;<italic>E</italic><sub><italic>O</italic></sub> &#x0003C; - LayerNorm(<italic>N</italic><sub><italic>E</italic></sub> &#x0002B; <italic>A</italic><sub><italic>O</italic></sub>)</td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0;&#x000A0;// pass encoderOutput into decoder</td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0;&#x000A0;// decoder part</td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0;&#x000A0;<italic>A</italic><sub><italic>D</italic></sub> &#x0003C; - MultiHeadAtt(<italic>E</italic><sub><italic>O</italic></sub>)</td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0;&#x000A0;<italic>D</italic><sub><italic>E</italic></sub> &#x0003C; - LayerNorm(<italic>A</italic><sub><italic>D</italic></sub> &#x0002B; <italic>E</italic><sub><italic>O</italic></sub>)</td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0;&#x000A0;<italic>D</italic><sub><italic>A</italic></sub> &#x0003C; - DecoderMultiHeadAtt(<italic>D</italic><sub><italic>E</italic></sub>, MP) // MP = matrix profile value</td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0;&#x000A0;<italic>D</italic><sub><italic>O</italic></sub> &#x0003C; - LayerNorm(<italic>D</italic><sub><italic>E</italic></sub> &#x0002B; <italic>D</italic><sub><italic>A</italic></sub>)</td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x00176; &#x0003C; - <italic>ReLU</italic>(<italic>D</italic><sub><italic>O</italic></sub><italic>W</italic><sub><italic>D</italic><sub><italic>O</italic></sub></sub> &#x0002B; <italic>B</italic><sub><italic>D</italic><sub><italic>O</italic></sub></sub>) // Predicted output</td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0;&#x000A0;// update weightsloss</td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0;&#x000A0;&#x0003C; - RMSELoss(&#x00176;, Y)</td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0;&#x000A0;backprop(loss)</td>
</tr>
<tr>
<td valign="top" align="left"><bold>End For</bold></td></tr>
</tbody>
</table>
</table-wrap>
<p>Given the combination of the normalized observed raw data, matrix profiles, and the attention mechanism, we evaluated different LSTM models for each of the three COVID-19 indicators used in the study, which include the following LSTM-related models:</p>
<list list-type="simple">
<list-item><p>1) <italic>LSTM</italic>: using only the normalized observed raw data.</p></list-item>
<list-item><p>2) <italic>CNN-LSTM</italic>: using only the normalized observed raw data with convolutional LSTM network.</p></list-item>
<list-item><p>3) <italic>LSTM-Att</italic>: using only the normalized observed raw data with attention mechanism in LSTM.</p></list-item>
<list-item><p>4) <italic>LSTM-MatAtt</italic>: using the normalized observed raw data and the distance profile fed into the attention-based LSTM network.</p></list-item>
<list-item><p>5) <italic>LSTM-RelAtt</italic>: using the normalized observed raw data and the relative position matrix profile feature fed into the attention-based LSTM network.</p></list-item>
</list>
</sec>
</sec>
</sec>
<sec>
<title>Hyperparameter Tuning</title>
<p>The number of reservoirs of ESN was tuned manually, ranging from 50 to 300, the best one was 200. GRU and simple RNN have one hidden unit, thus, there is no need to tune the number of hidden layers. The step size of the input time series was set as 12 for ESN, GRU, simple RNN, and all LSTM models with and without attention or matrix profile. Epochs were tuned separately for different models to achieve the best losses. The additional hyperparameters that we tuned for the LSTM networks, such as hidden dimensions, dropout, and feedforward dimensions, can be found in <xref ref-type="table" rid="T1">Table 1</xref>. These hyperparameters were tuned separately for different models with and without the addition of attention and the matrix profile to achieve the best convergence.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Parameters tuning of LSTM models.</p></caption>
<table frame="hsides" rules="groups">
<tbody>
<tr>
<td valign="top" align="left" colspan="11">LSTM</td>
</tr>
<tr>
<td valign="top" align="left">Runs</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">7</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">9</td>
<td valign="top" align="center">10</td>
</tr>
<tr>
<td valign="top" align="left">Hidden dimension</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">128</td>
</tr>
<tr>
<td valign="top" align="left">Dropout</td>
<td valign="top" align="center">0.95</td>
<td valign="top" align="center">0.36</td>
<td valign="top" align="center">0.21</td>
<td valign="top" align="center">0.34</td>
<td valign="top" align="center">0.85</td>
<td valign="top" align="center">0.77</td>
<td valign="top" align="center">0.39</td>
<td valign="top" align="center">0.02</td>
<td valign="top" align="center">0.07</td>
<td valign="top" align="center">0.09</td>
</tr>
<tr>
<td valign="top" align="left">Loss</td>
<td valign="top" align="center">404.74</td>
<td valign="top" align="center">229.33</td>
<td valign="top" align="center">316.65</td>
<td valign="top" align="center">188.39</td>
<td valign="top" align="center">293.42</td>
<td valign="top" align="center">297.75</td>
<td valign="top" align="center">242.4</td>
<td valign="top" align="center">258.1</td>
<td valign="top" align="center">287.42</td>
<td valign="top" align="center">309.82</td>
</tr>
<tr>
<td valign="top" align="left" colspan="11">CNN-LSTM</td>
</tr>
<tr>
<td valign="top" align="left">Runs</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">7</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">9</td>
<td valign="top" align="center">10</td>
</tr>
<tr>
<td valign="top" align="left">Hidden dimension</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">128</td>
</tr>
<tr>
<td valign="top" align="left">Dropout</td>
<td valign="top" align="center">0.95</td>
<td valign="top" align="center">0.36</td>
<td valign="top" align="center">0.21</td>
<td valign="top" align="center">0.34</td>
<td valign="top" align="center">0.85</td>
<td valign="top" align="center">0.77</td>
<td valign="top" align="center">0.39</td>
<td valign="top" align="center">0.02</td>
<td valign="top" align="center">0.07</td>
<td valign="top" align="center">0.09</td>
</tr>
<tr>
<td valign="top" align="left">Loss</td>
<td valign="top" align="center">827.48</td>
<td valign="top" align="center">801.15</td>
<td valign="top" align="center">820.6</td>
<td valign="top" align="center">807.1</td>
<td valign="top" align="center">828.87</td>
<td valign="top" align="center">812.52</td>
<td valign="top" align="center">763.49</td>
<td valign="top" align="center">811.17</td>
<td valign="top" align="center">716.7</td>
<td valign="top" align="center">805.17</td>
</tr>
<tr>
<td valign="top" align="left" colspan="11">LSTM attention (LSTM-Att)</td>
</tr>
<tr>
<td valign="top" align="left">Runs</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">7</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">9</td>
<td valign="top" align="center">10</td>
</tr>
<tr>
<td valign="top" align="left">Hidden dimension</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">128</td>
</tr>
<tr>
<td valign="top" align="left">Dropout</td>
<td valign="top" align="center">0.95</td>
<td valign="top" align="center">0.36</td>
<td valign="top" align="center">0.21</td>
<td valign="top" align="center">0.34</td>
<td valign="top" align="center">0.85</td>
<td valign="top" align="center">0.77</td>
<td valign="top" align="center">0.39</td>
<td valign="top" align="center">0.02</td>
<td valign="top" align="center">0.07</td>
<td valign="top" align="center">0.09</td>
</tr>
<tr>
<td valign="top" align="left">ff_dim&#x0002A;</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">16</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">64</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">16</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">16</td>
</tr>
<tr>
<td valign="top" align="left">Loss</td>
<td valign="top" align="center">421.18</td>
<td valign="top" align="center">505.58</td>
<td valign="top" align="center">265.48</td>
<td valign="top" align="center">444.86</td>
<td valign="top" align="center">534.89</td>
<td valign="top" align="center">435.63</td>
<td valign="top" align="center">530.28</td>
<td valign="top" align="center">506.72</td>
<td valign="top" align="center">571.58</td>
<td valign="top" align="center">461.99</td>
</tr>
<tr>
<td valign="top" align="left" colspan="11">LSTM matrix attention (LSTM-MatAtt)</td>
</tr>
<tr>
<td valign="top" align="left">Runs</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">7</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">9</td>
<td valign="top" align="center">10</td>
</tr>
<tr>
<td valign="top" align="left">Hidden dimension</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">128</td>
</tr>
<tr>
<td valign="top" align="left">Dropout</td>
<td valign="top" align="center">0.95</td>
<td valign="top" align="center">0.36</td>
<td valign="top" align="center">0.21</td>
<td valign="top" align="center">0.34</td>
<td valign="top" align="center">0.85</td>
<td valign="top" align="center">0.77</td>
<td valign="top" align="center">0.39</td>
<td valign="top" align="center">0.02</td>
<td valign="top" align="center">0.07</td>
<td valign="top" align="center">0.09</td>
</tr>
<tr>
<td valign="top" align="left">ff_dim</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">16</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">64</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">16</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">16</td>
</tr>
<tr>
<td valign="top" align="left">Loss</td>
<td valign="top" align="center">1377.05</td>
<td valign="top" align="center">1765.31</td>
<td valign="top" align="center">983.64</td>
<td valign="top" align="center">1618.63</td>
<td valign="top" align="center">677.07</td>
<td valign="top" align="center">1639.26</td>
<td valign="top" align="center">1194.62</td>
<td valign="top" align="center">887.21</td>
<td valign="top" align="center">1114.19</td>
<td valign="top" align="center">946.79</td>
</tr>
<tr>
<td valign="top" align="left" colspan="11">LSTM attention (LSTM-Att)</td>
</tr>
<tr>
<td valign="top" align="left">Runs</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">7</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">9</td>
<td valign="top" align="center">10</td>
</tr>
<tr>
<td valign="top" align="left">Hidden dimension</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">128</td>
</tr>
<tr>
<td valign="top" align="left">Dropout</td>
<td valign="top" align="center">0.95</td>
<td valign="top" align="center">0.36</td>
<td valign="top" align="center">0.21</td>
<td valign="top" align="center">0.34</td>
<td valign="top" align="center">0.85</td>
<td valign="top" align="center">0.77</td>
<td valign="top" align="center">0.39</td>
<td valign="top" align="center">0.02</td>
<td valign="top" align="center">0.07</td>
<td valign="top" align="center">0.09</td>
</tr>
<tr>
<td valign="top" align="left">ff_dim</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">16</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">64</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">16</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">16</td>
</tr>
<tr>
<td valign="top" align="left">Loss</td>
<td valign="top" align="center">1542.63</td>
<td valign="top" align="center">516.51</td>
<td valign="top" align="center">1605.26</td>
<td valign="top" align="center">556.59</td>
<td valign="top" align="center">570.86</td>
<td valign="top" align="center">436.76</td>
<td valign="top" align="center">511.54</td>
<td valign="top" align="center">786.18</td>
<td valign="top" align="center">454.08</td>
<td valign="top" align="center">566.33</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><italic>ff_dim is the feedforward dimension layer after the attention module</italic>.</p>
</table-wrap-foot>
</table-wrap>
</sec> 
<sec>
<title>Performance Evaluation</title>
<p>After building the models for each indicator, it is important to evaluate the prediction accuracy and compare the forecasting performance to other proposed models in forecasting the number of COVID-19 cases. Traditional K-fold cross-validation is designated for independent data. However, time-series data are considered as dependent data in which we use past events to forecast the future ones. Therefore, we consider a rep-holdout cross-validation method. In this method, we divided the time series data into training and testing sets by ascending time order. To conduct the model performance evaluation, we applied the rep-holdout strategy, in which the test sets are the last (most recent) 10, 20, 30, 40, and 50 percentage of all time points (<xref ref-type="fig" rid="F2">Figure 2</xref>). We measured the forecasting accuracy by root mean square error (RMSE), mean absolute error (MAE), and mean absolute percentage error (MAPE) using the test set.</p>
<disp-formula id="E22"><label>(20)</label><mml:math id="M24"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext>RMSE</mml:mtext><mml:mo>=</mml:mo><mml:msqrt><mml:mrow><mml:mfrac><mml:mrow><mml:mstyle displaystyle="true"><mml:msubsup><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msubsup></mml:mstyle><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x000A0;</mml:mtext><mml:msubsup><mml:mrow><mml:mo>-</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:msqrt><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E23"><label>(21)</label><mml:math id="M25"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext>MAE</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle="true"><mml:msubsup><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msubsup></mml:mstyle><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x000A0;</mml:mtext><mml:msubsup><mml:mrow><mml:mo>-</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E24"><label>(22)</label><mml:math id="M26"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext>MAPE</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle="true"><mml:msubsup><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msubsup></mml:mstyle><mml:mo>|</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x000A0;</mml:mtext><mml:msubsup><mml:mrow><mml:mo>-</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mtext>&#x000A0;</mml:mtext><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>n</italic> is the sample size for each testing set, <italic>y</italic><sub><italic>t</italic></sub> is the actual data, <inline-formula><mml:math id="M27"><mml:msubsup><mml:mrow><mml:mtext>&#x000A0;</mml:mtext><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> is the predicted data.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>The rep-holdout strategy for model training and validation. The X axis is the time points.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpubh-09-741030-g0002.tif"/>
</fig>

</sec> 
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<sec>
<title>Matrix Profiles</title>
<p>The observed signals of the three indicators and their matrix profiles can be found in <xref ref-type="fig" rid="F3">Figure 3</xref>. The trends of the <italic>hospital admission</italic> and <italic>death cases</italic> showed two peaks in April 2020 and January 2021 (top panels of <xref ref-type="fig" rid="F3">Figures 3A,C</xref>). In addition, some minor decreasing and increasing trends were also observed. Matrix profiles of these two indicators were able to detect these ups and downs (bottom panels of <xref ref-type="fig" rid="F3">Figures 3A,C</xref>). It should be noted that the beginning of the <italic>death cases</italic> time series was also detected as an anomaly because the death cases were pretty low at the first few days. The <italic>confirmed cases</italic> time series was stable overall until around November 2020 (top panel of <xref ref-type="fig" rid="F3">Figure 3B</xref>). The top 1 anomaly in its matrix profile was located in around November 2020, which is consistent with the visual observation (bottom panel of <xref ref-type="fig" rid="F3">Figure 3B</xref>). Overall, the data-driven unsupervised matrix profile successfully detected the intrinsic anomalous data in the time series of the three indicators.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>The raw time series and the matrix profile of the three indicators. <bold>(A)</bold> is the time series of COVID-19 hospital admissions and its matrix profile. <bold>(B)</bold> is the time series of the daily confirmed COVID-19 cases and its matrix profile. <bold>(C)</bold> is the time series of the daily death cases caused by COVID-19 and its matrix profile. The red segments in the matrix profile plots indicate the corresponding weeks that have large Euclidean distances to their nearest neighbor compared to all the other weeks, which also means these weeks marked as red are the top anomalies within the whole time series.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpubh-09-741030-g0003.tif"/>
</fig>
</sec> 
<sec>
<title>Seasonal ARIMA</title>
<p>The performance of the SARIMA is summarized in <xref ref-type="table" rid="T2">Table 2</xref>. Overall, SARIMA had a decent average performance of predicting the three indicators. Rep-holdout 1 (last 10% of time points as a testing set) did not always have best performance in predicting the three indicators among all other rep-holdout, and rep-holdout 5 (last 50% of time points as a testing set) was not the worst strategy, which indicates that the performance of the SARIMA model was not linearly related with the size of the training data. This is also applied to other models.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Model performance.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th/>
<th/>
<th valign="top" align="center" style="border-bottom: thin solid #000000;" colspan="6"><bold>Admission</bold></th>
<th valign="top" align="center" style="border-bottom: thin solid #000000;" colspan="6"><bold>Confirmed</bold></th>
<th valign="top" align="center" style="border-bottom: thin solid #000000;" colspan="6"><bold>Death</bold></th>
</tr>
<tr>
<th/>
<th valign="top" align="left"><bold>Rep-hold</bold></th>
<th valign="top" align="center"><bold>1</bold></th>
<th valign="top" align="center"><bold>2</bold></th>
<th valign="top" align="center"><bold>3</bold></th>
<th valign="top" align="center"><bold>4</bold></th>
<th valign="top" align="center"><bold>5</bold></th>
<th valign="top" align="center"><bold>Average</bold></th>
<th valign="top" align="center"><bold>1</bold></th>
<th valign="top" align="center"><bold>2</bold></th>
<th valign="top" align="center"><bold>3</bold></th>
<th valign="top" align="center"><bold>4</bold></th>
<th valign="top" align="center"><bold>5</bold></th>
<th valign="top" align="center"><bold>Average</bold></th>
<th valign="top" align="center"><bold>1</bold></th>
<th valign="top" align="center"><bold>2</bold></th>
<th valign="top" align="center"><bold>3</bold></th>
<th valign="top" align="center"><bold>4</bold></th>
<th valign="top" align="center"><bold>5</bold></th>
<th valign="top" align="center"><bold>Average</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Seasonal autoregressive integrated moving average (SARIMA)</td>
<td valign="top" align="left">Root mean square error (rmse)</td>
<td valign="top" align="center">2.93</td>
<td valign="top" align="center">1.13</td>
<td valign="top" align="center">5.58</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">2.86</td>
<td valign="top" align="center">3.7</td>
<td valign="top" align="center">48,893.89</td>
<td valign="top" align="center">43,200.22</td>
<td valign="top" align="center">43,890.86</td>
<td valign="top" align="center">285,457.64</td>
<td valign="top" align="center">182,560.72</td>
<td valign="top" align="center">120,800.67</td>
<td valign="top" align="center">192.03</td>
<td valign="top" align="center">339.58</td>
<td valign="top" align="center">1,401.85</td>
<td valign="top" align="center">2,103.07</td>
<td valign="top" align="center">1,254.41</td>
<td valign="top" align="center">1,058.19</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">mean absolute error (mae)</td>
<td valign="top" align="center">2.25</td>
<td valign="top" align="center">0.96</td>
<td valign="top" align="center">4.73</td>
<td valign="top" align="center">5.62</td>
<td valign="top" align="center">2.3</td>
<td valign="top" align="center">3.17</td>
<td valign="top" align="center">29,682.94</td>
<td valign="top" align="center">38,672.56</td>
<td valign="top" align="center">37,310.42</td>
<td valign="top" align="center">269,991.12</td>
<td valign="top" align="center">158,514.88</td>
<td valign="top" align="center">106,834.38</td>
<td valign="top" align="center">159.21</td>
<td valign="top" align="center">282.7</td>
<td valign="top" align="center">1,340.84</td>
<td valign="top" align="center">1,944.12</td>
<td valign="top" align="center">932.97</td>
<td valign="top" align="center">931.97</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">Mean absolute percentage error (mape)</td>
<td valign="top" align="center">0.89</td>
<td valign="top" align="center">0.53</td>
<td valign="top" align="center">2.2</td>
<td valign="top" align="center">2.42</td>
<td valign="top" align="center">0.67</td>
<td valign="top" align="center">1.34</td>
<td valign="top" align="center">0.76</td>
<td valign="top" align="center">3.65</td>
<td valign="top" align="center">3.05</td>
<td valign="top" align="center">14.88</td>
<td valign="top" align="center">8.7</td>
<td valign="top" align="center">6.21</td>
<td valign="top" align="center">2.51</td>
<td valign="top" align="center">3.15</td>
<td valign="top" align="center">7.94</td>
<td valign="top" align="center">9.55</td>
<td valign="top" align="center">2.49</td>
<td valign="top" align="center">5.13</td>
</tr>
<tr>
<td valign="top" align="left">Echo state network (ESN)</td>
<td valign="top" align="left">rmse</td>
<td valign="top" align="center">2.85</td>
<td valign="top" align="center">1.39</td>
<td valign="top" align="center">1.82</td>
<td valign="top" align="center">7.03</td>
<td valign="top" align="center">3.23</td>
<td valign="top" align="center">3.27</td>
<td valign="top" align="center">42,964.79</td>
<td valign="top" align="center">48,636.51</td>
<td valign="top" align="center">46,766.28</td>
<td valign="top" align="center">178,680.95</td>
<td valign="top" align="center">103,114.16</td>
<td valign="top" align="center">84,032.54</td>
<td valign="top" align="center">513.06</td>
<td valign="top" align="center">800.7</td>
<td valign="top" align="center">600.22</td>
<td valign="top" align="center">1,805.84</td>
<td valign="top" align="center">1,279.4</td>
<td valign="top" align="center">999.84</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">mae</td>
<td valign="top" align="center">2.46</td>
<td valign="top" align="center">1.21</td>
<td valign="top" align="center">1.54</td>
<td valign="top" align="center">6.67</td>
<td valign="top" align="center">2.93</td>
<td valign="top" align="center">2.96</td>
<td valign="top" align="center">25,838.09</td>
<td valign="top" align="center">43,065.5</td>
<td valign="top" align="center">42,408.47</td>
<td valign="top" align="center">170,927.49</td>
<td valign="top" align="center">92,714.16</td>
<td valign="top" align="center">74,990.74</td>
<td valign="top" align="center">433.16</td>
<td valign="top" align="center">712.51</td>
<td valign="top" align="center">452.26</td>
<td valign="top" align="center">1,622.81</td>
<td valign="top" align="center">992.24</td>
<td valign="top" align="center">842.59</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">mape</td>
<td valign="top" align="center">1.34</td>
<td valign="top" align="center">0.72</td>
<td valign="top" align="center">0.84</td>
<td valign="top" align="center">2.82</td>
<td valign="top" align="center">0.96</td>
<td valign="top" align="center">1.34</td>
<td valign="top" align="center">0.85</td>
<td valign="top" align="center">4.23</td>
<td valign="top" align="center">2.31</td>
<td valign="top" align="center">9.13</td>
<td valign="top" align="center">4.83</td>
<td valign="top" align="center">4.27</td>
<td valign="top" align="center">6.37</td>
<td valign="top" align="center">5.87</td>
<td valign="top" align="center">3.24</td>
<td valign="top" align="center">6.28</td>
<td valign="top" align="center">3.58</td>
<td valign="top" align="center">5.07</td>
</tr>
<tr>
<td valign="top" align="left">Recurrent neural network (RNN)</td>
<td valign="top" align="left">rmse</td>
<td valign="top" align="center">1.50</td>
<td valign="top" align="center">1.14</td>
<td valign="top" align="center">1.04</td>
<td valign="top" align="center">1.24</td>
<td valign="top" align="center">1.25</td>
<td valign="top" align="center">1.25</td>
<td valign="top" align="center">36,301.76</td>
<td valign="top" align="center">48,930.03</td>
<td valign="top" align="center">48,582.78</td>
<td valign="top" align="center">44,627.97</td>
<td valign="top" align="center">46,251.33</td>
<td valign="top" align="center">44,938.77</td>
<td valign="top" align="center">618.95</td>
<td valign="top" align="center">516.87</td>
<td valign="top" align="center">697.39</td>
<td valign="top" align="center">924.08</td>
<td valign="top" align="center">1,033.18</td>
<td valign="top" align="center">758.09</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">mae</td>
<td valign="top" align="center">1.24</td>
<td valign="top" align="center">0.86</td>
<td valign="top" align="center">0.76</td>
<td valign="top" align="center">0.96</td>
<td valign="top" align="center">0.98</td>
<td valign="top" align="center">0.96</td>
<td valign="top" align="center">27,413.55</td>
<td valign="top" align="center">30,122.34</td>
<td valign="top" align="center">28,976.36</td>
<td valign="top" align="center">28,576.04</td>
<td valign="top" align="center">30,771.02</td>
<td valign="top" align="center">29,171.86</td>
<td valign="top" align="center">458.87</td>
<td valign="top" align="center">354.16</td>
<td valign="top" align="center">463.03</td>
<td valign="top" align="center">594.38</td>
<td valign="top" align="center">681.61</td>
<td valign="top" align="center">510.41</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">mape</td>
<td valign="top" align="center">0.48</td>
<td valign="top" align="center">0.34</td>
<td valign="top" align="center">0.27</td>
<td valign="top" align="center">0.26</td>
<td valign="top" align="center">0.23</td>
<td valign="top" align="center">0.32</td>
<td valign="top" align="center">2.64</td>
<td valign="top" align="center">2.04</td>
<td valign="top" align="center">1.66</td>
<td valign="top" align="center">1.15</td>
<td valign="top" align="center">0.76</td>
<td valign="top" align="center">1.65</td>
<td valign="top" align="center">5.78</td>
<td valign="top" align="center">3.28</td>
<td valign="top" align="center">2.69</td>
<td valign="top" align="center">1.72</td>
<td valign="top" align="center">1.64</td>
<td valign="top" align="center">3.02</td>
</tr>
<tr>
<td valign="top" align="left">Gated recurrent unit (GRU)</td>
<td valign="top" align="left">rmse</td>
<td valign="top" align="center">1.57</td>
<td valign="top" align="center">1.2</td>
<td valign="top" align="center">1.03</td>
<td valign="top" align="center">1.22</td>
<td valign="top" align="center">1.24</td>
<td valign="top" align="center">1.25</td>
<td valign="top" align="center">48,021.18</td>
<td valign="top" align="center">30,072.92</td>
<td valign="top" align="center">52,378.17</td>
<td valign="top" align="center">27,062.73</td>
<td valign="top" align="center">39,457.26</td>
<td valign="top" align="center">39,398.45</td>
<td valign="top" align="center">560.39</td>
<td valign="top" align="center">546.44</td>
<td valign="top" align="center">676.17</td>
<td valign="top" align="center">827.9</td>
<td valign="top" align="center">941.93</td>
<td valign="top" align="center">710.56</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">mae</td>
<td valign="top" align="center">1.26</td>
<td valign="top" align="center">0.91</td>
<td valign="top" align="center">0.77</td>
<td valign="top" align="center">0.97</td>
<td valign="top" align="center">1.0</td>
<td valign="top" align="center">0.98</td>
<td valign="top" align="center">30,588.13</td>
<td valign="top" align="center">20,542.1</td>
<td valign="top" align="center">31,658.96</td>
<td valign="top" align="center">19,509.24</td>
<td valign="top" align="center">25,801</td>
<td valign="top" align="center">25,619.89</td>
<td valign="top" align="center">405.84</td>
<td valign="top" align="center">391.32</td>
<td valign="top" align="center">440.34</td>
<td valign="top" align="center">557.95</td>
<td valign="top" align="center">665.99</td>
<td valign="top" align="center">492.29</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">mape</td>
<td valign="top" align="center">0.5</td>
<td valign="top" align="center">0.35</td>
<td valign="top" align="center">0.27</td>
<td valign="top" align="center">0.26</td>
<td valign="top" align="center">0.23</td>
<td valign="top" align="center">0.32</td>
<td valign="top" align="center">2.3</td>
<td valign="top" align="center">1.57</td>
<td valign="top" align="center">1.76</td>
<td valign="top" align="center">0.9</td>
<td valign="top" align="center">0.82</td>
<td valign="top" align="center">1.47</td>
<td valign="top" align="center">5.38</td>
<td valign="top" align="center">3.17</td>
<td valign="top" align="center">2.15</td>
<td valign="top" align="center">1.8</td>
<td valign="top" align="center">1.96</td>
<td valign="top" align="center">2.89</td>
</tr>
<tr>
<td valign="top" align="left">Long short-term memory (LSTM)</td>
<td valign="top" align="left">rmse</td>
<td valign="top" align="center">0.99</td>
<td valign="top" align="center">0.58</td>
<td valign="top" align="center">0.50</td>
<td valign="top" align="center">2.27</td>
<td valign="top" align="center">3.18</td>
<td valign="top" align="center">1.50</td>
<td valign="top" align="center">40,777.76</td>
<td valign="top" align="center">8,524.12</td>
<td valign="top" align="center">6,615.82</td>
<td valign="top" align="center">54,148.58</td>
<td valign="top" align="center">76,302.02</td>
<td valign="top" align="center">37,273.66</td>
<td valign="top" align="center">199.09</td>
<td valign="top" align="center">387.76</td>
<td valign="top" align="center">709.34</td>
<td valign="top" align="center">1,191.08</td>
<td valign="top" align="center">588.35</td>
<td valign="top" align="center">615.12</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">mae</td>
<td valign="top" align="center">0.65</td>
<td valign="top" align="center">0.58</td>
<td valign="top" align="center">0.50</td>
<td valign="top" align="center">1.80</td>
<td valign="top" align="center">2.55</td>
<td valign="top" align="center">1.22</td>
<td valign="top" align="center">26,680.90</td>
<td valign="top" align="center">6,745.04</td>
<td valign="top" align="center">6,524.89</td>
<td valign="top" align="center">37,704.15</td>
<td valign="top" align="center">58,703.45</td>
<td valign="top" align="center">27,271.69</td>
<td valign="top" align="center">162.23</td>
<td valign="top" align="center">372.73</td>
<td valign="top" align="center">576.72</td>
<td valign="top" align="center">874.00</td>
<td valign="top" align="center">543.57</td>
<td valign="top" align="center">505.85</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">mape</td>
<td valign="top" align="center">0.23</td>
<td valign="top" align="center">0.17</td>
<td valign="top" align="center">0.14</td>
<td valign="top" align="center">0.60</td>
<td valign="top" align="center">0.44</td>
<td valign="top" align="center">0.31</td>
<td valign="top" align="center">1.03</td>
<td valign="top" align="center">0.13</td>
<td valign="top" align="center">0.11</td>
<td valign="top" align="center">0.74</td>
<td valign="top" align="center">0.59</td>
<td valign="top" align="center">0.52</td>
<td valign="top" align="center">1.88</td>
<td valign="top" align="center">0.54</td>
<td valign="top" align="center">0.52</td>
<td valign="top" align="center">0.61</td>
<td valign="top" align="center">0.32</td>
<td valign="top" align="center">0.77</td>
</tr>
<tr>
<td valign="top" align="left">Convolutional neural network (CNN)-LSTM</td>
<td valign="top" align="left">rmse</td>
<td valign="top" align="center">1.37</td>
<td valign="top" align="center">0.96</td>
<td valign="top" align="center">0.92</td>
<td valign="top" align="center">2.08</td>
<td valign="top" align="center">2.80</td>
<td valign="top" align="center">1.62</td>
<td valign="top" align="center">40,404.43</td>
<td valign="top" align="center">16,663.18</td>
<td valign="top" align="center">13,230.96</td>
<td valign="top" align="center">45,052.20</td>
<td valign="top" align="center">74,871.81</td>
<td valign="top" align="center">38,044.51</td>
<td valign="top" align="center">837.00</td>
<td valign="top" align="center">535.74</td>
<td valign="top" align="center">589.30</td>
<td valign="top" align="center">1,175.13</td>
<td valign="top" align="center">458.34</td>
<td valign="top" align="center">719.10</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">mae</td>
<td valign="top" align="center">1.27</td>
<td valign="top" align="center">0.77</td>
<td valign="top" align="center">0.92</td>
<td valign="top" align="center">1.49</td>
<td valign="top" align="center">2.34</td>
<td valign="top" align="center">1.36</td>
<td valign="top" align="center">28413.84</td>
<td valign="top" align="center">15,329.70</td>
<td valign="top" align="center">11,412.55</td>
<td valign="top" align="center">28,374.41</td>
<td valign="top" align="center">55,020.34</td>
<td valign="top" align="center">27,710.17</td>
<td valign="top" align="center">813.65</td>
<td valign="top" align="center">519.59</td>
<td valign="top" align="center">354.60</td>
<td valign="top" align="center">800.40</td>
<td valign="top" align="center">396.15</td>
<td valign="top" align="center">576.88</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">mape</td>
<td valign="top" align="center">0.78</td>
<td valign="top" align="center">0.29</td>
<td valign="top" align="center">0.26</td>
<td valign="top" align="center">0.46</td>
<td valign="top" align="center">0.38</td>
<td valign="top" align="center">0.43</td>
<td valign="top" align="center">1.37</td>
<td valign="top" align="center">0.29</td>
<td valign="top" align="center">0.20</td>
<td valign="top" align="center">0.70</td>
<td valign="top" align="center">0.53</td>
<td valign="top" align="center">0.62</td>
<td valign="top" align="center">2.64</td>
<td valign="top" align="center">0.75</td>
<td valign="top" align="center">0.45</td>
<td valign="top" align="center">0.55</td>
<td valign="top" align="center">0.22</td>
<td valign="top" align="center">0.92</td>
</tr>
<tr>
<td valign="top" align="left">LSTM-Att</td>
<td valign="top" align="left">rmse</td>
<td valign="top" align="center">1.03</td>
<td valign="top" align="center">0.92</td>
<td valign="top" align="center">0.84</td>
<td valign="top" align="center">2.34</td>
<td valign="top" align="center">2.82</td>
<td valign="top" align="center">1.59</td>
<td valign="top" align="center">40,190.04</td>
<td valign="top" align="center">15,190.05</td>
<td valign="top" align="center">12,312.38</td>
<td valign="top" align="center">54,340.17</td>
<td valign="top" align="center">76,099.67</td>
<td valign="top" align="center">39,626.46</td>
<td valign="top" align="center">196.35</td>
<td valign="top" align="center">452.30</td>
<td valign="top" align="center">682.91</td>
<td valign="top" align="center">1,190.51</td>
<td valign="top" align="center">492.40</td>
<td valign="top" align="center">602.89</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">mae</td>
<td valign="top" align="center">0.69</td>
<td valign="top" align="center">0.92</td>
<td valign="top" align="center">0.80</td>
<td valign="top" align="center">1.96</td>
<td valign="top" align="center">2.54</td>
<td valign="top" align="center">1.38</td>
<td valign="top" align="center">27,157.88</td>
<td valign="top" align="center">13,731.57</td>
<td valign="top" align="center">10,316.78</td>
<td valign="top" align="center">40,776.23</td>
<td valign="top" align="center">58,692.52</td>
<td valign="top" align="center">30,135.00</td>
<td valign="top" align="center">156.49</td>
<td valign="top" align="center">390.36</td>
<td valign="top" align="center">537.24</td>
<td valign="top" align="center">907.07</td>
<td valign="top" align="center">434.72</td>
<td valign="top" align="center">485.18</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">mape</td>
<td valign="top" align="center">0.25</td>
<td valign="top" align="center">0.26</td>
<td valign="top" align="center">0.23</td>
<td valign="top" align="center">0.55</td>
<td valign="top" align="center">0.38</td>
<td valign="top" align="center">0.34</td>
<td valign="top" align="center">0.71</td>
<td valign="top" align="center">0.26</td>
<td valign="top" align="center">0.18</td>
<td valign="top" align="center">0.70</td>
<td valign="top" align="center">0.53</td>
<td valign="top" align="center">0.48</td>
<td valign="top" align="center">0.79</td>
<td valign="top" align="center">0.71</td>
<td valign="top" align="center">0.46</td>
<td valign="top" align="center">0.55</td>
<td valign="top" align="center">0.25</td>
<td valign="top" align="center">0.55</td>
</tr>
<tr>
<td valign="top" align="left">LSTM-MatAtt</td>
<td valign="top" align="left">rmse</td>
<td valign="top" align="center">0.91</td>
<td valign="top" align="center">0.31</td>
<td valign="top" align="center">0.44</td>
<td valign="top" align="center">1.01</td>
<td valign="top" align="center">3.49</td>
<td valign="top" align="center"><bold>1.23</bold></td>
<td valign="top" align="center">41,368.47</td>
<td valign="top" align="center">18,142.10</td>
<td valign="top" align="center">8,341.91</td>
<td valign="top" align="center">39,247.68</td>
<td valign="top" align="center">50,963.88</td>
<td valign="top" align="center"><bold>31,612.81</bold></td>
<td valign="top" align="center">205.34</td>
<td valign="top" align="center">321.18</td>
<td valign="top" align="center">668.23</td>
<td valign="top" align="center">885.96</td>
<td valign="top" align="center">255.13</td>
<td valign="top" align="center"><bold>467.17</bold></td>
</tr>
<tr>
<td/>
<td valign="top" align="left">mae</td>
<td valign="top" align="center">0.74</td>
<td valign="top" align="center">0.31</td>
<td valign="top" align="center">0.44</td>
<td valign="top" align="center">0.79</td>
<td valign="top" align="center">2.47</td>
<td valign="top" align="center"><bold>0.95</bold></td>
<td valign="top" align="center">34,559.99</td>
<td valign="top" align="center">17,091.87</td>
<td valign="top" align="center">8,304.28</td>
<td valign="top" align="center">32,947.74</td>
<td valign="top" align="center">38,393.86</td>
<td valign="top" align="center"><bold>26,259.55</bold></td>
<td valign="top" align="center">166.47</td>
<td valign="top" align="center">255.29</td>
<td valign="top" align="center">527.03</td>
<td valign="top" align="center">659.52</td>
<td valign="top" align="center">211.79</td>
<td valign="top" align="center"><bold>364.02</bold></td>
</tr>
<tr>
<td/>
<td valign="top" align="left">mape</td>
<td valign="top" align="center">0.32</td>
<td valign="top" align="center">0.09</td>
<td valign="top" align="center">0.13</td>
<td valign="top" align="center">0.29</td>
<td valign="top" align="center">0.44</td>
<td valign="top" align="center"><bold>0.25</bold></td>
<td valign="top" align="center">3.35</td>
<td valign="top" align="center">0.45</td>
<td valign="top" align="center">0.13</td>
<td valign="top" align="center">0.73</td>
<td valign="top" align="center">0.66</td>
<td valign="top" align="center"><bold>1.06</bold></td>
<td valign="top" align="center">0.69</td>
<td valign="top" align="center">0.83</td>
<td valign="top" align="center">0.45</td>
<td valign="top" align="center">0.60</td>
<td valign="top" align="center">0.17</td>
<td valign="top" align="center"><bold>0.55</bold></td>
</tr>
<tr>
<td valign="top" align="left">LSTM-RelAtt</td>
<td valign="top" align="left">rmse</td>
<td valign="top" align="center">0.71</td>
<td valign="top" align="center">0.63</td>
<td valign="top" align="center">0.78</td>
<td valign="top" align="center">2.17</td>
<td valign="top" align="center">3.23</td>
<td valign="top" align="center">1.50</td>
<td valign="top" align="center">33,139.11</td>
<td valign="top" align="center">24,263.36</td>
<td valign="top" align="center">683.29</td>
<td valign="top" align="center">51,316.31</td>
<td valign="top" align="center">70,101.06</td>
<td valign="top" align="center">35,900.62</td>
<td valign="top" align="center">194.75</td>
<td valign="top" align="center">229.07</td>
<td valign="top" align="center">698.72</td>
<td valign="top" align="center">1,180.69</td>
<td valign="top" align="center">197.48</td>
<td valign="top" align="center">500.14</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">mae</td>
<td valign="top" align="center">0.58</td>
<td valign="top" align="center">0.63</td>
<td valign="top" align="center">0.75</td>
<td valign="top" align="center">1.66</td>
<td valign="top" align="center">2.49</td>
<td valign="top" align="center">1.22</td>
<td valign="top" align="center">24,412.11</td>
<td valign="top" align="center">23,770.01</td>
<td valign="top" align="center">505.07</td>
<td valign="top" align="center">36,789.93</td>
<td valign="top" align="center">50,511.32</td>
<td valign="top" align="center">27,197.69</td>
<td valign="top" align="center">155.75</td>
<td valign="top" align="center">191.42</td>
<td valign="top" align="center">688.39</td>
<td valign="top" align="center">782.20</td>
<td valign="top" align="center">195.39</td>
<td valign="top" align="center">402.63</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">mape</td>
<td valign="top" align="center">0.27</td>
<td valign="top" align="center">0.18</td>
<td valign="top" align="center">0.23</td>
<td valign="top" align="center">0.58</td>
<td valign="top" align="center">0.48</td>
<td valign="top" align="center">0.35</td>
<td valign="top" align="center">1.34</td>
<td valign="top" align="center">0.61</td>
<td valign="top" align="center">0.01</td>
<td valign="top" align="center">0.68</td>
<td valign="top" align="center">0.61</td>
<td valign="top" align="center">0.65</td>
<td valign="top" align="center">0.81</td>
<td valign="top" align="center">1.38</td>
<td valign="top" align="center">0.37</td>
<td valign="top" align="center">0.56</td>
<td valign="top" align="center">0.17</td>
<td valign="top" align="center">0.66</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec> 
<sec>
<title>RNN Models</title>
<p>The results of the simple RNN, ESN, and GRU models can be found in <xref ref-type="table" rid="T2">Table 2</xref>. The simple RNN and GRU achieved similar overall performance in predicting the three indicators, but their performances were better than that of the ESN model. The ESN performed slightly better than the SARIMA, while the simple RNN and GRU were much better than the SARIMA.</p>
</sec> 
<sec>
<title>LSTM Models</title>
<p>The LSTM model with attention mechanism and matrix profile (<italic>Model: LSTM-MatAtt</italic>) has achieved better-averaged performance in predicting the three indicators (<xref ref-type="table" rid="T2">Table 2</xref>). Furthermore, the <italic>LSTM-MatAtt</italic> model with rep-holdout 3 was the best model in predicting <italic>hospital admission</italic> (RMSE = 1.23, MAE = 0.95, MAPE = 0.25). This performance was far better than other models we tested, no matter if they were classic statistic models or any other RNN models (<xref ref-type="table" rid="T2">Table 2</xref>).</p>
<p>Overall, the LSTM models with attention mechanism fed together with the matrix profile outperformed the classic statistic models, the other RNN models, and the LSTM models without the attention mechanism as well as the matrix profile assistance in predicting the three indicators. Different models may need to be equipped with different rep-holdouts.</p>
</sec> 
<sec>
<title>COVID-19 Case Forecasting</title>
<p>After the performances of the models were evaluated and the hyperparameters were fine-tuned using the rep-holdout strategy, the final forecasting for the three indicators was performed based on the whole data (March 1, 2020 to August 5, 2021). We let the selected and well-trained models run freely to forecast the future data between August 6, 2021 and October 31, 2021 (<xref ref-type="fig" rid="F4">Figure 4</xref>). Note: We can only access the data up to August 5, 2021 at the time we submit the report.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>The model fitting and forecasting results of the three COVID-19 indicators by the selected models. <bold>(A)</bold> is the results for <italic>hospital admission</italic>, <bold>(B)</bold> is the results for <italic>confirmed cases</italic>, and <bold>(C)</bold> is the results for <italic>death cases</italic>. We showed the model fitting results based on the observed data from March 1, 2020 to August 5, 2021. The selected models are based on the performance shown in <xref ref-type="table" rid="T2">Table 2</xref> for the categories of traditional statistical models, RNN-based models, and the family of the LSTM models, including our proposed LSTM-based models. The model forecasting results are based on the best model among the proposed and the compared models for each indicator.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpubh-09-741030-g0004.tif"/>
</fig>
<p>Using the best models show in <xref ref-type="table" rid="T2">Table 2</xref>, we forecasted the cases of <italic>hospital admission, confirmed cases</italic>, and <italic>death cases</italic> using the trained LSTM-MatAtt models for the period from August 6, 2021 to October 31, 2021 (<xref ref-type="fig" rid="F4">Figure 4</xref>). The forecasting results of <italic>hospital admission show</italic> a significant rise between July 2021 and October 2021. The <italic>confirmed cases</italic> and <italic>death cases</italic> indicate a relatively stable trend in the next few months, but they have significantly decreased from the peaks in January&#x02013;February in 2021.</p>
</sec>
</sec>
<sec id="s4">
<title>Discussions and Conclusions</title>
<p>Classic statistic time series forecasting models and the baseline RNN models, which are the benchmarks of this study, are able to achieve descent predictions of the three indicators of COVID-19. The proposed novel LSTM models combining matrix profile and attention mechanism achieved the overall best performance. A different number of time points was assigned to the training set according to the rep-holdout strategy. According to <xref ref-type="table" rid="T1">Table 1</xref>, the model performances were not associated with the size of the training sets, which means a larger training set may not guarantee a better performance. Careful selection of a proper training strategy could potentially increase the performance.</p>
<p>There are some limitations in this study. First, although the proposed models were selected through a five rep-holdout strategy, it was not validated in another totally independent dataset. Second, forecasting future cases is often not accurate while its uncertainty is seriously underestimated. One such example is the case of SARS, where the fear of becoming a pandemic was overblown, resulting in overspending and the application of restrictive measures to be taken that turned out to be unnecessary. Due to the uncertain measures taken, mathematical models overpredicted the number of cases. This calls us to fully take these potential uncertain factors into account when we build the forecasting models. In our modeling strategy, we used an indirect approach to first detect anomalies existing in the history data, which may be related to these uncertain measures. Despite the inaccuracies associated with the predictions, forecasting is still useful in allowing us to better understand the current situation and make plans.</p>
<p>In conclusion, a novel unsupervised matrix profile combined with an attention-based LSTM algorithm was proposed. Our experiments showed that the proposed algorithm has the best ability to forecast COVID-19 cases than the classical statistic methods and the baseline RNN models. The forecasted data may provide potentially useful information to help decision-makers to control the consequences of COVID-19.</p>
</sec>
<sec sec-type="data-availability" id="s5">
<title>Data Availability Statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://cmu-delphi.github.io/delphi-epidata/">https://cmu-delphi.github.io/delphi-epidata/</ext-link>.</p>
</sec>
<sec id="s6">
<title>Author Contributions</title>
<p>QL and DF were responsible for the conceptualization, development of methodologies and writing, and editing the manuscript. LL performed data analysis and wrote the manuscript. PH provided advice on data analysis and critically reviewed the manuscript and was also involved in supervision and project administration. All authors had full access to all of the data in the study and can take responsibility for the integrity of the data and the accuracy of the data analysis.</p>
</sec>
<sec sec-type="funding-information" id="s7">
<title>Funding</title>
<p>This work was supported in part by the Natural Sciences and Engineering Research Council of Canada and the University of Manitoba. PH is the holder of the Manitoba Medical Services Foundation (MMSF) Allen Rouse Basic Science Career Development Research Award.</p>
</sec> 
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec> 
<sec sec-type="disclaimer" id="s8">
<title>Publisher&#x00027;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>  </body>
<back>
<ack><p>Data were downloaded from the website of Carnegie Mellon University Delphi Research Group.</p>
</ack>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>Disease outbreak news</collab></person-group>. <source>WHO | Novel Coronavirus &#x02013; China</source>. WHO (<year>2020</year>). Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.who.int/csr/don/12-january-2020-novel-coronavirus-china/en/">https://www.who.int/csr/don/12-january-2020-novel-coronavirus-china/en/</ext-link> (accessed Sep 22, 2020).</citation>
</ref>
<ref id="B2">
<label>2.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dong</surname> <given-names>E</given-names></name> <name><surname>Du</surname> <given-names>H</given-names></name> <name><surname>Gardner</surname> <given-names>L</given-names></name></person-group>. <article-title>An interactive web-based dashboard to track COVID-19 in real time</article-title>. <source>Lancet Infect Dis.</source> (<year>2020</year>) <volume>20</volume>:<fpage>533</fpage>&#x02013;<lpage>4</lpage>. <pub-id pub-id-type="doi">10.1016/S1473-3099(20)30120-1</pub-id><pub-id pub-id-type="pmid">32087114</pub-id></citation></ref>
<ref id="B3">
<label>3.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gorbalenya</surname> <given-names>AE</given-names></name> <name><surname>Baker</surname> <given-names>SC</given-names></name> <name><surname>Baric</surname> <given-names>RS</given-names></name> <name><surname>Groot</surname> <given-names>RJ</given-names></name> <name><surname>De Gulyaeva</surname> <given-names>AA</given-names></name> <name><surname>Haagmans</surname> <given-names>BL</given-names></name> <etal/></person-group>. <article-title>The species Severe acute respiratory syndrome-related coronavirus: classifying 2019-nCoV and naming it SARS-CoV-2</article-title>. <source>Nature microbiology</source> (<year>2020</year>) <fpage>536</fpage>.<pub-id pub-id-type="pmid">32123347</pub-id></citation></ref>
<ref id="B4">
<label>4.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vellingiri</surname> <given-names>B</given-names></name> <name><surname>Jayaramayya</surname> <given-names>K</given-names></name> <name><surname>Iyer</surname> <given-names>M</given-names></name> <name><surname>Narayanasamy</surname> <given-names>A</given-names></name> <name><surname>Govindasamy</surname> <given-names>V</given-names></name> <name><surname>Giridharan</surname> <given-names>B</given-names></name> <etal/></person-group>. <article-title>COVID-19: a promising cure for the global panic</article-title>. <source>Sci Total Environ.</source> (<year>2020</year>) <volume>725</volume>:<fpage>138277</fpage>. <pub-id pub-id-type="doi">10.1016/j.scitotenv.2020.138277</pub-id><pub-id pub-id-type="pmid">32278175</pub-id></citation></ref>
<ref id="B5">
<label>5.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Abd</surname> <given-names>El-Aziz TM</given-names></name> <name><surname>Stockand</surname> <given-names>JD</given-names></name></person-group>. <article-title>Recent progress and challenges in drug development against COVID-19 coronavirus (SARS-CoV-2) - an update on the status</article-title>. <source>Infect Genet Evol.</source> (<year>2020</year>) <volume>83</volume>:<fpage>104327</fpage>. <pub-id pub-id-type="doi">10.1016/j.meegid.2020.104327</pub-id><pub-id pub-id-type="pmid">32320825</pub-id></citation></ref>
<ref id="B6">
<label>6.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cucinotta</surname> <given-names>D</given-names></name> <name><surname>Vanelli</surname> <given-names>M</given-names></name></person-group>. <article-title>WHO declares COVID-19 a pandemic</article-title>. <source>Acta Biomed.</source> (<year>2020</year>) <volume>91</volume>:<fpage>157</fpage>&#x02013;<lpage>160</lpage>. <pub-id pub-id-type="doi">10.23750/abm.v91i1.9397</pub-id><pub-id pub-id-type="pmid">32191675</pub-id></citation></ref>
<ref id="B7">
<label>7.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Box</surname> <given-names>GEP</given-names></name> <name><surname>Jenkins</surname> <given-names>GM</given-names></name> <name><surname>Reinsel</surname> <given-names>GC</given-names></name></person-group>. <article-title>Time series analysis: forecasting and control</article-title>. <source>J Market Res.</source> (<year>1977</year>) <volume>14</volume>:<fpage>269</fpage>. <pub-id pub-id-type="doi">10.2307/3150485</pub-id></citation>
</ref>
<ref id="B8">
<label>8.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Heisterkamp</surname> <given-names>SH</given-names></name> <name><surname>Dekkers</surname> <given-names>ALM</given-names></name> <name><surname>Heijne</surname> <given-names>JCM</given-names></name></person-group>. <article-title>Automated detection of infectious disease outbreaks: hierarchical time series models</article-title>. <source>Stat Med.</source> (<year>2006</year>) <volume>25</volume>:<fpage>4179</fpage>&#x02013;<lpage>96</lpage>. <pub-id pub-id-type="doi">10.1002/sim.2674</pub-id><pub-id pub-id-type="pmid">16958149</pub-id></citation></ref>
<ref id="B9">
<label>9.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Choi</surname> <given-names>K</given-names></name> <name><surname>thacker</surname> <given-names>SB</given-names></name></person-group>. <article-title>An evaluation of influenza mortality surveillance, 1962&#x02013;1979</article-title>. <source>Am J Epidemiol.</source> (<year>1981</year>) <volume>113</volume>:<fpage>215</fpage>&#x02013;<lpage>26</lpage>. <pub-id pub-id-type="doi">10.1093/oxfordjournals.aje.a113090</pub-id><pub-id pub-id-type="pmid">6258426</pub-id></citation></ref>
<ref id="B10">
<label>10.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Benvenuto</surname> <given-names>D</given-names></name> <name><surname>Giovanetti</surname> <given-names>M</given-names></name> <name><surname>Vassallo</surname> <given-names>L</given-names></name> <name><surname>Angeletti</surname> <given-names>S</given-names></name> <name><surname>Ciccozzi</surname> <given-names>M</given-names></name></person-group>. <article-title>Application of the ARIMA model on the COVID-2019 epidemic dataset</article-title>. <source>Data Brief.</source> (<year>2020</year>) <volume>29</volume>:<fpage>105340</fpage>. <pub-id pub-id-type="doi">10.1016/j.dib.2020.105340</pub-id><pub-id pub-id-type="pmid">32181302</pub-id></citation></ref>
<ref id="B11">
<label>11.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ceylan</surname> <given-names>Z</given-names></name></person-group>. <article-title>Estimation of COVID-19 prevalence in Italy, Spain, and France</article-title>. <source>Sci Total Environ.</source> (<year>2020</year>) <volume>729</volume>:<fpage>138817</fpage>. <pub-id pub-id-type="doi">10.1016/j.scitotenv.2020.138817</pub-id><pub-id pub-id-type="pmid">32360907</pub-id></citation></ref>
<ref id="B12">
<label>12.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chintalapudi</surname> <given-names>N</given-names></name> <name><surname>Battineni</surname> <given-names>G</given-names></name> <name><surname>Amenta</surname> <given-names>F</given-names></name></person-group>. <article-title>COVID-19 virus outbreak forecasting of registered and recovered cases after sixty day lockdown in Italy: a data driven model approach</article-title>. <source>J Microbiol Immunol Infect.</source> (<year>2020</year>) <volume>53</volume>:<fpage>396</fpage>&#x02013;<lpage>403</lpage>. <pub-id pub-id-type="doi">10.1016/j.jmii.2020.04.004</pub-id><pub-id pub-id-type="pmid">32305271</pub-id></citation></ref>
<ref id="B13">
<label>13.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alzahrani</surname> <given-names>SI</given-names></name> <name><surname>Aljamaan</surname> <given-names>IA</given-names></name> <name><surname>Al-Fakih</surname> <given-names>EA</given-names></name></person-group>. <article-title>Forecasting the spread of the COVID-19 pandemic in Saudi Arabia using ARIMA prediction model under current public health interventions</article-title>. <source>J Infect Public Health.</source> (<year>2020</year>) <volume>13</volume>:<fpage>914</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1016/j.jiph.2020.06.001</pub-id><pub-id pub-id-type="pmid">32546438</pub-id></citation></ref>
<ref id="B14">
<label>14.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chaurasia</surname> <given-names>V</given-names></name> <name><surname>Pal</surname> <given-names>S</given-names></name></person-group>. <article-title>COVID-19 Pandemic: ARIMA and Regression Model-Based Worldwide Death Cases predictions</article-title>. <source>SN Comput Sci.</source> (<year>2020</year>) <volume>1</volume>:<fpage>288</fpage>. <pub-id pub-id-type="doi">10.1007/s42979-020-00298-6</pub-id><pub-id pub-id-type="pmid">33063056</pub-id></citation></ref>
<ref id="B15">
<label>15.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chaurasia</surname> <given-names>V</given-names></name> <name><surname>Pal</surname> <given-names>S</given-names></name></person-group>. <article-title>Application of machine learning time series analysis for prediction COVID-19 pandemic</article-title>. <source>Res Biomed Eng.</source> (<year>2020</year>) <volume>24</volume>:<fpage>1</fpage>&#x02013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1007/s42600-020-00105-4</pub-id></citation>
</ref>
<ref id="B16">
<label>16.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hernandez-Matamoros</surname> <given-names>A</given-names></name> <name><surname>Fujita</surname> <given-names>H</given-names></name> <name><surname>Hayashi</surname> <given-names>T</given-names></name> <name><surname>Perez-Meana</surname> <given-names>H</given-names></name></person-group>. <article-title>Forecasting of COVID19 per regions using ARIMA models and polynomial functions</article-title>. <source>Appl Soft Comput J.</source> (<year>2020</year>) <volume>96</volume>:<fpage>106610</fpage>. <pub-id pub-id-type="doi">10.1016/j.asoc.2020.106610</pub-id><pub-id pub-id-type="pmid">32834798</pub-id></citation></ref>
<ref id="B17">
<label>17.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sahai</surname> <given-names>AK</given-names></name> <name><surname>Rath</surname> <given-names>N</given-names></name> <name><surname>Sood</surname> <given-names>V</given-names></name> <name><surname>Singh</surname> <given-names>MP</given-names></name></person-group>. <article-title>ARIMA modelling &#x00026; forecasting of COVID-19 in top five affected countries</article-title>. <source>Diabetes Metab Syndrome Clin Res Rev.</source> (<year>2020</year>) <volume>14</volume>:<fpage>1419</fpage>&#x02013;<lpage>27</lpage>. <pub-id pub-id-type="doi">10.1016/j.dsx.2020.07.042</pub-id><pub-id pub-id-type="pmid">32755845</pub-id></citation></ref>
<ref id="B18">
<label>18.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Y</given-names></name> <name><surname>Xu</surname> <given-names>C</given-names></name> <name><surname>Yao</surname> <given-names>S</given-names></name> <name><surname>Zhao</surname> <given-names>Y</given-names></name></person-group>. <article-title>Forecasting the epidemiological trends of COVID-19 prevalence and mortality using the advanced &#x003B1;-Sutte Indicator</article-title>. <source>Epidemiol Infect.</source> (<year>2020</year>) <volume>148</volume>: <pub-id pub-id-type="doi">10.1017/S095026882000237X</pub-id><pub-id pub-id-type="pmid">33012300</pub-id></citation></ref>
<ref id="B19">
<label>19.</label>
<citation citation-type="journal"><person-group person-group-type="author"><collab>M. WB, A. HL. Modeling and forecasting vehicular traffic flow as a seasonal ARIMA process: theoretical basis and empirical results</collab></person-group>. <source>J Transport Eng.</source> (<year>2003</year>) <volume>129</volume>:<fpage>664</fpage>&#x02013;<lpage>72</lpage>. <pub-id pub-id-type="doi">10.1061/(ASCE)0733-947X(</pub-id><year>2003</year>)129:6(664)</citation>
</ref>
<ref id="B20">
<label>20.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chandola</surname> <given-names>V</given-names></name> <name><surname>Banerjee</surname> <given-names>A</given-names></name> <name><surname>Kumar</surname> <given-names>V</given-names></name></person-group>. <article-title>Anomaly detection: a survey</article-title>. <source>ACM Comput Surveys.</source> (<year>2009</year>) <volume>41</volume>:<fpage>1</fpage>&#x02013;<lpage>58</lpage>. <pub-id pub-id-type="doi">10.1145/1541880.1541882</pub-id></citation>
</ref>
<ref id="B21">
<label>21.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Yeh</surname> <given-names>C-CM</given-names></name> <name><surname>Zhu</surname> <given-names>Y</given-names></name> <name><surname>Ulanova</surname> <given-names>L</given-names></name> <name><surname>Begum</surname> <given-names>N</given-names></name> <name><surname>Ding</surname> <given-names>Y</given-names></name> <name><surname>Dau</surname> <given-names>HA</given-names></name> <etal/></person-group>. <article-title>Matrix profile I: all pairs similarity joins for time series: a unifying view that includes motifs, discords and shapelets</article-title>. In: <source>2016 IEEE 16th International Conference on Data Mining (ICDM)</source>. <publisher-loc>Barcelona</publisher-loc>: <publisher-name>Institute of Electrical and Electronics Engineers (IEEE)</publisher-name> (<year>2016</year>). p. <fpage>1317</fpage>&#x02013;<lpage>22</lpage>.</citation>
</ref>
<ref id="B22">
<label>22.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>Y</given-names></name> <name><surname>Zimmerman</surname> <given-names>Z</given-names></name> <name><surname>Senobari</surname> <given-names>NS</given-names></name> <name><surname>Yeh</surname> <given-names>C-CM</given-names></name> <name><surname>Funning</surname> <given-names>G</given-names></name> <name><surname>Mueen</surname> <given-names>A</given-names></name> <etal/></person-group>. <article-title>Matrix profile II: exploiting a novel algorithm and gpus to break the one hundred million barrier for time series motifs and joins</article-title>. In: <source>2016 IEEE 16th International Conference on Data Mining (ICDM)</source>. <publisher-loc>Barcelona</publisher-loc>: <publisher-name>Institute of Electrical and Electronics Engineers (IEEE)</publisher-name> (<year>2016</year>),<fpage>739</fpage>&#x02013;<lpage>48</lpage>.</citation>
</ref>
<ref id="B23">
<label>23.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Yeh</surname> <given-names>C-CM</given-names></name> <name><surname>Herle</surname> <given-names>H</given-names></name> <name><surname>Van Keogh</surname> <given-names>E</given-names></name></person-group>. <article-title>Matrix profile III: the matrix profile allows visualization of salient subsequences in massive time series</article-title>. In: : <source>2016 IEEE 16th International Conference on Data Mining (ICDM)</source>. <publisher-loc>Barcelona</publisher-loc>: <publisher-name>Institute of Electrical and Electronics Engineers (IEEE)</publisher-name>.p. <fpage>579</fpage>&#x02013;<lpage>88</lpage>.</citation>
</ref>
<ref id="B24">
<label>24.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yeh</surname> <given-names>CCM</given-names></name> <name><surname>Zhu</surname> <given-names>Y</given-names></name> <name><surname>Ulanova</surname> <given-names>L</given-names></name> <name><surname>Begum</surname> <given-names>N</given-names></name> <name><surname>Ding</surname> <given-names>Y</given-names></name> <name><surname>Anh</surname> <given-names>H</given-names></name> <etal/></person-group>. <article-title>Time series joins, motifs, discords and shapelets: a unifying view that exploits the matrix profile</article-title>. <source>Data Mining Knowl Discov.</source> (<year>2018</year>) <volume>32</volume>:<fpage>83</fpage>&#x02013;<lpage>123</lpage>. <pub-id pub-id-type="doi">10.1007/s10618-017-0519-9</pub-id></citation>
</ref>
<ref id="B25">
<label>25.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yeh</surname> <given-names>CCM</given-names></name> <name><surname>Kavantzas</surname> <given-names>N</given-names></name> <name><surname>Keogh</surname> <given-names>E</given-names></name></person-group>. <article-title>Matrix profile IV: using weakly labeled time series to predict outcomes</article-title>. <source>Proc VLDB Endow.</source> (<year>2017</year>) <volume>10</volume>:<fpage>1802</fpage>&#x02013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.14778/3137765.3137784</pub-id></citation>
</ref>
<ref id="B26">
<label>26.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jaeger</surname> <given-names>H</given-names></name> <name><surname>Haas</surname> <given-names>H</given-names></name></person-group>. <article-title>Harnessing nonlinearity: predicting chaotic systems and saving energy in wireless communication</article-title>. <source>Science.</source> (<year>2004</year>) <volume>304</volume>:<fpage>78</fpage>&#x02013;<lpage>80</lpage>. <pub-id pub-id-type="doi">10.1126/science.1091277</pub-id><pub-id pub-id-type="pmid">15064413</pub-id></citation></ref>
<ref id="B27">
<label>27.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cho</surname> <given-names>K</given-names></name> <name><surname>Van</surname> <given-names>Merri&#x000EB;nboer B</given-names></name> <name><surname>Gulcehre</surname> <given-names>C</given-names></name> <name><surname>Bahdanau</surname> <given-names>D</given-names></name> <name><surname>Bougares</surname> <given-names>F</given-names></name> <name><surname>Schwenk</surname> <given-names>H</given-names></name> <etal/></person-group>. <article-title>Learning phrase representations using RNN encoder-decoder for statistical machine translation</article-title>. In: <source>EMNLP 2014 - 2014 Conference on Empirical Methods in Natural Language Processing, Proceedings of the Conference</source>. Doha, Qatar, Association for Computational Linguistics (<year>2014</year>). p. <fpage>1724</fpage>&#x02013;<lpage>34</lpage>. <pub-id pub-id-type="doi">10.3115/v1/D14-1179</pub-id></citation>
</ref>
<ref id="B28">
<label>28.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hochreiter</surname> <given-names>S</given-names></name> <name><surname>Schmidhuber</surname> <given-names>J</given-names></name></person-group>. <article-title>Long short-term memory</article-title>. <source>Neural Comput.</source> (<year>1997</year>) <volume>9</volume>:<fpage>1735</fpage>&#x02013;<lpage>80</lpage>. <pub-id pub-id-type="doi">10.1162/neco.1997.9.8.1735</pub-id><pub-id pub-id-type="pmid">9377276</pub-id></citation></ref>
<ref id="B29">
<label>29.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hochreiter</surname> <given-names>S</given-names></name></person-group>. <article-title>The vanishing gradient problem during learning recurrent neural nets and problem solutions</article-title>. <source>Int J Uncertain Fuzziness Knowl Based Syst.</source> (<year>1998</year>) <volume>6</volume>:<fpage>107</fpage>&#x02013;<lpage>16</lpage>. <pub-id pub-id-type="doi">10.1142/S0218488598000094</pub-id></citation>
</ref>
<ref id="B30">
<label>30.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Oztuik</surname> <given-names>MC</given-names></name> <name><surname>Xu</surname> <given-names>D</given-names></name> <name><surname>Principe</surname> <given-names>JC</given-names></name></person-group>. <article-title>Analysis and design of echo state networks</article-title>. <source>Neural Comput.</source> (<year>2007</year>) <volume>19</volume>:<fpage>111</fpage>&#x02013;<lpage>38</lpage>. <pub-id pub-id-type="doi">10.1162/neco.2007.19.1.111</pub-id><pub-id pub-id-type="pmid">17134319</pub-id></citation></ref>
<ref id="B31">
<label>31.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shahid</surname> <given-names>F</given-names></name> <name><surname>Zameer</surname> <given-names>A</given-names></name> <name><surname>Muneeb</surname> <given-names>M</given-names></name></person-group>. <article-title>Predictions for COVID-19 with deep learning models of LSTM, GRU and Bi-LSTM</article-title>. <source>Chaos Solitons Fractals.</source> (<year>2020</year>) <volume>140</volume>:<fpage>110212</fpage>. <pub-id pub-id-type="doi">10.1016/j.chaos.2020.110212</pub-id><pub-id pub-id-type="pmid">32839642</pub-id></citation></ref>
<ref id="B32">
<label>32.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chimmula</surname> <given-names>VKR</given-names></name> <name><surname>Zhang</surname> <given-names>L</given-names></name></person-group>. <article-title>Time series forecasting of COVID-19 transmission in Canada using LSTM networks</article-title>. <source>Chaos Solitons Fractals.</source> (<year>2020</year>) <volume>135</volume>:<fpage>109864</fpage>. <pub-id pub-id-type="doi">10.1016/j.chaos.2020.109864</pub-id><pub-id pub-id-type="pmid">32390691</pub-id></citation></ref>
<ref id="B33">
<label>33.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Barman</surname> <given-names>A</given-names></name></person-group>. <article-title>Time series analysis and forecasting of COVID-19 cases using LSTM and ARIMA models[J]</article-title>. (<year>2020</year>). <source>arXiv [Preprint]. arXiv</source> 2006.13852</citation>
</ref>
<ref id="B34">
<label>34.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shastri</surname> <given-names>S</given-names></name> <name><surname>Singh</surname> <given-names>K</given-names></name> <name><surname>Kumar</surname> <given-names>S</given-names></name> <name><surname>Kour</surname> <given-names>P</given-names></name> <name><surname>Mansotra</surname> <given-names>V</given-names></name></person-group>. <article-title>Time series forecasting of Covid-19 using deep learning models: India-USA comparative case study</article-title>. <source>Chaos Solitons Fractals.</source> (<year>2020</year>) <volume>140</volume>:<fpage>110227</fpage>. <pub-id pub-id-type="doi">10.1016/j.chaos.2020.110227</pub-id><pub-id pub-id-type="pmid">32843824</pub-id></citation></ref>
<ref id="B35">
<label>35.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Y</given-names></name> <name><surname>Huang</surname> <given-names>M</given-names></name> <name><surname>Zhao</surname> <given-names>L</given-names></name> <name><surname>Zhu</surname> <given-names>X</given-names></name></person-group>. <article-title>Attention-based LSTM for aspect-level sentiment classification</article-title>. In: <source>EMNLP 2016 - Conference on Empirical Methods in Natural Language Processing, Proceedings</source>. Austin, Texas, Association for Computational Linguistics (<year>2016</year>). p. <fpage>606</fpage>&#x02013;<lpage>15</lpage>. <pub-id pub-id-type="doi">10.18653/v1/D16-1058</pub-id></citation>
</ref>
<ref id="B36">
<label>36.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>J</given-names></name> <name><surname>Yao</surname> <given-names>T</given-names></name> <name><surname>Zhang</surname> <given-names>Y</given-names></name> <name><surname>Mei</surname> <given-names>T</given-names></name></person-group>. <article-title>Learning multimodal attention LSTM networks for video captioning</article-title>. In: <source>MM 2017 - Proceedings of the 2017 ACM Multimedia Conference</source>. New York, United States, Association for Computing Machinery (<year>2017</year>). p. <fpage>537</fpage>&#x02013;<lpage>45</lpage>. <pub-id pub-id-type="doi">10.1145/3123266.3123448</pub-id></citation>
</ref>
<ref id="B37">
<label>37.</label>
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Farrow</surname> <given-names>DC</given-names></name> <name><surname>Brooks</surname> <given-names>LC</given-names></name> <name><surname>Rumack</surname> <given-names>A</given-names></name> <name><surname>Tibshirani</surname> <given-names>RJ</given-names></name> <name><surname>Rosenfeld</surname> <given-names>R</given-names></name></person-group>. <source>Delphi Epidata API</source>. Delphi Research Group at Carnegie Mellon University, Available online at: <ext-link ext-link-type="uri" xlink:href="http://cmu-delphi.github.io/delphi-epidata/api/covidcast.html">cmu-delphi.github.io/delphi-epidata/api/covidcast.html</ext-link> (<year>2021</year>).</citation>
</ref>
<ref id="B38">
<label>38.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Van Benschoten</surname> <given-names>A</given-names></name> <name><surname>Ouyang</surname> <given-names>A</given-names></name> <name><surname>Bischoff</surname> <given-names>F</given-names></name> <name><surname>Marrs</surname> <given-names>T</given-names></name></person-group>. <article-title>MPA: a novel cross-language API for time series analysis</article-title>. <source>J Open Source Softw.</source> (<year>2020</year>) <volume>5</volume>:<fpage>2179</fpage>. <pub-id pub-id-type="doi">10.21105/joss.02179</pub-id></citation>
</ref>
<ref id="B39">
<label>39.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hyndman</surname> <given-names>RJ</given-names></name> <name><surname>Khandakar</surname> <given-names>Y</given-names></name></person-group>. <article-title>Automatic time series forecasting: the forecast package for R</article-title>. <source>J Stat Softw.</source> (<year>2008</year>) <volume>27</volume>:<fpage>1</fpage>&#x02013;<lpage>22</lpage>. <pub-id pub-id-type="doi">10.18637/jss.v027.i03</pub-id></citation>
</ref>
<ref id="B40">
<label>40.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Petersen</surname> <given-names>NC</given-names></name> <name><surname>Christoffer</surname> <given-names>R</given-names></name> <name><surname>Rodrigues</surname> <given-names>F</given-names></name> <name><surname>Pereira</surname> <given-names>FC</given-names></name></person-group>. <article-title>Multi-output bus travel time prediction with convolutional LSTM neural network</article-title>. <source>Expert Syst Applic.</source> (<year>2019</year>) <volume>120</volume>:<fpage>426</fpage>&#x02013;<lpage>35</lpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2018.11.028</pub-id></citation>
</ref>
</ref-list>
</back>
</article>