<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="review-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Physiol.</journal-id>
<journal-title>Frontiers in Physiology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Physiol.</abbrev-journal-title>
<issn pub-type="epub">1664-042X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1386760</article-id>
<article-id pub-id-type="doi">10.3389/fphys.2024.1386760</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Physiology</subject>
<subj-group>
<subject>Review</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Predictive modeling of biomedical temporal data in healthcare applications: review and future directions</article-title>
<alt-title alt-title-type="left-running-head">Patharkar et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fphys.2024.1386760">10.3389/fphys.2024.1386760</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Patharkar</surname>
<given-names>Abhidnya</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2554577/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Cai</surname>
<given-names>Fulin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2272174/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Al-Hindawi</surname>
<given-names>Firas</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2859575/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Wu</surname>
<given-names>Teresa</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1674281/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Computing and Augmented Intelligence</institution>, <institution>Arizona State University</institution>, <addr-line>Tempe</addr-line>, <addr-line>AZ</addr-line>, <country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>ASU-Mayo Center for Innovative Imaging</institution>, <institution>Arizona State University</institution>, <addr-line>Tempe</addr-line>, <addr-line>AZ</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2381054/overview">Hyo Kyung Lee</ext-link>, Korea University, Republic of Korea</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1315523/overview">Ricardo Valentim</ext-link>, Federal University of Rio Grande do Norte, Brazil</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2054538/overview">Xu Huang</ext-link>, Nanjing University of Science and Technology, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Teresa Wu, <email>Teresa.Wu@asu.edu</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>15</day>
<month>10</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1386760</elocation-id>
<history>
<date date-type="received">
<day>16</day>
<month>02</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>26</day>
<month>09</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Patharkar, Cai, Al-Hindawi and Wu.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Patharkar, Cai, Al-Hindawi and Wu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Predictive modeling of clinical time series data is challenging due to various factors. One such difficulty is the existence of missing values, which leads to irregular data. Another challenge is capturing correlations across multiple dimensions in order to achieve accurate predictions. Additionally, it is essential to take into account the temporal structure, which includes both short-term and long-term recurrent patterns, to gain a comprehensive understanding of disease progression and to make accurate predictions for personalized healthcare. In critical situations, models that can make multi-step ahead predictions are essential for early detection. This review emphasizes the need for forecasting models that can effectively address the aforementioned challenges. The selection of models must also take into account the data-related constraints during the modeling process. Time series models can be divided into statistical, machine learning, and deep learning models. This review concentrates on the main models within these categories, discussing their capability to tackle the mentioned challenges. Furthermore, this paper provides a brief overview of a technique aimed at mitigating the limitations of a specific model to enhance its suitability for clinical prediction. It also explores ensemble forecasting methods designed to merge the strengths of various models while reducing their respective weaknesses, and finally discusses hierarchical models. Apart from the technical details provided in this document, there are certain aspects in predictive modeling research that have arisen as possible obstacles in implementing models using biomedical data. These obstacles are discussed leading to the future prospects of model building with artificial intelligence in healthcare domain.</p>
</abstract>
<kwd-group>
<kwd>biomedical temporal data</kwd>
<kwd>biomedical data challenges</kwd>
<kwd>forecasting</kwd>
<kwd>clinical predictive modeling</kwd>
<kwd>temporal data modeling problems</kwd>
<kwd>statistical time-series models</kwd>
<kwd>temporal machine learning models</kwd>
<kwd>deep temporal models</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Physiology and Medicine</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<sec id="s1-1">
<title>1.1 Biomedical time series data</title>
<p>Clinical or biomedical data advances medical research by providing insights into patient health, disease progression, and treatment efficacy. It underpins new diagnostics, therapies, and personalized medicine, improving outcomes and understanding complex conditions. In predictive modeling, biomedical data is categorized as spatial, temporal, and spatio-temporal (<xref ref-type="bibr" rid="B76">Khalique et al., 2020</xref>; <xref ref-type="bibr" rid="B139">Veneri et al., 2012</xref>). Temporal data is key, capturing health evolution over time and offering insights into disease progression and treatment effectiveness. Time series data, collected at successive time points, shows complex patterns with short- and long-term dependencies, crucial for forecasting and analysis (<xref ref-type="bibr" rid="B160">Zou et al., 2019</xref>; <xref ref-type="bibr" rid="B77">Lai et al., 2018</xref>). Properly harnessed, this data advances personalized medicine and treatment optimization, making it essential in contemporary research.</p>
</sec>
<sec id="s1-2">
<title>1.2 Applications of predictive modeling in biomedical time series analysis</title>
<p>Predictive modeling with artificial intelligence (AI) has gained significant traction across various domains, including manufacturing (<xref ref-type="bibr" rid="B9">Altarazi et al., 2019</xref>), heat transfer (<xref ref-type="bibr" rid="B4">Al-Hindawi et al., 2023</xref>; <xref ref-type="bibr" rid="B3">2024</xref>), energy systems (<xref ref-type="bibr" rid="B63">Huang et al., 2024</xref>), and notably, the biomedical field (<xref ref-type="bibr" rid="B22">Cai et al., 2023</xref>; <xref ref-type="bibr" rid="B101">Patharkar et al., 2024</xref>). Predictive modeling in biomedical time series data involves various approaches for specific predictions and data characteristics. Forecasting models predict future outcomes based on historical data, such as forecasting blood glucose levels for diabetic patients using past measurements, insulin doses, and dietary information (<xref ref-type="bibr" rid="B107">Plis et al., 2014</xref>). Classification models predict categorical outcomes, like detecting cardiac arrhythmias from ECG data by classifying segments into categories such as normal, atrial fibrillation, or other arrhythmias, aiding in early diagnosis and treatment (<xref ref-type="bibr" rid="B29">Daydulo et al., 2023</xref>; <xref ref-type="bibr" rid="B158">Zhou et al., 2019</xref>; <xref ref-type="bibr" rid="B28">Chuah and Fu, 2007</xref>). Anomaly detection in biomedical time series identifies outliers or abnormal patterns, signifying unusual events or conditions. For example, monitoring ICU patients&#x2019; vital signs can detect early signs of sepsis (<xref ref-type="bibr" rid="B94">Mollura et al., 2021</xref>; <xref ref-type="bibr" rid="B123">Shashikumar et al., 2017</xref>; <xref ref-type="bibr" rid="B93">Mitra and Ashraf, 2018</xref>), enabling timely intervention.</p>
<p>
<xref ref-type="table" rid="T1">Table 1</xref> summarizes the example applications of these models within the context of biomedical time series.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Overview of predictive modeling techniques for biomedical time series and their example applications across healthcare scenarios.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model type</th>
<th align="center">Description</th>
<th align="center">Bio-medical application example</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Forecasting</td>
<td align="center">Predicts a continuous value based on historical data</td>
<td align="center">Predicting blood glucose levels for diabetic patients using past glucose measurements, insulin doses, and dietary information to forecast potential hypo- or hyperglycemic events (<xref ref-type="bibr" rid="B107">Plis et al., 2014</xref>)</td>
</tr>
<tr>
<td align="center">Classification</td>
<td align="center">Predicts categorical outcomes based on temporal data</td>
<td align="center">Detecting cardiac arrhythmias (such as normal, atrial fibrillation, or other arrhythmias) (<xref ref-type="bibr" rid="B29">Daydulo et al., 2023</xref>; <xref ref-type="bibr" rid="B158">Zhou et al., 2019</xref>; <xref ref-type="bibr" rid="B28">Chuah and Fu, 2007</xref>)</td>
</tr>
<tr>
<td align="center">Anomaly Detection</td>
<td align="center">Identifies outliers or abnormal patterns within time series data</td>
<td align="center">Sepsis detection. (<xref ref-type="bibr" rid="B94">Mollura et al., 2021</xref>; <xref ref-type="bibr" rid="B123">Shashikumar et al., 2017</xref>; <xref ref-type="bibr" rid="B93">Mitra and Ashraf, 2018</xref>)</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s1-3">
<title>1.3 Challenges in biomedical time series data</title>
<p>Regardless of the particular medical application or predictive model type used, models that manage biomedical time series data must tackle the intrinsic challenges posed by clinical and biomedical data. This includes various categories, such as electronic health records (EHRs), administrative data, claims data, patient/disease registries, health surveys, and clinical trials data. As illustrated in <xref ref-type="table" rid="T2">Table 2</xref>, each biomedical data category presents distinct challenges regarding quality, privacy, and completeness. During predictive modeling, further challenges arise. Specifically, we will investigate problems associated with missing data and imputation methods, the intricate nature of high-dimensional temporal relationships, and factors concerning the size of the dataset. Addressing these issues is crucial for developing strong and accurate predictive models in medical research.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Overview of clinical data types and challenges. This table lists the main types of clinical and biomedical data, their definitions, and key challenges.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Data type</th>
<th align="center">Definition</th>
<th align="center">Challenges</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Electronic Health Records (EHRs)</td>
<td align="center">Digital records of patients medical history, treatments, and outcomes</td>
<td align="left">&#x2022; Data Standardization: Different formats across providers &#x2022; Data Quality: Missing, incomplete, or inaccurate data &#x2022; Privacy and Security: Ensuring compliance with regulations like HIPAA &#x2022; Interoperability: Difficulties in data exchange between systems</td>
</tr>
<tr>
<td align="center">Administrative Data</td>
<td align="center">Data related to healthcare administration, such as hospital admissions and discharge records</td>
<td align="left">&#x2022; Limited Clinical Detail: Lack of in-depth clinical information &#x2022; Data Timeliness: Potential delays in data availability &#x2022; Standardization Issues: Variability in recording and categorization &#x2022; Privacy Concerns: Maintaining patient confidentiality</td>
</tr>
<tr>
<td align="center">Claims Data</td>
<td align="center">Data from insurance claims used for billing and reimbursement</td>
<td align="left">&#x2022; Purpose and Detail: Primarily for billing, may lack clinical details &#x2022; Lag Time: Delays between care and data availability &#x2022; Coding Errors: Inaccuracies in coding (e.g., ICD codes) &#x2022; Complexity: Requires specialized knowledge for interpretation</td>
</tr>
<tr>
<td align="center">Patient/Disease Registries</td>
<td align="center">Databases that track patients with specific conditions or diseases</td>
<td align="left">&#x2022; Data Completeness: Ensuring all relevant data is captured &#x2022; Data Standardization: Different definitions and methods across registries &#x2022; Funding and Maintenance: Need for consistent resources &#x2022; Privacy Issues: Protecting patient confidentiality</td>
</tr>
<tr>
<td align="center">Health Surveys</td>
<td align="center">Data collected from health-related surveys and questionnaires</td>
<td align="left">&#x2022; Response Bias: Non-response or inaccurate self-reporting &#x2022; Sampling Issues: Ensuring representative samples &#x2022; Data Quality: Depends on survey design and execution &#x2022; Timeliness: Time-consuming design, conduct, and analysis</td>
</tr>
<tr>
<td align="center">Clinical Trials Data</td>
<td align="center">Data from controlled trials testing the efficacy of treatments or interventions</td>
<td align="left">&#x2022; Complexity and Cost: Expensive and logistically complex &#x2022; Regulatory Hurdles: Compliance with regulatory requirements &#x2022; Data Sharing: Balancing patient confidentiality and proprietary interests &#x2022; Generalizability: Trial participants may not represent the broader population</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s1-3-1">
<title>1.3.1 Challenges in handling missing values and imputation methods in biomedical time series</title>
<p>Clinical data is often confronted with the issue of missing values, which can be caused by irregular data collection schedules or unexpected events (<xref ref-type="bibr" rid="B150">Xu et al., 2020</xref>). Medical measurements, recorded variably and at different times, may be absent, not documented, or affected by recording errors (<xref ref-type="bibr" rid="B97">Mulyadi et al., 2022</xref>), which makes the data irregular. Dealing with missing values in data sets usually involves either directly modeling data sets with missing values or filling in the missing values (a.k.a. imputation) to complete datasets for traditional analysis methods using data imputation techniques.</p>
<p>Current imputation techniques can be divided into four categories: case deletion, basic statistical imputation, machine learning-based imputation (<xref ref-type="bibr" rid="B89">Luo et al., 2018</xref>), and aggregating irregularly sampled data into discrete time periods (<xref ref-type="bibr" rid="B48">Ghaderi et al., 2023</xref>). Each of these methods comes with specific challenges in the context of handling biomedical temporal data. The deletion or omission of cases may lead to the loss of important information, particularly when the rate of missingness is high, which is critical in sensitive applications such as biomedical predictive modeling, where data is scarce and human lives are at risk. However, in certain cases, it is possible to do data omission without any potential risk to the outcome of the study. For instance, (<xref ref-type="bibr" rid="B106">Pinto et al., 2022</xref>), employs interrupted time series analysis to assess the impact of the &#x201c;Syphilis No!&#x201d; initiative in reducing congenital syphilis rates in Brazil. The results indicate significant declines in priority municipalities after the intervention. The study showcases the efficacy of public health interventions in modifying disease trends using statistical analysis of temporal data. Data collection needed to be conducted consistently over time and at evenly spaced intervals for proper analysis. To prevent bias due to the COVID-19 pandemic, December 2019 was set as the final data collection point, encompassing 20 months before the intervention (September 2016 to April 2018) and 20 months after the intervention (May 2018 to December 2019). This approach illustrates how the author addressed potential issues of irregular data or missing values in this context.</p>
<p>Contrary to data omission, statistical imputation techniques, such as mean or median imputation offer an alternative that reduces the effect of missing data, however, such methods do not take into account the temporal information but rather offer a summarized statistical imputation that often does not provide accurate replacement of the missing data. This could be critical in biomedical applications with scarce datasets, where the weight of a single data point could heavily affect the predictive power of the model. The use of machine learning-based imputation methods, such as Maximum Likelihood Expectation-Maximization, k-Nearest Neighbors, and Matrix Factorization, might offer a more accurate imputation that takes into account the specificity of the data point contrary to statistical aggregation methods, however, many of them still do not consider temporal relations between observations (<xref ref-type="bibr" rid="B89">Luo et al., 2018</xref>; <xref ref-type="bibr" rid="B70">Jun et al., 2019</xref>), and they usually are computationally expensive. Furthermore, without incorporating domain knowledge, these approaches can introduce bias and lead to invalid conclusions. Both machine learning and statistical techniques may not consider data distribution or variable relationships and may fail to capture complex patterns in multivariate time-series data due to the neglect of correlated variables, potentially resulting in underestimated or overestimated imputed values (Jun et al., 2019). Additionally, in real-time clinical decision support systems, timely and accurate data is crucial, as delays or errors in imputation can lead to incorrect decisions that directly affect patient outcomes. These systems demand high-speed processing, requiring imputation algorithms to be both computationally efficient and accurate. Moreover, the dynamic nature of clinical environments, where patient conditions can change rapidly, necessitates imputation methods that can adapt quickly to evolving data.</p>
<p>Aggregating measurements into discrete time periods can address irregular intervals, but it may lead to a loss of granular information (<xref ref-type="bibr" rid="B48">Ghaderi et al., 2023</xref>). Additionally, in time series prediction, missing values and their patterns are often correlated with target labels, referred to as informative missingness (<xref ref-type="bibr" rid="B26">Che et al., 2018</xref>). These limitations make it ill-advised to ignore, impute, or aggregate these values when handling biomedical time series data, but rather employ a model that is capable of handling the sparsity and the irregularity of clinical time series data.</p>
</sec>
<sec id="s1-3-2">
<title>1.3.2 Complexities of high-dimensional temporal dependencies in biomedical data</title>
<p>Besides missing data challenges, hospitalized patients have a wide range of clinical events that are recorded in their electronic health records (EHRs). EHRs contain two different kinds of data: structured information, like diagnoses, treatments, medication prescriptions, vital signs, and laboratory tests, and unstructured data, like clinical notes and physiological signals (<xref ref-type="bibr" rid="B149">Xie et al., 2022</xref>; <xref ref-type="bibr" rid="B78">Lee and Hauskrecht, 2021</xref>), making them multivariate or high-dimensional (<xref ref-type="bibr" rid="B99">Niu et al., 2022</xref>).</p>
<p>The complexity of the relationships existing in such high-dimensional multivariate time series data can be difficult to capture and analyze. Analysts often try to predict future outcomes based on past data, and the accuracy of these predictions depends on how well the interdependencies between the various series are modeled (<xref ref-type="bibr" rid="B124">Shih et al., 2019</xref>). It is often beneficial to consider all relevant variables together rather than focusing on individual variables to build a prediction model, as this provides a comprehensive understanding of correlations in multivariate time series (MTS) data (<xref ref-type="bibr" rid="B34">Du et al., 2020</xref>). It thus becomes a requirement for predictive models employed in biomedical applications to take into account correlations among multiple dimensions and make predictions accordingly. It is equally crucial to ensure that only the features with a direct impact on the outcome are considered in the analysis. For instance, the study by <xref ref-type="bibr" rid="B12">Barreto et al. (2023)</xref> investigates the deployment of machine learning and deep learning models to forecast patient outcomes and allocate beds efficiently during the COVID-19 crisis in Rio Grande do Norte, Brazil. Out of 20 available features, nine were chosen based on their clinical importance and their correlation with patient outcomes, selected through discussions with clinical experts to guarantee the model&#x2019;s accuracy and interpretability.</p>
<p>In addition to the inherent high dimensionality of biomedical data sourced from diverse platforms such as EHRs, wearable devices monitoring neurophysiological functions, and intensive care units tracking disease progression through physiological measurements (<xref ref-type="bibr" rid="B7">Allam et al., 2021</xref>), also display a natural temporal ordering. This temporal structure demands a specialized analytical approach distinct from that applied to non-temporal datasets (<xref ref-type="bibr" rid="B160">Zou et al., 2019</xref>). The temporal dependency adds significant complexity to modeling due to the presence of two distinct recurring patterns: short-term and long-term. For instance, short-term patterns may repeat daily, whereas long-term patterns might span quarterly or yearly intervals within the time series (<xref ref-type="bibr" rid="B77">Lai et al., 2018</xref>). Biomedical data often exhibit long-term dependencies, such as those seen in biosignals like electroencephalograms (EEGs) and electrocardiograms (ECGs), which may span tens of thousands of time steps or involve specific medical conditions such as acute kidney injury (AKI) leading to subsequent dialysis (<xref ref-type="bibr" rid="B130">Sun et al., 2021</xref>; <xref ref-type="bibr" rid="B78">Lee and Hauskrecht, 2021</xref>). Concurrently, short-term dependencies can manifest in immediate physiological responses to medical interventions, such as the administration of norepinephrine and subsequent changes in blood pressure (<xref ref-type="bibr" rid="B78">Lee and Hauskrecht, 2021</xref>). Another instance is presented by <xref ref-type="bibr" rid="B134">Valentim et al. (2022)</xref>, who have created a model to forecast congenital syphilis (CS) cases in Brazil based on maternal syphilis (MS) incidences. The model takes into account the probability of proper diagnosis and treatment during prenatal care. It integrates short-term dependencies by assessing the immediate effects of prenatal care on birth outcomes, and long-term dependencies by analyzing syphilis case trends over a 10-year period. This strategy aids in enhancing public health decision-making and syphilis prevention planning.</p>
<p>Analyzing these recurrent patterns and longitudinal structures in biomedical data is essential to facilitate the creation of time-based patient trajectory representations of health events that facilitate more precise disease progression modeling and personalized treatment predictions (<xref ref-type="bibr" rid="B7">Allam et al., 2021</xref>; <xref ref-type="bibr" rid="B149">Xie et al., 2022</xref>). By incorporating both short-term fluctuations and long-term trends, robust predictive models can uncover hidden patterns in patient health records, advancing our understanding and application of digital medicine. Failing to consider these recurrent patterns can undermine the accuracy of time series forecasting in biomedical contexts such as digital medicine, which involves continuous recording of health events over time.</p>
<p>Additionally, early detection of diseases is of paramount importance. This can be achieved by utilizing existing biomarkers along with advanced predictive modeling techniques, or by introducing new biomarkers or devices aimed at early disease detection. For instance, early diagnosis of osteoporosis is essential to mitigate the significant socioeconomic impacts of fractures and hospitalizations. The novel device, Osseus, as cited by <xref ref-type="bibr" rid="B2">Albuquerque et al. (2023)</xref>, addresses this by offering a cost-effective, portable screening method that uses electromagnetic waves. Osseus measures signal attenuation through the patient&#x27;s middle finger to predict changes in bone mineral density with the assistance of machine learning models. The advantages of using Osseus include enhanced accessibility to osteoporosis screening, reduced healthcare costs, and improved patient quality of life through timely intervention.</p>
</sec>
<sec id="s1-3-3">
<title>1.3.3 Dataset size considerations</title>
<p>The quantity of data available in a given dataset must be carefully considered, as it significantly influences model selection and overall analytical approach. For instance, when patients are admitted for brief periods, the clinical sequences generated are often fewer than 50 data points (<xref ref-type="bibr" rid="B85">Liu, 2016</xref>). Similarly, the number of data points for specific tests, such as mean corpuscular hemoglobin concentration (MCHC) lab results, can be limited due to the high cost of these tests, often resulting in less than 50 data points (<xref ref-type="bibr" rid="B86">Liu and Hauskrecht, 2015</xref>). Such limited data points pose challenges for predictive modeling, as models must be robust enough to derive meaningful insights from small samples without overfitting.</p>
<p>Conversely, some datasets may have a moderate sample length, ranging from 55 to 100 data points, such as the Physionet sepsis dataset (<xref ref-type="bibr" rid="B113">Reyna et al., 2019</xref>; <xref ref-type="bibr" rid="B114">2020</xref>; <xref ref-type="bibr" rid="B51">Goldberger et al., 2000</xref>). These moderate-sized datasets offer a balanced scenario where the data is sufficient to train more complex models, but still requires careful handling to avoid overfitting and ensure generalizability.</p>
<p>In other cases, datasets can be extensive, particularly when long-span time series data is collected via sensor devices. These devices continuously monitor physiological parameters, resulting in large datasets with thousands of time steps (<xref ref-type="bibr" rid="B85">Liu, 2016</xref>). For example, wearable devices tracking neurophysiological functions or intensive care unit monitors can generate vast amounts of data, providing a rich source of information for predictive modeling. However, handling such large datasets demands models that are computationally efficient and capable of capturing long-term dependencies and complex patterns within the data.</p>
<p>The amount of data available is a major factor in choosing the appropriate model. Sparse datasets require models that can effectively handle limited information, often necessitating advanced techniques for data augmentation and imputation to make the most out of available data. Moderate datasets allow for the application of more sophisticated models, including machine learning and deep learning techniques, provided they are carefully tuned to prevent overfitting. Large datasets, on the other hand, enable the use of highly complex models, such as deep neural networks, which can leverage the extensive data to uncover intricate patterns and relationships.</p>
</sec>
</sec>
<sec id="s1-4">
<title>1.4 Strategies in forecasting for biomedical time series data</title>
<p>While our discussion has generally revolved around the challenges in predictive modeling of biomedical temporal data, this review specifically emphasizes forecasting. From the earlier discourse, it is clear that a forecasting model for clinical or biomedical temporal data needs to adeptly manage missing, irregular, sparse, and multivariate data, while also considering its temporal properties and the capacity to model both short-term and long-term dependencies. The model should be able to make multi-step predictions, and the selection of a suitable model is determined by the amount of data available and the temporal length of the time series under consideration.</p>
<p>In this review, we initially examine three main categories of forecasting models: statistical, machine learning, and deep learning models. We look closely at the leading models within each category, assessing their ability to tackle the complexities of biomedical temporal data, including issues like data irregularity, sparsity, and the need to capture detailed temporal dependencies, alongside multi-step predictions. Since each category has its unique advantages as well as limitations in addressing the specific challenges of biomedical temporal datasets, other sets of models mentioned in the literature, known as hierarchical time series forecasting and combination or ensemble forecasting that merge the benefits of various forecasting models to produce more accurate forecasts are also covered.</p>
<p>The rest of the paper is structured as follows: In <xref ref-type="sec" rid="s2">Section 2</xref>, statistical models are introduced. <xref ref-type="sec" rid="s3">Section 3</xref> covers machine learning models, while <xref ref-type="sec" rid="s4">Section 4</xref> focuses on deep learning models. This is followed by <xref ref-type="sec" rid="s5">Section 5</xref>, which is a discussion section that summarizes the findings, discusses ensemble as well as hierarchical models, and explores future directions for the application of AI in clinical datasets. Finally, <xref ref-type="sec" rid="s6">Section 6</xref> concludes the paper.</p>
</sec>
</sec>
<sec id="s2">
<title>2 Statistical models</title>
<p>The most popular predictive statistical models for temporal data are Auto-Regressive Integrated Moving Average (ARIMA) models, Exponential Weighted Moving Average (EWMA) models, and Regression models which are reviewed in the following sections.</p>
<sec id="s2-1">
<title>2.1 Auto-Regressive Integrated Moving Average models</title>
<p>(<xref ref-type="bibr" rid="B152">Yule, 1927</xref>) proposed an autoregressive (AR) model, and (<xref ref-type="bibr" rid="B146">Wold, 1948</xref>) introduced the Moving Averaging (MA) model, which were later combined by Box and Jenkins into the ARMA model (<xref ref-type="bibr" rid="B66">Janacek, 2010</xref>) for modeling stationary time series. The ARIMA model, an extension of ARMA, incorporates differencing to make the time series stationary before forecasting, represented by ARIMA (p,d,q), where p is the number of autoregressive terms, d is the degree of differencing, and q is the number of moving average terms. ARIMA models have been applied in real-world scenarios, such as predicting COVID-19 cases. <xref ref-type="bibr" rid="B32">Ding et al. (2020)</xref> used an ARIMA (1,1,2) model to forecast COVID-19 in Italy. In another study, (<xref ref-type="bibr" rid="B18">Bayyurt and Bayyurt, 2020</xref>), utilized ARIMA models for predictions in Italy, Turkey, and Spain, achieving a Mean Absolute Percentage Error (MAPE) value below 10%. Similarly, (<xref ref-type="bibr" rid="B131">Tandon et al., 2022</xref>), employed an ARIMA (2,2,2) model to forecast COVID-19 cases in India, reporting a MAPE of 5%, along with corresponding mean absolute deviation (MAD) and multiple seasonal decomposition (MSD) values.</p>
<p>When applying ARIMA models to biomedical data, we select the appropriate model using criteria like Akaike Information Criterion (AIC) or Bayesian information criterion (BIC), estimate parameters using tools like R or Python&#x27;s statsmodels, and validate the model through residual analysis. ARIMA models are effective for univariate time series with clear patterns, supported by extensive documentation and software, but they require stationarity and may be less effective for data with complex seasonality. Moreover, if a time series exhibits long-term memory, ARIMA models may produce unreliable forecasts (<xref ref-type="bibr" rid="B10">Al Zahrani et al., 2020</xref>), signifying that they are inadequate for capturing long-term dependencies. Additionally, ARIMA models necessitate a minimum of 50 data points in the time series to generate accurate forecasts (<xref ref-type="bibr" rid="B95">Montgomery et al., 2015</xref>). Therefore, ARIMA models should not be used for biomedical data that require the modeling of long-term relationships or have a small number of data points.</p>
<p>Several extensions such as Seasonal ARIMA (SARIMA) have been introduced for addressing seasonality. For instance, the research by <xref ref-type="bibr" rid="B84">Liu et al. (2023)</xref> examined 10 years of inpatient data on Acute Mountain Sickness (AMS), uncovering evident periodicity and seasonality, thereby establishing its suitability for SARIMA modeling. The SARIMA model exhibited high accuracy for short-term forecasts, assisting in comprehending AMS trends and optimizing the allocation of medical resources. An additional extension of ARIMA, proposed for long-term forecasts, is ARFIMA. In the study by <xref ref-type="bibr" rid="B109">Qi et al. (2020)</xref>, the Seasonal Autoregressive Fractionally Integrated Moving Average (SARFIMA) model was utilized to forecast the incidence of hemorrhagic fever with renal syndrome (HFRS). The SARFIMA model showed a better fit and forecasting accuracy compared to the SARIMA model, indicating its superior capability for early warning and control of infectious diseases by capturing long-range dependencies. Additionally, it is apparent that ARIMA models cannot incorporate exogenous variables. Therefore, a variation incorporating exogenous variables, known as the ARIMAX model, has been proposed. The study by <xref ref-type="bibr" rid="B90">Mahmudimanesh et al. (2022)</xref> applied the ARIMAX model to forecast cardiac and respiratory mortality in Tehran by analyzing the effects of air pollution and environmental factors. The key variables encompass air pollutants (CO, NO2, SO2, PM10) and environmental data (temperature, humidity). The ARIMAX model is selected for its capacity to include exogenous variables and manage non-static time series data.</p>
<p>For multi-step ahead forecasting in temporal prediction models, two methods exist. The first, known as the plug-in or iterated multi-step (IMS) prediction that involves successively using the single step predictor, treating each prediction as if it were an observed value to obtain the expected future value. The second approach is to create a direct multi-step (DMS) prediction as a function of the observations, and to select the coefficients in this predictor by minimizing the sum of squares of the multi-step forecast errors. <xref ref-type="bibr" rid="B56">Haywood and Wilson (2009)</xref> developed a test to decide which of two approaches is more dependable based on a given lead-time. In addition to this test, there are other ways to decide which technique is most suitable for forecasting multiple steps ahead. One of these methods can be used to decide the best choice for multi-step ahead prediction either for ARIMA or other types of models depending on the amount of historical data and the lead-time.</p>
</sec>
<sec id="s2-2">
<title>2.2 Exponential weighted moving average models</title>
<p>The EWMA method, based on <xref ref-type="bibr" rid="B115">Roberts (2000)</xref>, uses first-order exponential smoothing as a linear combination of the current observation and the previous smoothed observation. The smoothed observation <inline-formula id="inf25">
<mml:math id="m25">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> at time t is given by the equation <inline-formula id="inf26">
<mml:math id="m26">
<mml:mrow>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf27">
<mml:math id="m27">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the weight assigned to the latest observation. This recursive equation requires an initial value <inline-formula id="inf28">
<mml:math id="m28">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula>. Common choices for <inline-formula id="inf29">
<mml:math id="m29">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> include setting it equal to the first observation <inline-formula id="inf30">
<mml:math id="m30">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> or the average of available data, depending on the expected changes in the process. The smoothing parameter <inline-formula id="inf31">
<mml:math id="m31">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is typically chosen by minimizing metrics such as Mean Squared Error (MSE) or MAPE (<xref ref-type="bibr" rid="B95">Montgomery et al., 2015</xref>).</p>
<p>Several modifications of simple exponential smoothing exist to account for trends and seasonal variations, such as Holt&#x27;s method (<xref ref-type="bibr" rid="B61">Holt, 2004</xref>) and Holt-Winter&#x27;s method (<xref ref-type="bibr" rid="B145">Winters, 1960</xref>). These can be used in either additive or multiplicative forms. For modeling and forecasting biomedical temporal data, the choice of method depends on the data characteristics. Holt&#x27;s method is more appropriate for data with trends. On the other hand, EWMA is suitable for stationary or relatively stable data, making it effective in scenarios without a clear trend, such as certain biomedical measurements. For instance, <xref ref-type="bibr" rid="B112">Rachmat and Suhartono (2020)</xref> performed a comparative analysis of the simple exponential smoothing model and Holt&#x2019;s method for forecasting the number of goods required in a hospital&#x2019;s inpatient service, assessing performance using error percentage and MAD. Their findings indicated that the EWMA model outperformed Holt&#x2019;s method, as it produced lower forecast errors. This outcome is logical since the historical data of hospitalized patients lack any discernible trend.</p>
<p>EWMA models are also intended for univariate, regularly-spaced temporal data, as demonstrated in the example above (<xref ref-type="bibr" rid="B112">Rachmat and Suhartono, 2020</xref>), which uses a single variable (number of goods) over a period of time as input for model construction. This model is not suitable for biomedical data that involves multiple variables influencing the forecast unless its extention for multivariate data is employed. As highlighted by <xref ref-type="bibr" rid="B30">De Gooijer and Hyndman (2006)</xref>, there has been surprisingly little progress in developing multivariate versions of exponential smoothing methods for forecasting. <xref ref-type="bibr" rid="B108">Poloni and Sbrana (2015)</xref> attributes this to the challenges in parameter estimation for high-dimensional systems. Conventional multivariate maximum likelihood methods are prone to numerical convergence issues and high complexity, which escalate with model dimensionality. They propose a novel strategy that simplifies the high-dimensional maximum likelihood problem into several manageable univariate problems, rendering the algorithm largely unaffected by dimensionality.</p>
<p>EWMA models cannot directly handle data that is not evenly spaced, and thus cannot be used to directly model biomedical data with a large number of missing values without imputation. These models are capable of multi-step ahead prediction either through DMS or IMS approach. To emphasize long-range dependencies, the parameter <inline-formula id="inf32">
<mml:math id="m32">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> can be set to a low value, while a higher value will give more importance to recent past value (<xref ref-type="bibr" rid="B111">Rabyk and Schmid, 2016</xref>). The range of <inline-formula id="inf33">
<mml:math id="m33">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> values typically used for reasonable forecasting is 0.1&#x2013;0.4, depending on the amount of historical data available for modeling (<xref ref-type="bibr" rid="B95">Montgomery et al., 2015</xref>).</p>
</sec>
<sec id="s2-3">
<title>2.3 Regression models</title>
<p>Several regression models are available, and in this discussion, we focus on two specific types: multiple linear regression (MLR) (<xref ref-type="bibr" rid="B44">Galton, 1886</xref>; <xref ref-type="bibr" rid="B103">Pearson, 1922</xref>; <xref ref-type="bibr" rid="B104">Pearson, 2023</xref>) and multiple polynomial regression (MPR) (<xref ref-type="bibr" rid="B79">Legendre, 1806</xref>; <xref ref-type="bibr" rid="B46">Gauss, 1823</xref>). These models are particularly relevant for biomedical data analysis as they accommodate the use of two or more variables to forecast values. In MLR, there is one continuous dependent variable and two or more independent variables, which may be either continuous or categorical. This model operates under the assumption of a linear relationship between the variables. On the other hand, MPR shares the same structure as MLR but differs in that it assumes a polynomial or non-linear relationship between the independent and dependent variables. This review provides examination of these two regression models.</p>
<sec id="s2-3-1">
<title>2.3.1 Multiple linear regression models</title>
<p>The estimated value of output <inline-formula id="inf34">
<mml:math id="m34">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> at time <inline-formula id="inf35">
<mml:math id="m35">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, denoted as <inline-formula id="inf36">
<mml:math id="m36">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> with a MLR model for a certain set of predictors is given by the following <xref ref-type="disp-formula" rid="e1">Equation 1</xref>.<disp-formula id="e1">
<mml:math id="m37">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where, <inline-formula id="inf37">
<mml:math id="m38">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is a vector of <inline-formula id="inf38">
<mml:math id="m39">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> explanatory variables at time <inline-formula id="inf39">
<mml:math id="m40">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf40">
<mml:math id="m41">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> are regression coefficients, and <inline-formula id="inf41">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is a random error term at time <inline-formula id="inf42">
<mml:math id="m43">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf43">
<mml:math id="m44">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> (<xref ref-type="bibr" rid="B41">Fang and Lahdelma, 2016</xref>). It can be solved with least squares method (<xref ref-type="bibr" rid="B102">Pearson, 1901</xref>) to obtain the regression coefficients.</p>
<p>
<inline-formula id="inf44">
<mml:math id="m45">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> value can be calculated to check the accuracy of model fitting. The value of <inline-formula id="inf45">
<mml:math id="m46">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> that is closer to 1 indicates better model performance. Metrics such as Root Mean Squared Error (RMSE), Mean Absolute Percentage Error (MAPE), and Theil&#x2019;s inequality coefficient (TIC) are commonly utilized to assess the forecasting model&#x2019;s performance. While RMSE is scale-sensitive, MAPE and TIC are scale-insensitive. Lower values for these three metrics signify a well-fitting forecasting model.</p>
<p>
<xref ref-type="bibr" rid="B156">Zhang et al. (2021)</xref> developed an MLR model aimed at being computationally efficient and accurate for forecasting blood glucose levels in individuals with type 1 diabetes. These MLR models can predict specific future intervals (e.g., 30 or 60 min ahead). The dataset is divided into training, validation, and testing subsets; missing values are handled using interpolation and forward filling, and the data is normalized for uniformity. The MLR model showed strong performance, especially in 60-min forward predictions, and was noted for its computational efficiency in comparison to deep learning models. It excelled in short-term time series forecasts with significant data variability, making it optimal for real-time clinical applications.</p>
</sec>
<sec id="s2-3-2">
<title>2.3.2 Multiple polynomial regression models</title>
<p>The estimated value of <inline-formula id="inf46">
<mml:math id="m47">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> with say a second-order MPR model for a certain set of predictors is given by the following <xref ref-type="disp-formula" rid="e2">Equation 2</xref>.<disp-formula id="e2">
<mml:math id="m48">
<mml:mrow>
<mml:mtable class="aligned">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right">
<mml:mspace width="-3em"/>
<mml:mo>&#x2b;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>n</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>n</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3f5;</mml:mi>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where, <inline-formula id="inf47">
<mml:math id="m49">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf48">
<mml:math id="m50">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are regression coefficients, <inline-formula id="inf49">
<mml:math id="m51">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are predictor variables, and <inline-formula id="inf50">
<mml:math id="m52">
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is a random error. The ordinary least squares method (<xref ref-type="bibr" rid="B79">Legendre, 1806</xref>; <xref ref-type="bibr" rid="B46">Gauss, 1823</xref>) is applicable for solving this, similar to how it is used with MLR models. Furthermore, the evaluation metrics utilized for MLR are also suitable for MPR models.</p>
<p>
<xref ref-type="bibr" rid="B147">Wu et al. (2021)</xref> utilized US COVID-19 data from January 22 to July 20 (2020), categorizing it into nationwide and state-level data sets. Positive cases were identified as Temporal Features (TF), whereas negative cases, total tests, and daily positive case increases were identified as Characteristic Features (CF). Various other features were employed in different manners, such as the daily increment of hospitalized COVID-19 patients. An MPR model was created for forecasting single-day outcomes. The model consisted of pre-processing and forecasting phases. The pre-processing phase included quantifying temporal dependency through time-window lag adjustment, selecting CFs, and performing bias correction. The forecasting phase involved developing MPR models on pre-processed data sets, tuning parameters, and employing cross-validation techniques to forecast daily positive cases based on state classification.</p>
<p>The various applications of multiple regression models stated above, linear or polynomial, reveal their inability to directly capture temporal patterns. Although these models can accommodate multiple input variables, their design limits them to forecasting a singular outcome with one model. One of the extentions proposed to tackle this problem is multivariate MLR (MVMLR). <xref ref-type="bibr" rid="B129">Suganya et al. (2020)</xref> employs MVMLR to forecast four continuous COVID-19 target variables (confirmed cases and death counts after one and 2 weeks) using cumulative confirmed cases and death counts as independent variables. The methodology includes data preprocessing, feature selection, and model evaluation using metrics like Accuracy, <inline-formula id="inf51">
<mml:math id="m53">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> score, Mean Absolute Error (MAE), Mean Squared Error (MSE), and Root Mean Squared Error (RMSE).</p>
<p>It is clear from the design of the regression models that they are unable to process missing input data. Unless all the predictor variables are present or substituted, the value of the output variable cannot be determined. Therefore, it becomes essential to apply imputation techniques prior to employing the regression models for forecasting.</p>
<p>The regression models do not usually require a large amount of data; it has been demonstrated to be effective with as few as 15 data points per case (<xref ref-type="bibr" rid="B42">Filipow et al., 2023</xref>). Multi-step ahead prediction can be accomplished with either IMS or DMS approaches when dealing with temporal data like previous cases. Nonetheless, as mentioned previously, since these methods do not inherently capture temporal dependencies, forecasts can be generated as long as the temporal order is maintained while training, and testing the model.</p>
</sec>
</sec>
</sec>
<sec id="s3">
<title>3 Machine learning models</title>
<p>Many machine learning models are employed to construct forecasting models for temporal data sets. The most popular models for temporal data sets include Support vector regression (SVR), k-nearest neighbors regression (KNNR), Regression trees (Random forest regression [RFR]), Markov process (MP) models, Gaussian process (GP) models. We will examine these techniques in the following sections.</p>
<sec id="s3-1">
<title>3.1 Support vector regression</title>
<p>The origin of Support Vector Machines (SVMs) can be traced back to <xref ref-type="bibr" rid="B135">Vapnik (1999)</xref>. Initially, SVMs were designed to address the issue of classification, but they have since been extended to the realm of regression or forecasting problems (<xref ref-type="bibr" rid="B136">Vapnik et al., 1996</xref>). The SVR approach has the benefit of transforming a nonlinear problem into a linear one. This is done by mapping the data set <inline-formula id="inf52">
<mml:math id="m54">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> into a higher-dimensional, linear feature space. This allows linear regression to be performed on the new feature space. Various kernels are employed to convert non-linear data into linear data. The most commonly used are linear kernel, polynomial kernel, and radial basis or Gaussian kernel.</p>
<p>Upon transforming a nonlinear dataset <inline-formula id="inf53">
<mml:math id="m55">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> into a higher-dimensional, linear feature space, the prediction function <inline-formula id="inf54">
<mml:math id="m56">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is expressed by <xref ref-type="disp-formula" rid="e3">Equation 3</xref>.<disp-formula id="e3">
<mml:math id="m57">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mi>&#x3d5;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
<p>The SVR algorithm solves a nonlinear regression problem by transforming the training data <inline-formula id="inf55">
<mml:math id="m58">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> (where <inline-formula id="inf56">
<mml:math id="m59">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> ranges from one to <inline-formula id="inf57">
<mml:math id="m60">
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, with <inline-formula id="inf58">
<mml:math id="m61">
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> being the size of the training data set) into a new feature space, denoted by <inline-formula id="inf59">
<mml:math id="m62">
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. This transformation allows establishing a linear relationship between input and output, using the weight matrix <inline-formula id="inf60">
<mml:math id="m63">
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and bias matrix <inline-formula id="inf61">
<mml:math id="m64">
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to further refine the model.</p>
<p>In SVR, selecting optimal hyperparameters <inline-formula id="inf62">
<mml:math id="m65">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is crucial for accurate forecasting. The parameter <inline-formula id="inf63">
<mml:math id="m66">
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> controls the balance between minimizing training error and generalization. A higher <inline-formula id="inf64">
<mml:math id="m67">
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> reduces training errors but may overfit, while a lower <inline-formula id="inf65">
<mml:math id="m68">
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> results in a smoother decision function, possibly sacrificing training accuracy. The parameter <inline-formula id="inf66">
<mml:math id="m69">
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> sets a tolerance margin where errors are not penalized, forming an <inline-formula id="inf67">
<mml:math id="m70">
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-tube around predictions. A larger <inline-formula id="inf68">
<mml:math id="m71">
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> simplifies the model but may underfit, whereas a smaller <inline-formula id="inf69">
<mml:math id="m72">
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> provides more detail, potentially leading to overfitting. Optimal values for <inline-formula id="inf70">
<mml:math id="m73">
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf71">
<mml:math id="m74">
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> may require additional methods (<xref ref-type="bibr" rid="B88">Liu et al., 2021</xref>).</p>
<p>SVR is often combined with other algorithms for parameter optimization. Evolutionary algorithms frequently determine SVR parameters. For example, <xref ref-type="bibr" rid="B54">Hamdi et al. (2018)</xref> used a combination of SVR and differential evolution (DE) to predict blood glucose levels with continuous glucose monitoring (CGM) data. The DE algorithm was used to determine the optimal parameters of the SVR model, which was then built based on these parameters. The model was tested using real CGM data from 12 patients, and RMSE was used to evaluate its performance for different prediction horizons. The RMSE values obtained were 9.44, 10.78, 11.82, and 12.95 mg/dL for prediction horizons (PH) of 15, 30, 45, and 60 min, respectively. It should be noted that when these evolutionary algorithms are employed for determining parameters, SVR encounters notable disadvantages, including a propensity to get stuck in local minima (premature convergence).</p>
<p>Moreover, SVR can occasionally lack robustness, resulting in inconsistent outcomes. To mitigate these challenges, hybrid algorithms and innovative approaches are applied. For instance, Empirical Mode Decomposition (EMD) is employed to extract non-linear or non-stationary elements from the initial dataset. EMD facilitates the decomposition of data, thereby improving the effectiveness of the kernel function <xref ref-type="bibr" rid="B40">Fan et al. (2017)</xref>.</p>
<p>Essentially, SVR is an effective method for dealing with MTS data (<xref ref-type="bibr" rid="B154">Zhang et al., 2019</xref>). SVR, which operates on regression-based extrapolation, fits a curve to the training data and then uses this curve to predict future samples. It allows for continuous predictions rather than only at fixed intervals, making it applicable to irregularly spaced time series (<xref ref-type="bibr" rid="B50">Godfrey and Gashler, 2017</xref>). Nonetheless, due to its structure, SVR struggles to capture complex temporal dependencies (<xref ref-type="bibr" rid="B142">Weerakody et al., 2021</xref>).</p>
<p>It is suitable for smaller data sets as the computational complexity of the problem increases with the size of the sample <xref ref-type="bibr" rid="B88">Liu et al. (2021)</xref>. It excels at forecasting datasets with high dimensionality <xref ref-type="bibr" rid="B47">Gavrishchaka and Banerjee (2006)</xref> due to the advanced mapping capabilities of kernel functions <xref ref-type="bibr" rid="B40">Fan et al. (2017)</xref>. Additionally, multi-step ahead prediction in the context of SVR&#x2019;s application to temporal data can be achieved either with the DMS or IMS approach (<xref ref-type="bibr" rid="B11">Bao et al., 2014</xref>).</p>
</sec>
<sec id="s3-2">
<title>3.2 K-nearest neighbors regression</title>
<p>In 1951, Evelyn Fix and Joseph Hodges developed the KNN algorithm for discriminant examination analysis (<xref ref-type="bibr" rid="B43">Fix and Hodges, 1989</xref>). This algorithm was then extended to be used for regression or forecasting. The KNN method assumes that the current time series segment will evolve in the future in a similar way to a past time series segment (not necessarily a recent one) that has already been observed (<xref ref-type="bibr" rid="B73">Kantz and Schreiber, 2004</xref>). The task is thus to identify past segments of the time series that are similar to the present one according to a certain norm. Given a time series <inline-formula id="inf72">
<mml:math id="m75">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> with <inline-formula id="inf73">
<mml:math id="m76">
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> samples, the segment made of the last <inline-formula id="inf74">
<mml:math id="m77">
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> samples is denoted as <inline-formula id="inf75">
<mml:math id="m78">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, reflecting the current disturbance pattern. The KNN algorithm searches for <inline-formula id="inf76">
<mml:math id="m79">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> past time series intervals most comparable to <inline-formula id="inf77">
<mml:math id="m80">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> within the memory <inline-formula id="inf78">
<mml:math id="m81">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> using various distance metrics. For each nearest neighbor, a following time series of length <inline-formula id="inf79">
<mml:math id="m82">
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is generated, known as prediction contributions. Forecasting can then be done using unweighted or weighted approaches. In the unweighted approach, the prediction is the mean of the prediction contributions. In the weighted approach, the prediction is a weighted average based on the distance of each nearest neighbor from the current segment. Weights are assigned inversely proportional to the distances.</p>
<p>
<xref ref-type="bibr" rid="B52">Gopakumar et al. (2016)</xref> employed the KNN algorithm to forecast the total number of discharges from an open ward in an Australian hospital, which lacked real-time clinical data. To estimate the next-day discharge, they used the median of similar discharges from the past. The quality of the forecast was evaluated using the mean forecast error (MFE), MAE, symmetric MAE (SMAPE), and RMSE. The results of these metrics were reported to be 1.09, 2.88, 34.92%, and 3.84, respectively, with an MAE error improvement of 16.3% over the naive forecast.</p>
<p>KNN regression is viable for multivariate temporal datasets, as illustrated by <xref ref-type="bibr" rid="B8">Al-Qahtani and Crone (2013)</xref>. Nevertheless, its forecasting accuracy diminishes as the dimensionality of the data escalates. Consequently, it is critical to meticulously select pertinent features that impact the target variable to enhance model performance.</p>
<p>KNN proves effective for irregular temporal datasets (<xref ref-type="bibr" rid="B50">Godfrey and Gashler, 2017</xref>) due to its ability to identify previous matching patterns rather than solely depending on recent data. This distinctive characteristic renders KNN regression a favored choice for imputing missing data (<xref ref-type="bibr" rid="B6">Aljuaid and Sasi, 2016</xref>) prior to initiating any forecasting. Furthermore, it excels in capturing seasonal variations or local trends, such as aligning the administration of a medication that elevates blood pressure with a low blood pressure condition. Conversely, its efficacy in identifying global trends is limited, particularly in scenarios like septic shock, where multiple health parameters progressively deteriorate over time (<xref ref-type="bibr" rid="B142">Weerakody et al., 2021</xref>).</p>
<p>The KNN algorithm necessitates distance computations for k-nearest neighbors. Selecting an appropriate distance metric aligned with the dataset&#x27;s attributes is essential, with Euclidean distance being prevalent, though other metrics may be more suitable for specific datasets. <xref ref-type="bibr" rid="B36">Ehsani and Drabl&#xf8;s (2020)</xref> examines the impact of various distance measures on cancer data classification, using both common and novel measures, including Sobolev and Fisher distances. The findings reveal that novel measures, especially Sobolev, perform comparably to established measures.</p>
<p>As the size of the training dataset increases, the computational demands of the algorithm also rise. To mitigate this issue, approximate nearest neighbor search algorithms can be employed (<xref ref-type="bibr" rid="B68">Jones et al., 2011</xref>). Furthermore, the algorithm requires a large amount of data to accurately detect similar patterns. Several methods have been suggested to accelerate the process; for example, (<xref ref-type="bibr" rid="B45">Garcia et al., 2010</xref>), presented two GPU-based implementations of the brute-force kNN search algorithm using CUDA and CUBLAS, achieving speed-ups of up to 64X and 189X over the ANN C&#x2b;&#x2b; library on synthetic data.</p>
<p>Similarly to other forecasting models, KNN is applicable for multistep ahead predictions using strategies such as IMS or DMS (<xref ref-type="bibr" rid="B92">Mart&#xed;nez et al., 2019</xref>). It is imperative to thoroughly analyze the clinical application and characteristics of the clinical data prior to employing KNN regression for forecasting, given its unique attributes. Optimizing the number of neighbors <inline-formula id="inf80">
<mml:math id="m83">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and the segment length <inline-formula id="inf81">
<mml:math id="m84">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> through cross-validation is crucial. Employing appropriate evaluation metrics (e.g., MFE, MAE, SMAPE, RMSE) is necessary to assess the model&#x2019;s performance.</p>
</sec>
<sec id="s3-3">
<title>3.3 Random forest regression</title>
<p>Random Forests (RFs), introduced by <xref ref-type="bibr" rid="B21">Breiman (2001)</xref>, are a widely-used forecasting data mining technique. According to <xref ref-type="bibr" rid="B20">Bou-Hamad and Jamali (2020)</xref>, they are tree-based ensemble methods used for predicting either categorical (classification) or numerical (regression) responses. In the context of regression, known as Random Forest Regression (RFR), RF models strive to derive a prediction function <inline-formula id="inf82">
<mml:math id="m85">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> that reduces the expected value of a loss function <inline-formula id="inf83">
<mml:math id="m86">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>Y</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, with the output <inline-formula id="inf84">
<mml:math id="m87">
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> typically evaluated using the squared error loss. RFR builds on base learners, where each learner is a tree trained on bootstrap samples of the data. The final prediction is the average of all tree predictions as shown by <xref ref-type="disp-formula" rid="e4">Equation 4</xref>.<disp-formula id="e4">
<mml:math id="m88">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>where <inline-formula id="inf85">
<mml:math id="m89">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the number of trees, and <inline-formula id="inf86">
<mml:math id="m90">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the <inline-formula id="inf87">
<mml:math id="m91">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th tree. Trees are constructed using binary recursive partitioning based on criteria such as MSE.</p>
<p>
<xref ref-type="bibr" rid="B157">Zhao et al. (2019)</xref> developed a RFR model to forecast the future estimated glomerular filtration rate (eGFR) values of patients to predict the progression of Chronic Kidney Disease (CKD). The data set used was from a regional health system and included 120,495 patients from 2009 to 2017. The data was divided into three tables: eGFR, demographic, and disease information. The model was optimized through grid-search and showed good fit and accuracy in forecasting eGFR for 2015&#x2013;2017 using the historical data from the past years. The forecasting accuracy decreased over time, indicating the importance of previous eGFR records. The model was successful in predicting CKD stages, with an average <inline-formula id="inf88">
<mml:math id="m92">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> of 0.95, 88% Macro Recall, and 96% Macro Precision over 3 years.</p>
<p>The study presented in <xref ref-type="bibr" rid="B157">Zhao et al. (2019)</xref> indicates that RFR is effective for forecasting multivariate data. Another research by <xref ref-type="bibr" rid="B62">Hosseinzadeh et al. (2023)</xref> found that RFR performs better with multivariate data than with univariate data, especially when the features hold substantial information about the target. Research by <xref ref-type="bibr" rid="B133">Tyralis and Papacharalampous (2017)</xref> indicated that RF incorporating many predictor variables without selecting key features exhibited inferior performance relative to other methods. Conversely, optimized RF utilizing a more refined set of variables showed consistent reliability, highlighting the importance of thoughtful variable selection.</p>
<p>Similar to SVR, RFR is able to process non-linear information, although it does not have a specific design for capturing temporal patterns (<xref ref-type="bibr" rid="B57">Helmini et al., 2019</xref>). RFR is capable of handling irregular or missing data. <xref ref-type="bibr" rid="B37">El Mrabet et al. (2022)</xref> compared RFR for fault detection with Deep Neural Networks (DNNs), and found that RFR was more resilient to missing data than DNNs, showing its superior ability to manage missing values. To apply RFR to temporal data, it must be suitably modeled. As an example, <xref ref-type="bibr" rid="B62">Hosseinzadeh et al. (2023)</xref> has demonstrated one of the techniques, which involves forecasting stream flow by modeling the RFR as a supervised learning task with 24 months of input data and corresponding 24 months of output sequence. The construction of sequences involves going through the entire data set, shifting 1 month at a time. The study showed that extending the look-back window beyond a certain time frame decreases accuracy, indicating RFR&#x2019;s difficulty in capturing long-term dependencies when used in temporal modeling context. For a forecasting window of 24 months, the look-back window must be at least 24 months to avoid an increase in MAPE. This implies that although RFR can be used for temporal modeling, its effectiveness is more in capturing short-term dependencies rather than long-term ones. The experiments conducted by <xref ref-type="bibr" rid="B133">Tyralis and Papacharalampous (2017)</xref> also support this, showing that utilizing a small number of recent variables as predictors during the fitting process significantly improves the RFR&#x2019;s forecasting accuracy.</p>
<p>RFR can be used to forecast multiple steps ahead, similar to other regression models used for temporal forecasting (<xref ref-type="bibr" rid="B5">Alhnaity et al., 2021</xref>). Regarding data management, RFR necessitates a considerable volume of data to adjust its hyperparameters. It can swiftly handle such extensive datasets, leading to a more accurate model (<xref ref-type="bibr" rid="B96">Moon et al., 2018</xref>).</p>
</sec>
<sec id="s3-4">
<title>3.4 Markov process models</title>
<p>Two types of Markov Process (MP) models exist: Linear Dynamic System (LDS) and Hidden Markov Model (HMM). Both of these models are based on the same concept: a hidden state variable that changes according to Markovian dynamics can be measured. The learning and inference algorithms for both models are similar in structure. The only difference is that the HMM uses a discrete state variable with any type of dynamics and measurements, while the LDS uses a continuous state variable with linear-Gaussian dynamics and measurements. These models are discussed in more detail in the following sections.</p>
<sec id="s3-4-1">
<title>3.4.1 Linear dynamic system</title>
<p>LDS, introduced by <xref ref-type="bibr" rid="B71">Kalman (1963)</xref>, models the dynamics of sequences using hidden states and discrete time. It assumes evenly spaced time intervals within sequences, where the state transition and state-observation probabilities are given by <inline-formula id="inf89">
<mml:math id="m93">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf90">
<mml:math id="m94">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> respectively. These probabilities are determined by the <xref ref-type="disp-formula" rid="e5">Equations 5</xref>, <xref ref-type="disp-formula" rid="e6">6</xref>.<disp-formula id="e5">
<mml:math id="m95">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>A</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
<disp-formula id="e6">
<mml:math id="m96">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>B</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b6;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
</p>
<p>The terms <inline-formula id="inf91">
<mml:math id="m97">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf92">
<mml:math id="m98">
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represent the transition and emission matrices, respectively, whereas <inline-formula id="inf93">
<mml:math id="m99">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf94">
<mml:math id="m100">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b6;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denote Gaussian noise components. Specifically, the stochastic element <inline-formula id="inf95">
<mml:math id="m101">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> adheres to a zero-mean Gaussian distribution <inline-formula id="inf96">
<mml:math id="m102">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x223c;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, characterized by a zero-mean vector and covariance matrix <inline-formula id="inf97">
<mml:math id="m103">
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. On the other hand, the stochastic component <inline-formula id="inf98">
<mml:math id="m104">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b6;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> follows a zero-mean Gaussian distribution <inline-formula id="inf99">
<mml:math id="m105">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b6;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x223c;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, which is also characterized by a zero-mean vector and covariance matrix <inline-formula id="inf100">
<mml:math id="m106">
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. The initial state distribution <inline-formula id="inf101">
<mml:math id="m107">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is defined, with mean <inline-formula id="inf102">
<mml:math id="m108">
<mml:mrow>
<mml:mi>&#x3be;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and covariance matrix <inline-formula id="inf103">
<mml:math id="m109">
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, i.e., <inline-formula id="inf104">
<mml:math id="m110">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x223c;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3be;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. The set of LDS parameters is denoted as <inline-formula id="inf105">
<mml:math id="m111">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>B</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>P</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>R</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3be;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. In applied scenarios, these parameters necessitate estimation from empirical data. Two standard approaches for learning LDS are the Expectation-Maximization (EM) (<xref ref-type="bibr" rid="B49">Ghahramani and Hinton, 1996</xref>) and spectral learning algorithms (<xref ref-type="bibr" rid="B75">Katayama, 2005</xref>; <xref ref-type="bibr" rid="B100">Overchee and Moor, 1996</xref>; <xref ref-type="bibr" rid="B33">Doretto et al., 2003</xref>). EM iteratively maximizes the likelihood of observations by cycling between expectation (E-step) and maximization (M-step). It is precise but can be slow and prone to local optima, especially with limited training data. Spectral learning algorithms provide a non-iterative, closed-form solution using singular value decomposition (SVD) to estimate LDS parameters. They are faster but may be less precise than EM.</p>
<p>A new data-driven state-space dynamic model was developed by <xref ref-type="bibr" rid="B140">Wang et al. (2014)</xref> using an extended Kalman filter to estimate time-varying coefficients based on three variate time series data corresponding to glucose, insulin, and meal intake from type 1 diabetic subjects. This model was used to forecast blood glucose levels and was evaluated against a standard model (forgetting-factor-based recursive ARX). The results showed that the proposed model was superior in terms of fit, temporal gain, and J index, making it better for early detection of glucose trends. Furthermore, the model parameters could be estimated in real time, making it suitable for adaptive control. This model was tested for various prediction horizons, demonstrating the model&#x2019;s suitability for multi-step ahead prediction.</p>
<p>The LDS is apt for modelling multivariate temporal data, yet it is confined to data sampled at regular time intervals. As a result, its application to irregularly spaced data (<xref ref-type="bibr" rid="B122">Shamout et al., 2021</xref>) or time series with missing values may be problematic. In such instances, modifications and extension are needed. For example, <xref ref-type="bibr" rid="B87">Liu et al. (2013)</xref> presented a novel probabilistic method for modeling clinical time series data that accommodates irregularly sampled observations using LDS combined with GP models. They defined the model by a series of GPs, each confined to a finite window, with dependencies between consecutive GPs represented via an LDS. Their experiments on real-world clinical time series data demonstrate that their model excels in modeling clinical time series and either outperforms or matches alternative time series prediction models.</p>
<p>Typically, implementing the LDS model starts with thorough data preparation, requiring uniform sampling. In cases of irregular sampling or datasets with missing values, proper management through interpolation or imputation is essential for using the model without alterations, as mentioned above. The model architecture is constructed using hidden state variables <inline-formula id="inf106">
<mml:math id="m112">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> to encapsulate the latent processes, alongside measurable observation variables <inline-formula id="inf107">
<mml:math id="m113">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> representing directly observable quantities. Parameters such as the state transition matrix (A), the emission matrix (B), and the covariance matrices for process noise (P) and observation noise (R) should be initialized based on prior knowledge or through randomization techniques. Parameter learning is facilitated through the EM algorithm or spectral learning methods, with practical considerations dictating the choice: EM being preferred for its precision with limited datasets and spectral methods for their computational expediency.</p>
<p>The LDS or Kalman filter remains a cornerstone for tracking and estimation due to its attributes of simplicity, optimality, tractability, and robustness. However, nonlinear system applications present complex challenges, often mitigated by the Extended Kalman Filter (EKF) (<xref ref-type="bibr" rid="B80">Lewis, 1986</xref>) which linearizes nonlinear models to leverage the linear Kalman filter. Also, various advancements have been proposed for LDS, particularly when addressing nonlinear or non-Gaussian dynamics. For example, approximate filtering methodologies such as the unscented Kalman filter (<xref ref-type="bibr" rid="B69">Julier and Uhlmann, 1997</xref>), alongside Monte Carlo-based techniques including the particle filter (<xref ref-type="bibr" rid="B53">Gordon et al., 1993</xref>) and the ensemble Kalman filter (<xref ref-type="bibr" rid="B38">Evensen, 1994</xref>), are also utilized similar to EKF. Model evaluation is conducted through cross-validation employing metrics such as MSE or RMSE. For forecasting applications, the model can be employed for one-step ahead forecasts or extended to iterative multi-step predictions.</p>
</sec>
<sec id="s3-4-2">
<title>3.4.2 Hidden markov model</title>
<p>Hidden Markov Models (HMMs), introduced by Baum and colleagues in the late 1960s and early 1970s (<xref ref-type="bibr" rid="B15">Baum and Petrie, 1966</xref>; <xref ref-type="bibr" rid="B13">Baum and Eagon, 1967</xref>; <xref ref-type="bibr" rid="B17">Baum and Sell, 1968</xref>; <xref ref-type="bibr" rid="B16">Baum et al., 1970</xref>; <xref ref-type="bibr" rid="B14">Baum, 1972</xref>), are powerful tools for linking hidden states with observed events, assuming an underlying stochastic process. An HMM consists of a set of hidden states, a transition probability matrix, a sequence of observations, observation likelihoods, and an initial state distribution. A critical assumption in HMMs is output independence, where the probability of an observation depends solely on the state that produced it.</p>
<p>HMMs address three fundamental problems: (1) Likelihood estimation: Using the forward or backward algorithm to compute the probability of an observed sequence given the model parameters; (2) Decoding: Employing the Viterbi algorithm to determine the optimal sequence of hidden states corresponding to a sequence of observations; and (3) Learning: Applying the Baum-Welch algorithm, a special case of the EM algorithm, to estimate HMM parameters from observation sequences.</p>
<p>
<xref ref-type="bibr" rid="B127">Sotoodeh and Ho (2019)</xref> proposed a novel feature representation based on the HMM to predict the length of stay of patients admitted to the ICU. This representation was composed of a specified time resolution and a summary statistic calculated for a specific time window for each feature (e.g., average, most recent, maximum, etc.). An HMM was then trained on these features, and used to generate a series of states for each patient, with the first and last states being used as it was thought that these could better explain the variance in the length of stay. This feature matrix was then used as the input to a regression model to estimate the length of stay. Experiments were conducted to determine the optimal number of states, overlapping or non-overlapping time windows, aggregation of ICU types, summary measure for each time window, and selection of time window probabilities. The model was compared to other baseline models, and was found to have a lower RMSE than all of them.</p>
<p>It is evident from the application here that HMM is capable of dealing with multivariate data. Additionally, it is designed to process temporal data that is spaced at regular intervals of time (<xref ref-type="bibr" rid="B122">Shamout et al., 2021</xref>). Unfortunately, it is not able to process temporal data that is irregular or has missing values. <xref ref-type="bibr" rid="B24">Cao et al. (2015)</xref> employed both DMS and IMS strategies to forecast multiple future system states and anticipate the evolution of a fault in the Tennessee Eastman (TE) chemical process using HMM. They reported the accuracy of 1,2,3,.,20 step-ahead predictions, which were similar for both approaches, with the DMS approach being slightly more accurate than the IMS approach. This is understandable, as the IMS approach has to contend with additional complexities, such as cumulative errors, decreased precision, and increased uncertainty. This demonstrates the capability of HMM to make predictions for multiple steps in the future.</p>
<p>HMM can be constructed using either raw time series data or extracted features. <xref ref-type="bibr" rid="B121">Samaee and Kobravi (2020)</xref> introduced a forecasting model aimed at forecasting the timing of tremor bursts with a nonlinear hidden Markov model. This model was trained using the Baum-Welch algorithm, employing both raw Electromyogram (EMG) data and extracted features such as integrated EMG, mean frequency, and peak frequency. The study found that an HMM trained on raw EMG data performed better at forecasting tremor occurrences, suggesting that raw data more accurately captures tremor dynamics compared to extracted features. This is likely due to the short time window being insufficient for feature-based methods. Therefore, it is crucial to determine whether raw time series data or extracted features yield better performance in HMM construction.</p>
<p>In general, MP models are well-recognized for their efficacy in capturing short-term relationships (<xref ref-type="bibr" rid="B91">Manaris et al., 2011</xref>) between adjacent symbols or sequences with strong inter-symbol ties. However, they prove inadequate for representing long-distance dependencies between symbols that are spatially or temporally distant (<xref ref-type="bibr" rid="B151">Yoon and Vaidyanathan, 2006</xref>; <xref ref-type="bibr" rid="B91">Manaris et al., 2011</xref>). To enhance the representational scope of these models, certain methodologies must be employed. For instance, <xref ref-type="bibr" rid="B151">Yoon and Vaidyanathan (2006)</xref> proposed context-sensitive HMMs capable of capturing long-distance dependencies, thereby enabling robust pairwise correlations between distant symbols.</p>
<p>Additionally, a limitation of Markov models is that the intrinsic dimensionality of its hidden states is not known beforehand. If the dimensionality is too large, there is a risk of the model becoming overfitted. Therefore, it is often necessary to try out different training sizes and intrinsic dimensionality of the hidden states to create a model that fits (<xref ref-type="bibr" rid="B85">Liu, 2016</xref>).</p>
</sec>
</sec>
<sec id="s3-5">
<title>3.5 Gaussian process models</title>
<p>The Gaussian process (GP), introduced by Williams and Rasmussen (<xref ref-type="bibr" rid="B144">Williams and Rasmussen, 2006</xref>), is a non-parametric, non-linear Bayesian model in statistical machine learning. A GP is a collection of random variables, any finite number of which have a joint Gaussian distribution. This model extends the multivariate Gaussian to infinite-sized collections of real-valued variables, defining the distribution over random functions. A GP is represented by the mean function: <inline-formula id="inf108">
<mml:math id="m114">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="double-struck">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, and the covariance function: <inline-formula id="inf109">
<mml:math id="m115">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="double-struck">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where f(x) is a real-valued process and, <inline-formula id="inf110">
<mml:math id="m116">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf111">
<mml:math id="m117">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> are two input vectors.</p>
<p>In the context of biomedical temporal data, GP shows promises for modeling and forecasting due to their flexibility and ability to incorporate uncertainty. For example, GP can be used to model patient vital signs over time or predict disease progression (<xref ref-type="bibr" rid="B126">Siami-Namini et al., 2019</xref>). The key advantage of GP is their ability to provide uncertainty estimates along with predictions, which is crucial in biomedical applications where uncertainty quantification can inform clinical decisions. The GP can compute the distribution of function values for any set of inputs. This initial distribution, known as the prior, is a multivariate Gaussian represented by <xref ref-type="disp-formula" rid="e7">Equation 7</xref>.<disp-formula id="e7">
<mml:math id="m118">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x223c;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msup>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
</p>
<p>When given observed data, the GP updates this to the posterior distribution, which also follows a multivariate Gaussian. This updated distribution incorporates the observed data, providing more accurate predictions. The posterior distribution is influenced by the observed values and accounts for noise in the data.</p>
<p>GPs extend the multivariate Gaussian distribution into an infinite function space, making them suitable for time series modeling. They can handle observations taken at any time, whether regularly or irregularly spaced, and can make future predictions by calculating the posterior mean for any given time index. Additionally, GPs can act as non-linear transformation operators by replacing the linear transformations used in traditional temporal models with GP, offering a flexible approach to modeling complex data.</p>
<p>GP parameters consist of mean and covariance function parameters. The mean function, dependent on time, represents the expectation before observations. In cases of uncertain trend directions, constant-offset mean functions are common. If prior knowledge about the long-term trend exists, it can be incorporated into GP models, optimizing mean function parameters using gradient-based methods. In clinical scenarios with diverse patient ages and circumstances, aligning time origins is challenging. A practical approach is setting mean functions to a constant <inline-formula id="inf112">
<mml:math id="m119">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, making the GP time-invariant. The constant <inline-formula id="inf113">
<mml:math id="m120">
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is determined by averaging all patient observations. To optimize the covariance function parameters <inline-formula id="inf114">
<mml:math id="m121">
<mml:mrow>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, one can maximize the marginal likelihood <inline-formula id="inf115">
<mml:math id="m122">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>Y</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. The log marginal likelihood for GP is calculated where <inline-formula id="inf116">
<mml:math id="m123">
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> includes all training observations. The covariance matrix for noisy observations is represented by <inline-formula id="inf117">
<mml:math id="m124">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. It is calculated as <inline-formula id="inf118">
<mml:math id="m125">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf119">
<mml:math id="m126">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the covariance matrix for noise-free function values, and <inline-formula id="inf120">
<mml:math id="m127">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is a standard deviation of the noise, represented as, <inline-formula id="inf121">
<mml:math id="m128">
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
<mml:mo>&#x223c;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. The partial derivatives of the marginal likelihood with respect to each parameter in <inline-formula id="inf122">
<mml:math id="m129">
<mml:mrow>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> are then derived. These derivatives are used in gradient-based optimization methods to maximize <inline-formula id="inf123">
<mml:math id="m130">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>Y</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, thereby optimizing the covariance function parameters.</p>
<p>A prevalent limitation of GP models pertains to their high computational demands. Sparse GP methodologies have been devised to mitigate this challenge (<xref ref-type="bibr" rid="B144">Williams and Rasmussen, 2006</xref>; <xref ref-type="bibr" rid="B110">Quinonero-Candela and Rasmussen, 2005</xref>), primarily by identifying a subset of pseudo inputs to alleviate computational load. Further optimization of computational efficiency can be achieved through the application of the Kronecker product (<xref ref-type="bibr" rid="B128">Stegle et al., 2011</xref>), synchronization of training data across identical time intervals for each dimension (<xref ref-type="bibr" rid="B39">Evgeniou et al., 2005</xref>), or the implementation of recursive algorithms tailored for online settings (<xref ref-type="bibr" rid="B105">Pillonetto et al., 2008</xref>). Applications necessitating near real-time retraining are more apt to benefit from these approaches, whereas methods that extend over more prolonged temporal frameworks exhibit reduced sensitivity to such computational constraints. Another shortcoming of GP is that it models each time series separately, disregarding the interactions between multiple variables. To tackle this problem and capture the multivariate behavior of MTS, the multi-task Gaussian process (MTGP) was proposed (<xref ref-type="bibr" rid="B19">Bonilla et al., 2007</xref>).</p>
<sec id="s3-5-1">
<title>3.5.1 Multi-task Gaussian process</title>
<p>MTGP is an extension of GP that models multiple tasks (e.g., MTS) simultaneously by utilizing the learned covariance between related tasks. It uses <inline-formula id="inf124">
<mml:math id="m131">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> to model the similarities between tasks and <inline-formula id="inf125">
<mml:math id="m132">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> to capture the temporal dependence with respect to time stamps. The covariance function of MTGP is given by <xref ref-type="disp-formula" rid="e8">Equation 8</xref>.<disp-formula id="e8">
<mml:math id="m133">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2297;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>D</mml:mi>
<mml:mo>&#x2297;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>where <inline-formula id="inf126">
<mml:math id="m134">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is a positive semi-definite matrix and <inline-formula id="inf127">
<mml:math id="m135">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> measures the similarity between time series j and time series k. <inline-formula id="inf128">
<mml:math id="m136">
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is an <inline-formula id="inf129">
<mml:math id="m137">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> x <inline-formula id="inf130">
<mml:math id="m138">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> diagonal matrix in which <inline-formula id="inf131">
<mml:math id="m139">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the noise variance <inline-formula id="inf132">
<mml:math id="m140">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> for the <inline-formula id="inf133">
<mml:math id="m141">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> time series. <inline-formula id="inf134">
<mml:math id="m142">
<mml:mrow>
<mml:mo>&#x2297;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is the Kronecker product.</p>
<p>The parameters of GP-based models are composed of parameters that define the mean and covariance functions. Generally, the covariance function ensures that values of the function for two close times tend to have a high covariance, while values from inputs that are distant in time usually have a low covariance. These parameters can be acquired from data that includes one or multiple examples of time series. The predictions of values at future times are equivalent to the calculation of the posterior distribution for those times.</p>
<p>Proper data preprocessing is essential when building MTGP models for forecasting time series. This involves transformations such as detrending and applying logarithmic adjustments. Methods like spectral mixture kernels or Bayesian Nonparametric Spectral Estimation can be employed for initialization. Post-training, it is vital to visualize and interpret cross-channel correlations to better understand the inherent patterns, thereby supporting practical and accurate forecasting applications (<xref ref-type="bibr" rid="B31">de Wolff et al., 2021</xref>).</p>
<p>
<xref ref-type="bibr" rid="B125">Shukla (2017)</xref> proposed to use MTGP to forecast blood pressure from Photoplethysmogram (PPG) signals and compared its performance to Artificial Neural Networks (ANNs). Ten features were extracted from the PPG signal, and five of them were chosen as the tasks (or targets) to construct the MTGP model. These features were systolic blood pressure, diastolic blood pressure, systolic upstroke time, diastolic time and cardiac period. Four different ANN models were built based on one or more of the above tasks. The models were evaluated on clinical data from the MIMIC Database, with the absolute error <inline-formula id="inf135">
<mml:math id="m143">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> calculated for each heart beat as the performance measure. The results showed that the performance of MTGP was either comparable to or better than the ANNs and existing methods of computing BP from non-invasive data. MTGP is thus applicable for modeling multivariate temporal data with multiple prediction targets. In a study by <xref ref-type="bibr" rid="B35">D&#xfc;richen et al. (2014)</xref>, MTGP was employed on three diverse biomedical data sets. The experiments aimed to illustrate that forecasting all correlated variables simultaneously enhanced prediction performance, contrasting with individual variable predictions. MTGP has been demonstrated to be successful in multi-step ahead forecasting for a variety of biomedical domain applications mentioned here, as well as in other domains (<xref ref-type="bibr" rid="B23">Cai et al., 2020</xref>).</p>
<p>GP models, with an appropriate choice of covariance function, can capture rapid changes in a time series and can be applied to time series modeling problems by representing observations as a function of time. This means that there is no restriction on when the observations are made or if they are regularly or irregularly spaced in time. <xref ref-type="bibr" rid="B85">Liu (2016)</xref> and <xref ref-type="bibr" rid="B27">Cheng et al. (2020)</xref> demonstrated that, with the appropriate selection of a covariance function, it is possible to model both the short-term dependencies or long-term correlations of temporal data. GP models also work well with small amounts of data (<xref ref-type="bibr" rid="B85">Liu, 2016</xref>). It is possible to predict with a certain degree of certainty (confidence interval) using GP (<xref ref-type="bibr" rid="B116">Roberts et al., 2013</xref>), which is usually essential for temporal modeling of medical data that necessitates a certain degree of assurance to be employed by medical professionals to make their decisions. However, this approach has some limitations, the most serious being that the mean function of the GP is a function of time and must be set to a constant value in order to make the GP independent of the time origin. This significantly restricts its ability to represent changes or different modes in time series dynamics.</p>
</sec>
</sec>
</sec>
<sec id="s4">
<title>4 Deep learning models</title>
<p>The use of Deep Learning techniques for predicting time series data has gained significant attention. While there are various models available for handling time-series data, in this review, we will focus on some commonly used models for forecasting clinical data sets over time. Specifically, we will explore Recurrent Neural Networks (RNN), Long Short Term Memory Networks (LSTM), and Transformer models.</p>
<sec id="s4-1">
<title>4.1 Recurrent Neural Networks</title>
<p>The concept of RNN was introduced by Elman (1990) for identifying patterns in sequential data. RNNs accept sequential data as input and process it recursively. In an RNN, nodes are linked sequentially, where the input at time <inline-formula id="inf136">
<mml:math id="m144">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> depends on the output at time <inline-formula id="inf137">
<mml:math id="m145">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. The structure and functions of RNNs are depicted in <xref ref-type="fig" rid="F1">Figure 1</xref>.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>RNN structure (reproduced from <xref ref-type="bibr" rid="B88">Liu et al., 2021</xref>, licensed under CC BY 4.0).</p>
</caption>
<graphic xlink:href="fphys-15-1386760-g001.tif"/>
</fig>
<p>In this structure, the input layer <inline-formula id="inf138">
<mml:math id="m146">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is weighted by <inline-formula id="inf139">
<mml:math id="m147">
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, the hidden layer <inline-formula id="inf140">
<mml:math id="m148">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> by <inline-formula id="inf141">
<mml:math id="m149">
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and the output layer <inline-formula id="inf142">
<mml:math id="m150">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> by <inline-formula id="inf143">
<mml:math id="m151">
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. The equations employed for calculations are <xref ref-type="disp-formula" rid="e9">Equations 9</xref>, <xref ref-type="disp-formula" rid="e10">10</xref>.<disp-formula id="e9">
<mml:math id="m152">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>g</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>V</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>
<disp-formula id="e10">
<mml:math id="m153">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>U</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>W</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>The above formula is iterative in nature and can be expanded using the <xref ref-type="disp-formula" rid="e11">Equation 11</xref> as:<disp-formula id="e11">
<mml:math id="m154">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>V</mml:mi>
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>U</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>W</mml:mi>
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>U</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>W</mml:mi>
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>U</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mo>&#x22ef;</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>The equation above demonstrates that the RNN network&#x2019;s output <inline-formula id="inf144">
<mml:math id="m155">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is influenced by the current input <inline-formula id="inf145">
<mml:math id="m156">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, as well as the previous inputs <inline-formula id="inf146">
<mml:math id="m157">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. RNNs effectively handle sequential and correlated data by considering historical inputs. The work of <xref ref-type="bibr" rid="B25">Chandra et al. (2021)</xref> demonstrates its applicability in multi-step ahead prediction. Although their demonstration focuses on univariate cases, RNN has also been successfully applied to multivariate cases. In their study, <xref ref-type="bibr" rid="B159">Zhu et al. (2020)</xref> utilized four data fields for each instance: sampling time, CGM values, meal intake, and insulin dose. They employed a deep learning approach using an extention of RNN, dilated RNN (DRNN), to forecast glucose levels for the next 30 min. The DRNN model exhibits superior performance compared to current models like autoregressive (ARX), SVR, and neural networks for glucose prediction (NNPG), when evaluated on the OhioT1DM dataset. The RMSE values reported are ARX: 20.1 mg/dL, SVR: 21.7 mg/dL, NNPG: 22.9 mg/dL, and DRNN: 18.9 mg/dL. RNNs are frequently used to handle missing values or irregularities in multivariate temporal datasets. There are two main approaches to achieve this: imputation and data generation, or a forecasting approach. When using the first approach, RNNs leverage temporal correlations within each series and correlations among multiple features to fill in missing values or create a time series that captures the original characteristics. On the other hand, the latter approach involves the development of more advanced RNN-based solutions that provide a deeper understanding of the missing data, as well as the patterns and relationships within the data (<xref ref-type="bibr" rid="B142">Weerakody et al., 2021</xref>).</p>
<p>Implementing RNNs for modeling and forecasting biomedical temporal data necessitates meticulous attention to data preprocessing, model structure, tuning of hyperparameters, and evaluation techniques. The recommendations for each aspect are outlined as discussed in <xref ref-type="bibr" rid="B59">Hewamalage et al. (2021)</xref>. Deseasonalization is advised for datasets exhibiting seasonal trends unless consistent seasonal patterns exist, which RNNs can inherently manage. Data normalization enhances training convergence, while the sliding window approach divides the time series into overlapping sequences for model input. Hyperparameter tuning is crucial for achieving optimal RNN performance. Principal hyperparameters include the learning rate, batch size, and the number of layers. The learning rate must be selected judiciously; for ideal convergence, the Adagrad optimizer typically needs a higher learning rate ranging between 0.01 and 0.9, whereas the Adam optimizer performs effectively within a narrower range of 0.001&#x2013;0.1. The batch size should be commensurate with the dataset size, and usually, one or two layers are sufficient, as additional layers may result in overfitting. Setting high values for the standard deviation of regularization parameters for Gaussian noise and L2 weight regularization can cause significant underfitting, reducing the neural network&#x27;s efficacy in generating forecasts. One category of RNN models, stacked RNNs, which involve multiple RNN layers, are employed for forecasting and often utilize skip connections to alleviate vanishing gradient issues. Another category of RNN models, known as sequence-to-sequence (S2S) models, is typically applied in sequential data transformations and is useful for tasks like multi-step forecasting. Assessing RNN performance against traditional methods like ARIMA using standard metrics and cross-validation confirms their competitiveness. Enhancements to RNN methods, such as attention mechanisms and ensemble methods, further boost their performance. Attention mechanisms enable the model to concentrate on relevant parts of the input sequence, while ensemble methods combine several RNN models to produce robust forecasts, reducing biases and variances.</p>
<p>RNNs excel at capturing short-term dependencies (<xref ref-type="bibr" rid="B57">Helmini et al., 2019</xref>). They are more sensitive to time series data than traditional convolutional neural networks (CNNs) and can retain memory during data transmission. However, as previously mentioned, when the input sequence lengthens, the network demands more temporal references, leading to a deeper network. In longer sequences, it becomes challenging for the gradient to propagate back from later sequences to earlier ones, resulting in the vanishing gradient problem. Consequently, RNNs struggle with long-term dependencies. To mitigate this vanishing (or exploding) gradient issue, a modification of the RNN known as the long sshort-term memory (LSTM) model was introduced by <xref ref-type="bibr" rid="B60">Hochreiter and Schmidhuber (1997)</xref>.</p>
<sec id="s4-1-1">
<title>4.1.1 Long Short Term Memory Networks</title>
<p>To overcome the challenges of vanishing and exploding gradients in RNNs, the LSTM model was introduced. This architecture employs a cell state to maintain long-term dependencies, as discussed by <xref ref-type="bibr" rid="B57">Helmini et al. (2019)</xref>. The model effectively manages gradient dispersion by establishing a retention mechanism between input and feedback. <xref ref-type="fig" rid="F2">Figure 2</xref> illustrates the LSTM structure (<xref ref-type="bibr" rid="B142">Weerakody et al., 2021</xref>). Additionally, LSTM models are proficient in capturing short-term dependencies, primarily through the use of a hidden state. LSTM units are controlled by three gates: the input gate, the output gate, and the forget gate. These gates regulate the flow of information and maintain the cell state, enabling LSTMs to retain important information over long periods. The key equations (<xref ref-type="disp-formula" rid="e12">Equations 12</xref>&#x2013;<xref ref-type="disp-formula" rid="e17">17</xref>) governing LSTM operations are mentioned as follows:<disp-formula id="e12">
<mml:math id="m158">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(12)</label>
</disp-formula>
<disp-formula id="e13">
<mml:math id="m159">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(13)</label>
</disp-formula>
<disp-formula id="e14">
<mml:math id="m160">
<mml:mrow>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>tanh</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(14)</label>
</disp-formula>
<disp-formula id="e15">
<mml:math id="m161">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mspace width="0.17em"/>
</mml:mrow>
</mml:msub>
<mml:mo>&#x25e6;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.17em"/>
<mml:mo>&#x25e6;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(15)</label>
</disp-formula>
<disp-formula id="e16">
<mml:math id="m162">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>Y</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>Y</mml:mi>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(16)</label>
</disp-formula>
<disp-formula id="e17">
<mml:math id="m163">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.17em"/>
<mml:mo>&#x25e6;</mml:mo>
<mml:mi>tanh</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(17)</label>
</disp-formula>
</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>LSTM structure (reproduced from <xref ref-type="bibr" rid="B88">Liu et al., 2021</xref>, licensed under CC BY 4.0).</p>
</caption>
<graphic xlink:href="fphys-15-1386760-g002.tif"/>
</fig>
<p>In these equations, <inline-formula id="inf147">
<mml:math id="m164">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the sigmoid function, and <inline-formula id="inf148">
<mml:math id="m165">
<mml:mrow>
<mml:mo>&#x25e6;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> denotes element-wise multiplication. The forget gate <inline-formula id="inf149">
<mml:math id="m166">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> controls the retention of the previous cell state <inline-formula id="inf150">
<mml:math id="m167">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, the input gate <inline-formula id="inf151">
<mml:math id="m168">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> manages the incorporation of new information, and the output gate <inline-formula id="inf152">
<mml:math id="m169">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> determines the output based on the cell state <inline-formula id="inf153">
<mml:math id="m170">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf154">
<mml:math id="m171">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf155">
<mml:math id="m172">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf156">
<mml:math id="m173">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf157">
<mml:math id="m174">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf158">
<mml:math id="m175">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are different weights associated with the forget gate, input gate, and the current input unit state.</p>
<p>A deep learning neural network (NN) model based on LSTM with the addition of two fully connected layers was proposed by <xref ref-type="bibr" rid="B65">Idriss et al. (2019)</xref>, for forecasting blood glucose levels. To determine the optimal parameters for the model, several experiments were conducted using data from 10 diabetic patients. The performance of the proposed LSTM NN, as measured by RMSE, was compared to that of a simple LSTM model and an autoregressive (AR) model. The results indicated that the LSTM NN achieved higher accuracy (mean RMSE &#x3d; 12.38 mg/dL) compared to both the existing LSTM model (mean RMSE &#x3d; 28.84 mg/dL) for all patients and the AR model (mean RMSE &#x3d; 50.69 mg/dL) for 9 out of 10 patients. LSTM is therefore valuable in the representation of time-based information.</p>
<p>One popular extention of the LSTM network is a Bidirectional LSTM (BiLSTM) model which is obtained by modifying the architecture of the LSTM network to include two LSTM layers: one processing the input sequence from left to right (forward direction) and the other from right to left (backward direction). This bidirectional traversal allows the model to have information from both past and future contexts, enhancing its ability to capture complex patterns and dependencies. The outputs from both layers are concatenated at each time step, providing a richer representation of the input sequence. This approach results in improved performance for tasks like time series forecasting, as BiLSTM models can leverage additional training from both directions to better understand sequential data (<xref ref-type="bibr" rid="B1">Abbasimehr and Paki, 2022</xref>). For instance, in a study by <xref ref-type="bibr" rid="B119">Said et al. (2021)</xref>, a bidirectional LSTM (Bi-LSTM) was employed to analyze multivariate data from countries grouped based on demographic, socioeconomic, and health sector indicators alongwith the information on lockdown measures, to predict the cumulative number of COVID-19 cases in Qatar from December 1st to 31 December 2020.</p>
<p>LSTM is also combined with multi-head attention mechanisms. This approach aims to address the non-linear patterns and complexities often found in real-world time series data, which traditional forecasting techniques struggle to predict accurately (<xref ref-type="bibr" rid="B126">Siami-Namini et al., 2019</xref>). When dealing with irregular temporal data that contain missing values, traditional LSTM models face challenges and may produce suboptimal analyses and predictions. This is because applying the LSTM model to irregular temporal data, either by filling in missing values or using temporal smoothing, does not enable the model to differentiate between actual observations and imputed values. Therefore, caution is advised when using an LSTM model on a dataset where multiple missing values have been imputed.</p>
</sec>
</sec>
<sec id="s4-2">
<title>4.2 Transformer models</title>
<p>The Transformer model for natural language processing (NLP) was introduced by <xref ref-type="bibr" rid="B137">Vaswani et al. (2017)</xref>. This model is composed of an encoder-decoder network, which differs from the traditional sequential structure of RNN. Transformer model utilizes the Self-Attention mechanism to enable parallel training and capture global information. The encoder takes historical time series data as input, while the decoder predicts future values using an auto-regressive approach. This means that the decoder&#x2019;s generated output at each step is based on previously generated outputs. To establish a connection between the encoder and decoder, an attention mechanism is employed. This allows the decoder to learn how to effectively focus (&#x201c;pay attention&#x201d;) on relevant parts of the historical time series before making predictions. The decoder utilizes masked self-attention to prevent the network from accessing future values during training, thereby avoiding information leakage. The typical architecture of the Transformer model is depicted in the <xref ref-type="fig" rid="F3">Figure 3</xref>.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Transformer architecture (reproduced from <xref ref-type="bibr" rid="B88">Liu et al., 2021</xref>, licensed under CC BY 4.0).</p>
</caption>
<graphic xlink:href="fphys-15-1386760-g003.tif"/>
</fig>
<p>Originally designed for NLP tasks, the Transformer architecture has found application in temporal forecasting as well. To model irregular temporal data, various methods have been proposed. For instance, <xref ref-type="bibr" rid="B132">Tipirneni and Reddy (2022)</xref> introduced the Self-supervised Transformer for Time-Series (STraTS) model, which treats each time-series as observation triplets (time, variable, value) instead of matrices as done by conventional methods. This approach eliminates the need for aggregation or imputation. STraTS utilizes a Continuous Value Embedding (CVE) scheme to retain detailed time information without discretization.</p>
<p>The study by <xref ref-type="bibr" rid="B55">Harerimana et al. (2022)</xref> utilized a Multi-Headed Transformer (MHT) model to forecast clinical time-series variables from charted vital signs, leveraging the transformer architecture&#x2019;s attention mechanism to capture complex temporal dependencies. The dataset is split into training and testing sets per patient, using past 24-h data for recursive future predictions. Training involves a fixed dimension of 512 for all layers, and the model is evaluated using metrics like Area under the Receiver Operating Characteristic Curve (AUC-ROC), MSE, and MAPE. The MHT model outperforms traditional models (LSTM, Temporal Convolutional Network, TimeNet) in forecasting vital signs, length of stay, and in-hospital mortality, demonstrating superior accuracy and robustness by focusing on influential past time steps, validating its efficacy in handling clinical time-series data.</p>
<p>The Transformer architecture is a relatively new concept, and ongoing research is being conducted to explore its capabilities. For instance, <xref ref-type="bibr" rid="B83">Li et al. (2019)</xref> suggest that unlike RNN-based methods, the Transformer enables the model to access any part of the time series history, disregarding the distance. This characteristic potentially makes it more adept at capturing recurring patterns with long-term dependencies. However, <xref ref-type="bibr" rid="B153">Zeng et al. (2023)</xref> presented an opposing viewpoint, questioning the effectiveness of Transformer-based solutions in long-term time series forecasting (LTSF). They argue that while Transformers are adept at capturing semantic correlations in sequences, their self-attention mechanism, which is invariant to permutations, may result in the loss of crucial temporal information necessary for accurate time series modeling. In support of their claim, the researchers introduced LTSF-Linear, a simple one-layer linear model, and discovered that it outperformed more complex Transformer-based LTSF models on nine real-life data sets. In addition, a temporal fusion transformer (TFT) was suggested by <xref ref-type="bibr" rid="B155">Zhang et al. (2022)</xref> as a method that effectively captures both short-term and long-term dependencies. Hence, when employing Transformer-based approaches for temporal forecasting, it is crucial to take into account these distinct viewpoints and conduct experiments to determine the most effective modeling technique for the specific forecasting task, considering the presence of short-term and long-term dependencies.</p>
<p>While DL models are capable of generating precise predictions, they are frequently perceived as black-box models that lack interpretability and transparency in their internal processes (<xref ref-type="bibr" rid="B138">Vellido, 2019</xref>). This presents a significant issue as medical professionals are often hesitant to trust machine recommendations without a clear understanding of the underlying rationale. In addition, significant quantities of clinical data are utilized to generate standardized inputs for training DL models. The challenge of acquiring extensive clinical data sets poses a challenge in the integration of DL clinical models into real-world clinical systems (<xref ref-type="bibr" rid="B148">Xiao et al., 2018</xref>).</p>
</sec>
</sec>
<sec sec-type="discussion" id="s5">
<title>5 Discussion</title>
<p>This section is comprised of two subsections. The first subsection summarizes the overview of the models and their capacities in addressing the difficulties encountered in forecasting of clinical datasets. The second subsection explores the future prospects concerning the practical obstacles in implementing AI models for biomedical data modeling.</p>
<sec id="s5-1">
<title>5.1 Summary of models for biomedical temporal data forecasting</title>
<sec id="s5-1-1">
<title>5.1.1 Summary of statistical, ML, and DL models</title>
<p>This review focuses on predictive models for biomedical temporal data, which face several challenges such as missing values due to irregular data collection or errors. Traditional methods use imputation or deletion, but models that handle missing values without these steps are preferable, as patterns of missing data might hold valuable information termed as &#x201c;informative missingness&#x201d;. EHRs often feature MTS data, so models must capture these correlations. Temporal data complexity requires models to consider short-term and long-term patterns. Short-term patterns might involve events like norepinephrine administration linked to recent hypotension, while long-term patterns could involve past acute kidney injury necessitating dialysis. Models should account for these dependencies and support multi-step ahead forecasting for early disease detection. Data availability varies with clinical events, thus impacting model selection. These challenges are crucial for accurate, effective predictions in clinical settings. <xref ref-type="table" rid="T3">Table 3</xref> summarizes the advantages and disadvantages of the discussed models, supplemented by literature insights.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Advantages and disadvantages of models for handling biomedical temporal data.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Model type</th>
<th align="center">Advantages</th>
<th align="center">Disadvantages</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">ARIMA</td>
<td align="left">- Captures linear dependencies and trends<break/>- Interpretable parameters<break/>- Works well with stationary data</td>
<td align="left">- Needs data to be stationarized<break/>- Lacks ability to handle missing values<break/>- Inability to manage multivariate data<break/>- Can not capture long-range dependencies</td>
</tr>
<tr>
<td align="left">EWMA</td>
<td align="left">- Proficient in temporal modeling<break/>- Simple and computationally efficient<break/>- Adapts quickly to recent changes in data<break/>- Useful for smoothing noisy data<break/>- Effective in short-range modeling</td>
<td align="left">- Not suitable for complex patterns<break/>- Unsuitable for handling multivariate data<break/>- Critical initialization and parameter selection<break/>- Capable of long-range modeling with parameter adjustment</td>
</tr>
<tr>
<td align="left">MLR</td>
<td align="left">- Interpretable coefficients<break/>- Insights into variables&#x2019; relationships<break/>- Performs well with small-mid datasets<break/>- Efficiently manages multivariate data<break/>- Can be adapted for temporal modeling</td>
<td align="left">- Assumes linear relationships<break/>- Sensitive to multicollinearity<break/>- Requires features to be linearly related to the target.</td>
</tr>
<tr>
<td align="left">MPR</td>
<td align="left">- Can capture higher-order relationships<break/>- More flexible than MLR.<break/>- Suitable for polynomial relationships<break/>- Manages multivariate data efficiently</td>
<td align="left">- Prone to overfitting with high polynomial-degrees<break/>- Interpretation of coefficients can be complex<break/>- Unable to handle missing values</td>
</tr>
<tr>
<td align="left">SVR</td>
<td align="left">- Effective in high-dimensional spaces<break/>- Can capture nonlinear relationships<break/>- Robust to overfitting with regularization<break/>- Manages multivariate data efficiently<break/>- Although not designed for temporal modeling, but can be adapted to capture them</td>
<td align="left">- Computationally complex<break/>- Needs support from other algorithms for hyperparameter tuning<break/>- Lacks robustness resulting in inconsistent outcomes<break/>- Struggles to capture complex temporal dependencies<break/>- Memory intensive for large datasets</td>
</tr>
<tr>
<td align="left">KNNR</td>
<td align="left">- Non-parametric and flexible<break/>- Can be adapted for temporal modeling<break/>- Proficient in handling missing values<break/>- Efficiently manages multivariate data<break/>- Effective in short-range modeling due to its unique structure</td>
<td align="left">- Expensive for large datasets<break/>- Memory intensive<break/>- Falls short in capturing global dependencies</td>
</tr>
<tr>
<td align="left">RFR</td>
<td align="left">- Handles nonlinear relationships<break/>- Robust to overfitting<break/>- Can handle high-dimensional data<break/>- Manages multivariate data efficiently<break/>- Capable of handling irregular or missing data</td>
<td align="left">- Time consuming for large datasets<break/>- Requires careful tuning of hyperparameters<break/>- Difficulty in handling long-range dependencies</td>
</tr>
<tr>
<td align="left">LDS</td>
<td align="left">- Captures temporal dependencies<break/>- Efficiently handles multivariate data<break/>- Captures short-term relationships</td>
<td align="left">- Complex parameter tuning<break/>- Cannot deal with irregular data<break/>- Difficulty with nonlinear relationships</td>
</tr>
<tr>
<td align="left">HMM</td>
<td align="left">- Captures hidden influencing states<break/>- Useful for sequential data modeling<break/>- Efficiently handles multivariate data<break/>- Can capture short-term dependencies efficiently</td>
<td align="left">- Training complexity<break/>- Lacks interpretability of hidden states<break/>- Prone to overfitting when intrinsic dimensionality exceeds data<break/>- Struggles with capturing long-term dependencies</td>
</tr>
<tr>
<td align="left">MTGP</td>
<td align="left">- Models multiple tasks simultaneously<break/>- Captures correlations between tasks<break/>- Provides uncertainty estimates<break/>- Can forecast efficiently with irregular data<break/>- Flexible covariance function that can capture both short-range and long-range dependencies</td>
<td align="left">- Complex to implement and tune<break/>- If the GP is made time independent, it restricts the representation of changes in time series dynamics<break/>- Computationally intensive on large-scale</td>
</tr>
<tr>
<td align="left">RNN</td>
<td align="left">- Proficient in handling missing values<break/>- Can handle variable-length sequences<break/>- Effective for multivariate sequential data modeling</td>
<td align="left">- Vanishing/exploding gradient problem<break/>- Training can be slow<break/>- Difficulty with very long-term dependencies</td>
</tr>
<tr>
<td align="left">LSTM</td>
<td align="left">- Handles vanishing gradient problem<break/>- Captures long-term dependencies effectively<break/>- Robust to sequence length variations</td>
<td align="left">- Training complexity<break/>- Lacks interpretability<break/>- Requires careful hyperparameters tuning<break/>- May produce suboptimal analyses and predictions when modeling imputed data</td>
</tr>
<tr>
<td align="left">Transformer</td>
<td align="left">- Highly suitable for multivariate temporal modeling<break/>- Parallel processing of sequences<break/>- Scalable to large datasets<break/>- Effective in short-range modeling</td>
<td align="left">- Computationally intensive<break/>- Requires large amounts of data<break/>- Lacks interpretability<break/>- Fine-tuning can be complex<break/>- Uncertain effectiveness in managing long-term dependencies</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Forecasting is categorized into statistical, ML, and DL methods. We focused on models frequently used in biomedical temporal modeling, evaluating their effectiveness. For statistical methods, we analyzed ARIMA, EWMA, and regression models. In ML, we assessed SVR, RFR, KNNR, MP, and GP models. For DL methods, we evaluated RNN, LSTM, and Transformer models. Our analysis found the MTGP model effective for irregularly spaced data, capturing both short-term and long-term dependencies with an appropriate covariance function. It predicts multiple steps ahead and accounts for autocorrelation within and correlation between time series, making it suitable for multivariate temporal analysis with small to moderate data. However, MTGP&#x2019;s computational cost can be high with large data, and a constant mean function may limit its ability to represent time series dynamics. While MTGP is suitable for biomedical temporal modeling, alternative approaches include improving current models, adopting ensemble methods, or using hierarchical approaches discussed later in this paper.</p>
<p>Improving existing models by incorporating new techniques can address limitations in temporal analysis of biomedical data. For instance, while RNNs struggle with long-range dependencies, they handle other temporal challenges well. To overcome this, <xref ref-type="bibr" rid="B159">Zhu et al. (2020)</xref> introduced a dilated RNN, enhancing neuron receptive fields to capture long-term dependencies, enabling 30-min glucose level forecasts. Similarly, HMMs lack long-range correlation modeling. <xref ref-type="bibr" rid="B151">Yoon and Vaidyanathan (2006)</xref> introduced context-sensitive HMM (csHMM), capturing long-range correlations by adding context-sensitivity to model states. Additionally, the interpretability in DL models is essential. <xref ref-type="bibr" rid="B132">Tipirneni and Reddy (2022)</xref> proposed an interpretable model with outputs as linear combinations of individual feature components. Slight modifications to the original models can address specific limitations.</p>
<p>Even though various modifications have been suggested to address the shortcomings of individual models, certain limitations remain insurmountable. A recently emerging solution involves combining multiple models to create a fusion model, which allows for the integration of their strengths and mitigation of their weaknesses. These fusion models, also known as combination or ensemble forecasting models, is examined in the next subsection.</p>
</sec>
<sec id="s5-1-2">
<title>5.1.2 Fusion models</title>
<p>A different approach to enhance forecasting precision involves merging multiple models, also known as combination or ensemble forecasting models. The paper by Wang et al. (<xref ref-type="bibr" rid="B141">Wang et al., 2023</xref>) provides a comprehensive overview of the evolution and effectiveness of combining multiple forecasts to enhance prediction accuracy. Combining forecasts, known as &#x201c;ensemble forecasts,&#x201d; integrates information from various sources, avoiding the need to identify a single &#x201c;best&#x201d; forecast amidst model uncertainty and complex data patterns. The review covers simple combination methods, such as equally weighted averages, which surprisingly often outperform more sophisticated techniques due to their robustness and lower risk of overfitting. Linear combinations, which determine optimal weights based on historical performance, and nonlinear combinations, which account for nonlinear relationships using methods like neural networks, are also discussed. <xref ref-type="bibr" rid="B141">Wang et al. (2023)</xref> emphasize the potential of learning-based combination methods, such as stacking and cross-learning, which improve accuracy by training meta-models on multiple time series. In stacking, several forecasting models are trained on the original dataset, and their predictions are combined by a meta-model to provide an optimal forecast. Cross-learning builds on this by utilizing data from various time series to train the meta-model. The review also highlights the crucial role of diversity and precision in forecast combinations, pointing out that successful combinations are enhanced by diverse individual forecasts.</p>
<p>These techniques have been successfully applied to biomedical data forecasting. For example, <xref ref-type="bibr" rid="B98">Naemi et al. (2020)</xref> introduced a customizable real-time hybrid model, leveraging the Nonlinear Autoregressive Exogenous (NARX) model along with Ensemble Learning (EL) (RFR and AdaBoost), to forecast patient severity during their stay at Emergency Departments (ED). This model makes use of patient vital signs such as Pulse Rate (PR), Respiratory Rate (RR), Arterial Blood Oxygen Saturation (SpO2), and Systolic Blood Pressure (SBP), which are recorded during treatment. The model forecasts the severity of illness in hospitalized patients at ED for the upcoming hour based on their vital signs from the previous 2 hours. The effectiveness of the NARX-EL models is evaluated against other baseline models including ARIMA, a fusion of NARX and LR, SVR, and KNNR. The findings revealed that the proposed hybrid models could predict patient severity with significantly higher accuracy. Furthermore, it was noted that the NARX-RF model excels at predicting abrupt changes and unexpected adverse events in patients&#x2019; vital signs, exhibiting an <inline-formula id="inf159">
<mml:math id="m176">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> score of 0.978 and NRMSE of 6.16%. <xref ref-type="bibr" rid="B72">Kandula et al. (2018)</xref> used a super-ensemble technique to combine information from different forecasting methods robustly. This method yielded a more accurate comprehensive forecast on average than a single model. They compared three forecasting approaches for predicting seven characteristics of seasonal influenza during the 2016&#x2013;2017 USA season: a mechanistic method, a weighted average of two statistical methods, and a super-ensemble of eight statistical and mechanistic models. The study found the meta-ensemble approach to be the most accurate overall. <xref ref-type="bibr" rid="B74">Katari et al. (2023)</xref> employed a combination of Decision Tree (DT) and Ada Boosting algorithms for heart disease prediction. The study highlights the importance of early diagnosis due to high mortality rates. The hybrid model outperformed traditional methods in accuracy, true positive rate (TPR), and precision. Results indicate this combination approach enhances heart disease prediction and aids clinical decision-making.</p>
<p>It is evident that combining forecasts is a crucial component in contemporary forecasting methods for temporal biomedical datasets, providing notable benefits over using single models. Nevertheless, it is crucial to thoroughly understand the data and the aim of forecasting to create an effective ensemble model. Furthermore, it is essential to employ appropriate evaluation metrics for assessing biomedical temporal forecasts. Advancements in research on efficient combination techniques may arise from the capability to manage large and varied datasets, alongside the development of automatic selection methods that balance expertise and diversity when selecting and combining models for forecasting (<xref ref-type="bibr" rid="B141">Wang et al., 2023</xref>).</p>
</sec>
<sec id="s5-1-3">
<title>5.1.3 Coherent forecasting</title>
<p>This type of forecasting a.k.a. hierarchical time series (HTS) represents a set of data sequences organized by aggregation constraints, reflecting many real-world applications in research and industry. Forecasting in such hierarchical structures is challenging and time-consuming due to the need to ensure forecasting consistency among hierarchy levels based on their dimensional attributes, such as geography or product categories. Coherent forecasts are essential, meaning that higher-level forecasts must equal the sum of lower-level forecasts. This coherency requirement adds complexity to the original time series forecasting problem (<xref ref-type="bibr" rid="B118">Sagheer et al., 2021</xref>).</p>
<p>For biomedical data scenarios, HTS forecasting is applied in predicting instances similar to emergency medical services (EMS) requirements (<xref ref-type="bibr" rid="B117">Rostami-Tabar and Hyndman, 2024</xref>) and mortality rates across various U.S. states (<xref ref-type="bibr" rid="B81">Li and Hyndman, 2021</xref>; <xref ref-type="bibr" rid="B82">Li et al., 2024</xref>). Forecasting is crucial for EMS as it promotes consistency and synchronized resource allocation, enhancing decision-making processes and leading to better patient outcomes by avoiding the imbalance between demand and resources. In mortality rate predictions, forecasting addresses differences in mortality patterns across different geographic regions. Maintaining adherence between state-level and national-level mortality forecasts is vital for precise policy planning and resource management, aiding in reducing life expectancy disparities and enhancing public health results.</p>
<p>Different reconciliation procedures like top-down, bottom-up, and middle-out have been developed to maintain consistency across levels by generating base forecasts and then adjusting them. These procedures vary in approach: bottom-up starts from the lowest level and aggregates upwards, top-down begins at the highest level and disaggregates downwards, and middle-out combines both methods starting from an intermediate level. Each has its strengths and weaknesses, and none has proven universally superior. <xref ref-type="bibr" rid="B64">Hyndman et al. (2011)</xref> proposed an optimal combination approach, which independently forecasts all levels and then combines them using regression to ensure coherence. The Minimum Trace (MinT) method (<xref ref-type="bibr" rid="B143">Wickramasuriya et al., 2018</xref>) is another widely adopted approach for reconciliation. This technique uses the complete covariance matrix of forecast errors to generate a set of coherent forecasts. It aims to minimize the MSE of these coherent forecasts across the whole series, under the assumption of unbiasedness.</p>
<p>The approach detailed by <xref ref-type="bibr" rid="B117">Rostami-Tabar and Hyndman (2024)</xref> involves implementing forecast reconciliation for the hierarchical data of ambulance demand. It utilizes an ensemble of models: Exponential Smoothing State Space model (ETS), Poisson regression with Generalized Linear Model (GLM), and time series GLM (TSGLM). It generates base forecasts independently for each hierarchy level and reconcile them using the MinT method, minimizing forecast variances for coherence. Validation is done via time series cross-validation, with accuracy measured by mean absolute scaled error (MASE) and continuous ranked probability scores (CRPS). The methodology by <xref ref-type="bibr" rid="B81">Li and Hyndman (2021)</xref> ensures coherent mortality forecasts using a forecast reconciliation approach. Independent state-level forecasts are generated with the Lee-Carter model and then reconciled using the Minimum Trace (MinT) method together with the sampling approach by <xref ref-type="bibr" rid="B67">Jeon et al., (2019)</xref> to ensure consistency with national-level forecasts. Validation is performed using out-of-sample forecasting, with accuracy measured by MAPE and the Winkler score. The study uses U.S. mortality data from 1969 to 2017 and projects rates up to 2027. Another paper by <xref ref-type="bibr" rid="B82">Li et al. (2024)</xref> uses boosting with stochastic mortality models as weak learners. The authors extend gradient boosting with age-based and spatial shrinkage, iteratively fitting the Lee-Carter model to residuals and adding graph Laplacian-based penalties to align forecasts of adjacent age groups and states. Validation uses US male mortality data (1969&#x2013;2019), with forecasting performance assessed using MASE.</p>
<p>Traditionally, methods like ARIMA and exponential smoothing generate base forecasts but fail to capture individual and grouped time series dynamics, especially with time variation or sudden changes. They also struggle with exploiting complete hierarchical information, affecting forecasting efficiency. Recently, ML algorithms like artificial neural networks, extreme gradient boosting, and SVR have been employed to improve accuracy by considering nonlinear relationships and dynamic changes. However, they often still rely on traditional methods and may overlook useful hierarchical information. Overall, HTS forecasting remains a complex problem with ongoing research aimed at finding more efficient and accurate methods to ensure coherent and reliable forecasts across all levels of the hierarchy (<xref ref-type="bibr" rid="B118">Sagheer et al., 2021</xref>). Note: A list and description of open source tools for forecasting is provided in the <xref ref-type="sec" rid="s11">Supplementary Material</xref> of this article.</p>
</sec>
</sec>
<sec id="s5-2">
<title>5.2 Future directions</title>
<p>Extensive research has been conducted to interrogate biomedical temporal data in medical and health applications. Challenges remain, and are summarized into six key areas: (1) standardizing diverse data formats; (2) managing data quality; (3) ensuring model interpretability; (4) protecting patient privacy; (5) enabling real-time monitoring; and (6) addressing bias to create fair models. To grasp the potential future developments, we present a use case to illustrate six future directions within the clinical context. Specifically, taking Mr. Smith (45 years old) as a persona who is concerned about his risk of developing Alzheimer&#x2019;s Disease (AD).</p>
<sec id="s5-2-1">
<title>5.2.1 Data harmonization to standardize data format</title>
<p>Time series analysis plays a critical role in the early detection of AD by enabling the continuous monitoring of specific biomarkers over time. This approach is crucial for understanding the progression of the disease through its various stages, from preclinical AD to mild cognitive impairment (MCI), and ultimately to dementia. The primary biomarkers used in detecting and monitoring AD include beta-amyloid and tau proteins, which are typically measured in cerebrospinal fluid (CSF), along with imaging biomarkers such as PET scans for assessing beta-amyloid burden and MRI scans for detecting changes in brain volume. These biomarkers are indispensable for identifying the onset and progression of the disease, often before clinical symptoms become evident (<xref ref-type="bibr" rid="B58">Hernandez-Lorenzo et al., 2022</xref>).</p>
<p>During his visit to the physician, Mr. Smith is advised to undergo a series of tests, including genetic screening, neuroimaging, and cognitive assessments. These tests generate a diverse array of data types, ranging from genetic biomarkers to neuroimaging data (e.g., MRI scans) and time-series data derived from cognitive assessments. However, the data collected from Mr. Smith originate from multiple sources: a local hospital, a specialized lab for genetic testing, and a cognitive assessment app. To create a unified dataset, data harmonization is necessary, ensuring consistency across different formats, terminologies, and units. Implementing interoperable technologies can greatly facilitate seamless data exchange across disparate healthcare systems. Future research should focus on developing advanced harmonization techniques for time series data to ensure accurate and consistent integration from various sources. Additionally, integrating multi-modal data, such as clinical, genetic, and imaging information, will be crucial for creating personalized prediction models.</p>
</sec>
<sec id="s5-2-2">
<title>5.2.2 Data quality</title>
<p>As Mr. Smith assesses his risk of developing AD, data from various tests play a critical role in forecasting his condition. However, his data may contain missing values due to irregular monitoring, different data collection protocols, or the progression of his condition. Addressing these gaps is crucial for building a reliable predictive model. A promising approach involves filling these gaps and using the missing data as a valuable signal. Missing biomarker readings can be estimated using methods like forward-filling or zero imputation. The model can also incorporate indicators to highlight absent data points, learning from the pattern of missing data. For example, if Mr. Smith&#x27;s cognitive scores are missing for several months, the model can predict these values and use the absence of scores as a feature. This allows the model to detect patterns that may reveal insights such as health changes or inconsistencies in monitoring.</p>
<p>Ensuring data quality is essential for reliable predictive models in clinical research. Future directions should integrate advanced ML techniques that handle missing data and leverage the temporal patterns surrounding these gaps. By combining models that analyze available data and sequences of missing data, we can improve predictive accuracy, uncover hidden trends, and identify critical periods signaling disease progression. This approach enhances timely, personalized predictions for patients like Mr. Smith.</p>
</sec>
<sec id="s5-2-3">
<title>5.2.3 Interpretability</title>
<p>As Mr. Smith assesses his risk of developing AD, advanced ML models analyzing the biomarkers to identify the intervention strategies become crucial. Current models offer predictive power but often function as &#x201c;black boxes&#x201d; making it challenging to understand risk factors and the associated impacts. To address this, interpretability methods are essential to know the factors behind risk predictions. One important future direction on interpretability is to use attention mechanisms that prioritize key biomarkers and time points, focusing on early disease prediction characteristics. For example, attention-based models can highlight critical data points, such as changes in biomarkers that signal the onset of AD.</p>
<p>A significant biomarker decline flagged by the model would make the risk assessment more transparent, aiding the physician&#x27;s understanding and decisions (e.g., intervention). Alternatively, time-based SHAP (SHapley Additive exPlanations) techniques enhance model prediction transparency by assessing feature importance at specific times. Future work could focus on developing interpretability frameworks in personalized, real-time risk assessments for AD and other conditions, ensuring predictions are accurate and understandable for patients and clinicians.</p>
</sec>
<sec id="s5-2-4">
<title>5.2.4 Data privacy</title>
<p>As Mr. Smith evaluates his risk of AD, the sensitive data gathered requires rigorous privacy safeguards. Data privacy is crucial for legal compliance and maintaining trust in the healthcare system. Sharing sensitive information in research while retaining data utility is challenging. Anonymization is a technique to safeguard against reidentification while maintaining the usefulness of research data. Blockchain technology is another method, providing secure means for sharing data. Federated learning (FL) is also beneficial for collaborative studies, enabling ML models to be trained on Mr. Smith&#x2019;s data locally without the need for centralization, thus decreasing privacy risks. Informed consent is another essential aspect for research purposes. If consent is dynamic, it allows for real-time management, permitting alterations as new research develops. Future directions include implementing these techniques independently or as hybrid frameworks that improve privacy protection without sacrificing research utility. Establishing international standards for these methods is imperative for harmonizing global privacy practices and enhancing security and trust in collaborative research.</p>
</sec>
<sec id="s5-2-5">
<title>5.2.5 Real-time detection</title>
<p>Let&#x2019;s assume, the physician seeing Mr. Smith recommends the use of a wearable device that monitors essential physiological indicators such as sleep patterns and heart rate variability (HRV) to assess his AD risks. Note these devices have already demonstrated potential in identifying early signs of cognitive decline (<xref ref-type="bibr" rid="B120">Saif et al., 2020</xref>). With continuous, real-time monitoring, Mr. Smith would be empowered to take proactive actions&#x2014;such as making lifestyle changes or seeking further medical evaluations&#x2014;that could potentially delay the progression of the disease. We have observed an emerging trend in health domain to embed wearable devices into regular health surveillance, facilitating the early identification and treatment of AD or other disease conditions. A future direction in predictive modeling is high-fidelity model enabling real-time, or near real-time (e.g., 15 min) detection. Some related research questions include data storage (where data to be stored, cloud or locally), model calibration and fine tuning strategies (e.g., transfer learning).</p>
</sec>
<sec id="s5-2-6">
<title>5.2.6 Bias and fairness</title>
<p>A typical problem in AI models is the possibility of bias if they are trained on unrepresentative datasets. For example, if a model is trained mainly on data from old Asian females, it might inaccurately evaluate Mr. Smith, who is a middle-aged American male. Future directions for utilizing AI-driven models should emphasize making these models unbiased and dependable for various populations. A critical measure is the creation and validation of AI models with datasets that include a broad spectrum of demographics, such as different ages, ethnicities, and genders. Another approach to ensure fairness in AI algorithms is through regular audits and validation by independent experts. These audits can uncover and fix biases that could distort predictions. Independent audits help guarantee that AI models are equitable and effective for diverse groups thereby offering reliable health assessments. Additionally, it is essential for both healthcare providers and patients to recognize the potential biases in AI tools. By carefully reviewing AI-generated advice alongside clinical expertise and other diagnostic tools, healthcare providers can ensure that the AI model&#x27;s predictions are accurate and contextual.</p>
</sec>
</sec>
</sec>
<sec sec-type="conclusion" id="s6">
<title>6 Conclusion</title>
<p>In summary, the review paper outlines the challenges faced in predictive modeling for biomedical temporal data, such as managing missing values, addressing correlations between variables, capturing both short-term and long-term dependencies, performing multi-step ahead predictions, and considering data availability. It assesses models in three categories&#x2014;statistical, machine learning, and deep learning&#x2014;to evaluate their effectiveness in forecasting data amidst these challenges. Recognizing limitations in each approach, it discusses alternative methods like model enhancements or ensemble/combination forecasting techniques to potentially improve forecasting accuracy. The review also covers hierarchical forecasting for biomedical datasets with relevant structures. Moreover, it explores issues like data quality, privacy concerns, data harmonization, interpretability, real-time detection, and bias/fairness considerations in integrating AI or ML into clinical practices. These challenges underline the necessity for thorough data evaluation, strong privacy laws, and a deep understanding of the goals of predictive modeling. Moreover, successfully implementing these models necessitates a joint effort from the different fields, along with an inclusive approach that tackles not just the technical aspects of the model but also the broader ethical and fairness issues in healthcare environments.</p>
</sec>
</body>
<back>
<sec id="s7">
<title>Author contributions</title>
<p>AP: Conceptualization, Formal Analysis, Investigation, Writing&#x2013;original draft, Writing&#x2013;review and editing. FC: Writing&#x2013;original draft, Writing&#x2013;review and editing. FA-H: Writing&#x2013;review and editing. TW: Conceptualization, Supervision, Writing&#x2013;original draft, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. We extend our heartfelt appreciation to the National Science Foundation (NSF)&#x2014;Partnership for Innovation&#x2014;&#x201c;Avoiding Kidney Injuries&#x201d; (Award &#x23;2122901) and CPS: TTP Option: Medium: i-HEAR: immersive Human-On-the-Loop Environmental Adaptation for Stress Reduction (Award #2038905) for their generous support of this work.</p>
</sec>
<ack>
<p>&#x201c;Writefull&#x201d; Version 2.2.0 with Overleaf is utilized for improving the quality of the writing for this manuscript.</p>
</ack>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s11">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fphys.2024.1386760/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fphys.2024.1386760/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.PDF" id="SM1" mimetype="application/PDF" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abbasimehr</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Paki</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Improving time series forecasting using lstm and attention models</article-title>. <source>J. Ambient Intell. Humaniz. Comput.</source> <volume>13</volume>, <fpage>673</fpage>&#x2013;<lpage>691</lpage>. <pub-id pub-id-type="doi">10.1007/s12652-020-02761-x</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Albuquerque</surname>
<given-names>G. A.</given-names>
</name>
<name>
<surname>Carvalho</surname>
<given-names>D. D.</given-names>
</name>
<name>
<surname>Cruz</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>Santos</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Machado</surname>
<given-names>G. M.</given-names>
</name>
<name>
<surname>Gendriz</surname>
<given-names>I. S.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Osteoporosis screening using machine learning and electromagnetic waves</article-title>. <source>Sci. Rep.</source> <volume>13</volume>, <fpage>12865</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-023-40104-w</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Al-Hindawi</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Siddiquee</surname>
<given-names>M. M. R.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Domain-knowledge inspired pseudo supervision (dips) for unsupervised image-to-image translation models to support cross-domain classification</article-title>. <source>Eng. Appl. Artif. Intell.</source> <volume>127</volume>, <fpage>107255</fpage>. <pub-id pub-id-type="doi">10.1016/j.engappai.2023.107255</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Al-Hindawi</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Soori</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Siddiquee</surname>
<given-names>M. M. R.</given-names>
</name>
<name>
<surname>Yoon</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>A framework for generalizing critical heat flux detection models using unsupervised image-to-image translation</article-title>. <source>Expert Syst. Appl.</source> <volume>227</volume>, <fpage>120265</fpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2023.120265</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alhnaity</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Kollias</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Leontidis</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Schamp</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Pearson</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>An autoencoder wavelet based deep neural network with attention mechanism for multi-step prediction of plant growth</article-title>. <source>Inf. Sci.</source> <volume>560</volume>, <fpage>35</fpage>&#x2013;<lpage>50</lpage>. <pub-id pub-id-type="doi">10.1016/j.ins.2021.01.037</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Aljuaid</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Sasi</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Proper imputation techniques for missing values in data sets</article-title>,&#x201d; in <source>2016 international conference on data science and engineering (ICDSE)</source>, <fpage>1</fpage>&#x2013;<lpage>5</lpage>. <pub-id pub-id-type="doi">10.1109/ICDSE.2016.7823957</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Allam</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Feuerriegel</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rebhan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Krauthammer</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Analyzing patient trajectories with artificial intelligence</article-title>. <source>J. Med. internet Res.</source> <volume>23</volume>, <fpage>e29812</fpage>. <pub-id pub-id-type="doi">10.2196/29812</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Al-Qahtani</surname>
<given-names>F. H.</given-names>
</name>
<name>
<surname>Crone</surname>
<given-names>S. F.</given-names>
</name>
</person-group> (<year>2013</year>). &#x201c;<article-title>Multivariate k-nearest neighbour regression for time series data&#x2014;a novel algorithm for forecasting UK electricity demand</article-title>,&#x201d; in <source>The 2013 international joint conference on neural networks (IJCNN)</source> (<publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1109/IJCNN.2013.6706742</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Altarazi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Allaf</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Alhindawi</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Machine learning models for predicting and classifying the tensile strength of polymeric films fabricated via different production processes</article-title>. <source>Materials</source> <volume>12</volume>, <fpage>1475</fpage>. <pub-id pub-id-type="doi">10.3390/ma12091475</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Al Zahrani</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Al Sameeh</surname>
<given-names>F. A. R.</given-names>
</name>
<name>
<surname>Musa</surname>
<given-names>A. C.</given-names>
</name>
<name>
<surname>Shokeralla</surname>
<given-names>A. A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Forecasting diabetes patients attendance at al-baha hospitals using autoregressive fractional integrated moving average (arfima) models</article-title>. <source>J. Data Analysis Inf. Process.</source> <volume>8</volume>, <fpage>183</fpage>&#x2013;<lpage>194</lpage>. <pub-id pub-id-type="doi">10.4236/jdaip.2020.83011</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xiong</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Multi-step-ahead time series prediction using multiple-output support vector regression</article-title>. <source>Neurocomputing</source> <volume>129</volume>, <fpage>482</fpage>&#x2013;<lpage>493</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2013.09.010</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barreto</surname>
<given-names>T. d. O.</given-names>
</name>
<name>
<surname>Veras</surname>
<given-names>N. V. R.</given-names>
</name>
<name>
<surname>Cardoso</surname>
<given-names>P. H.</given-names>
</name>
<name>
<surname>Fernandes</surname>
<given-names>F. R. d. S.</given-names>
</name>
<name>
<surname>Medeiros</surname>
<given-names>L. P. d. S.</given-names>
</name>
<name>
<surname>Bezerra</surname>
<given-names>M. V.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Artificial intelligence applied to analyzes during the pandemic: covid-19 beds occupancy in the state of rio grande do norte, Brazil</article-title>. <source>Front. Artif. Intell.</source> <volume>6</volume>, <fpage>1290022</fpage>. <pub-id pub-id-type="doi">10.3389/frai.2023.1290022</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Baum</surname>
<given-names>L. E.</given-names>
</name>
<name>
<surname>Eagon</surname>
<given-names>J. A.</given-names>
</name>
</person-group> (<year>1967</year>). <source>An inequality with applications to statistical estimation for probabilistic functions of markov processes and to a model for ecology</source>.</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Baum</surname>
<given-names>L. E.</given-names>
</name>
</person-group> (<year>1972</year>). <article-title>An inequality and associated maximization technique in statistical estimation for probabilistic functions of markov processes</article-title>. <source>Inequalities</source> <volume>3</volume>, <fpage>1</fpage>&#x2013;<lpage>8</lpage>.</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Baum</surname>
<given-names>L. E.</given-names>
</name>
<name>
<surname>Petrie</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>1966</year>). <article-title>Statistical inference for probabilistic functions of finite state Markov chains</article-title>. <source>Ann. Math. statistics</source> <volume>37</volume>, <fpage>1554</fpage>&#x2013;<lpage>1563</lpage>. <pub-id pub-id-type="doi">10.1214/aoms/1177699147</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Baum</surname>
<given-names>L. E.</given-names>
</name>
<name>
<surname>Petrie</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Soules</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Weiss</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>1970</year>). <article-title>A maximization technique occurring in the statistical analysis of probabilistic functions of Markov chains</article-title>. <source>Ann. Math. statistics</source> <volume>41</volume>, <fpage>164</fpage>&#x2013;<lpage>171</lpage>. <pub-id pub-id-type="doi">10.1214/aoms/1177697196</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Baum</surname>
<given-names>L. E.</given-names>
</name>
<name>
<surname>Sell</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>1968</year>). <article-title>Growth transformations for functions on manifolds</article-title>. <source>Pac. J. Math.</source> <volume>27</volume>, <fpage>211</fpage>&#x2013;<lpage>227</lpage>. <pub-id pub-id-type="doi">10.2140/pjm.1968.27.211</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Bayyurt</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Bayyurt</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2020</year>). <source>Forecasting of covid-19 cases and deaths using arima models</source>, <fpage>2020</fpage>. <comment>medrxiv</comment>. <pub-id pub-id-type="doi">10.1101/2020.04.17.20069237</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bonilla</surname>
<given-names>E. V.</given-names>
</name>
<name>
<surname>Chai</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Williams</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Multi-task Gaussian process prediction</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>20</volume>.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bou-Hamad</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Jamali</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Forecasting financial time-series using data mining models: a simulation study</article-title>. <source>Res. Int. Bus. Finance</source> <volume>51</volume>, <fpage>101072</fpage>. <pub-id pub-id-type="doi">10.1016/j.ribaf.2019.101072</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Breiman</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Random forests</article-title>. <source>Mach. Learn.</source> <volume>45</volume>, <fpage>5</fpage>&#x2013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1023/A:1010933404324</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cai</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Patharkar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Lure</surname>
<given-names>F. Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>V. C.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Stride: systematic radar intelligence analysis for adrd risk evaluation with gait signature simulation and deep learning</article-title>. <source>IEEE sensors J.</source> <volume>23</volume>, <fpage>10998</fpage>&#x2013;<lpage>11006</lpage>. <pub-id pub-id-type="doi">10.1109/jsen.2023.3263071</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cai</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Hsu</surname>
<given-names>Y.-M.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Gaussian process regression for numerical wind speed prediction enhancement</article-title>. <source>Renew. energy</source> <volume>146</volume>, <fpage>2112</fpage>&#x2013;<lpage>2123</lpage>. <pub-id pub-id-type="doi">10.1016/j.renene.2019.08.018</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Cao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Multi-step ahead forecasting for fault prognosis using hidden markov model</article-title>,&#x201d; in <source>The 27th Chinese control and decision conference (2015 CCDC)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>1688</fpage>&#x2013;<lpage>1692</lpage>. <pub-id pub-id-type="doi">10.1109/CCDC.2015.7162191</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chandra</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Goyal</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Evaluation of deep learning models for multi-step ahead time series prediction</article-title>. <source>IEEE Access</source> <volume>9</volume>, <fpage>83105</fpage>&#x2013;<lpage>83123</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2021.3085085</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Che</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Purushotham</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Cho</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Sontag</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Recurrent neural networks for multivariate time series with missing values</article-title>. <source>Sci. Rep.</source> <volume>8</volume>, <fpage>6085</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-018-24271-9</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cheng</surname>
<given-names>L.-F.</given-names>
</name>
<name>
<surname>Dumitrascu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Darnell</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Chivers</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Draugelis</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Sparse multi-output Gaussian processes for online medical time series prediction</article-title>. <source>BMC Med. Inf. Decis. Mak.</source> <volume>20</volume>, <fpage>1</fpage>&#x2013;<lpage>23</lpage>. <pub-id pub-id-type="doi">10.1186/s12911-020-1069-4</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chuah</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2007</year>). &#x201c;<article-title>Ecg anomaly detection via time series analysis</article-title>,&#x201d; in <source>Frontiers of high performance computing and networking ISPA 2007 workshops: ISPA 2007 international workshops SSDSN, UPWN, WISH, SGC, ParDMCom, HiPCoMB, and IST-AWSN niagara falls, Canada, august 28-september 1, 2007 proceedings 5</source> (<publisher-name>Springer</publisher-name>), <fpage>123</fpage>&#x2013;<lpage>135</lpage>.</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Daydulo</surname>
<given-names>Y. D.</given-names>
</name>
<name>
<surname>Thamineni</surname>
<given-names>B. L.</given-names>
</name>
<name>
<surname>Dawud</surname>
<given-names>A. A.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Cardiac arrhythmia detection using deep learning approach and time frequency representation of ecg signals</article-title>. <source>BMC Med. Inf. Decis. Mak.</source> <volume>23</volume>, <fpage>232</fpage>. <pub-id pub-id-type="doi">10.1186/s12911-023-02326-w</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>De Gooijer</surname>
<given-names>J. G.</given-names>
</name>
<name>
<surname>Hyndman</surname>
<given-names>R. J.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>25 years of time series forecasting</article-title>. <source>Int. J. Forecast.</source> <volume>22</volume>, <fpage>443</fpage>&#x2013;<lpage>473</lpage>. <pub-id pub-id-type="doi">10.1016/j.ijforecast.2006.01.001</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>de Wolff</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Cuevas</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tobar</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Mogptk: the multi-output Gaussian process toolkit</article-title>. <source>Neurocomputing</source> <volume>424</volume>, <fpage>49</fpage>&#x2013;<lpage>53</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2020.09.085</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ding</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Jiao</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2020</year>). <source>Brief analysis of the arima model on the covid-19 in Italy</source>, <fpage>2020</fpage>. <comment>medRxiv</comment>. <pub-id pub-id-type="doi">10.1101/2020.04.08.20058636</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Doretto</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Chiuso</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y. N.</given-names>
</name>
<name>
<surname>Soatto</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Dynamic textures</article-title>. <source>Int. J. Comput. Vis.</source> <volume>51</volume>, <fpage>91</fpage>&#x2013;<lpage>109</lpage>. <pub-id pub-id-type="doi">10.1023/A:1021669406132</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Du</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Horng</surname>
<given-names>S.-J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Multivariate time series forecasting via attention-based encoder&#x2013;decoder framework</article-title>. <source>Neurocomputing</source> <volume>388</volume>, <fpage>269</fpage>&#x2013;<lpage>279</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2019.12.118</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>D&#xfc;richen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Pimentel</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Clifton</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Schweikard</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Clifton</surname>
<given-names>D. A.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>Multi-task Gaussian process models for biomedical applications</article-title>,&#x201d; in <source>
<italic>IEEE-EMBS international Conference on Biomedical and health informatics (BHI)</italic> (IEEE)</source>, <fpage>492</fpage>&#x2013;<lpage>495</lpage>. <pub-id pub-id-type="doi">10.1109/BHI.2014.6864410</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ehsani</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Drabl&#xf8;s</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Robust distance measures for k nn classification of cancer data</article-title>. <source>Cancer Inf.</source> <volume>19</volume>, <fpage>1176935120965542</fpage>. <pub-id pub-id-type="doi">10.1177/1176935120965542</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>El Mrabet</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Sugunaraj</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Ranganathan</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Abhyankar</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Random forest regressor-based approach for detecting fault location and duration in power systems</article-title>. <source>Sensors</source> <volume>22</volume>, <fpage>458</fpage>. <pub-id pub-id-type="doi">10.3390/s22020458</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Evensen</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>1994</year>). <article-title>Sequential data assimilation with a nonlinear quasi-geostrophic model using Monte Carlo methods to forecast error statistics</article-title>. <source>J. Geophys. Res. Oceans</source> <volume>99</volume>, <fpage>10143</fpage>&#x2013;<lpage>10162</lpage>. <pub-id pub-id-type="doi">10.1029/94jc00572</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Evgeniou</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Micchelli</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Pontil</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Shawe-Taylor</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Learning multiple tasks with kernel methods</article-title>. <source>J. Mach. Learn. Res.</source> <volume>6</volume>.</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fan</surname>
<given-names>G.-F.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>L.-L.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Hong</surname>
<given-names>W.-C.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Applications of hybrid emd with pso and ga for an svr-based load forecasting model</article-title>. <source>Energies</source> <volume>10</volume>, <fpage>1713</fpage>. <pub-id pub-id-type="doi">10.3390/en10111713</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Lahdelma</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Evaluation of a multiple linear regression model and sarima model in forecasting heat demand for district heating system</article-title>. <source>Appl. energy</source> <volume>179</volume>, <fpage>544</fpage>&#x2013;<lpage>552</lpage>. <pub-id pub-id-type="doi">10.1016/j.apenergy.2016.06.133</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Filipow</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Main</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Tanriver</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Raywood</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Davies</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Douglas</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Exploring flexible polynomial regression as a method to align routine clinical outcomes with daily data capture through remote technologies</article-title>. <source>BMC Med. Res. Methodol.</source> <volume>23</volume>, <fpage>114</fpage>. <pub-id pub-id-type="doi">10.1186/s12874-023-01942-4</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fix</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Hodges</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>1989</year>). <article-title>Discriminatory analysis. nonparametric discrimination: consistency properties</article-title>. <source>Int. Stat. Review/Revue Int. Stat.</source> <volume>57</volume>, <fpage>238</fpage>&#x2013;<lpage>247</lpage>. <pub-id pub-id-type="doi">10.2307/1403797</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Galton</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>1886</year>). <article-title>Regression towards mediocrity in hereditary stature</article-title>. <source>J. Anthropol. Inst. G. B. Irel.</source> <volume>15</volume>, <fpage>246</fpage>&#x2013;<lpage>263</lpage>. <pub-id pub-id-type="doi">10.2307/2841583</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Garcia</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Debreuve</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Nielsen</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Barlaud</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2010</year>). &#x201c;<article-title>K-nearest neighbor search: fast gpu-based implementations and application to high-dimensional feature matching</article-title>,&#x201d; in <source>2010 IEEE international Conference on image processing (IEEE)</source>, <fpage>3757</fpage>&#x2013;<lpage>3760</lpage>.</citation>
</ref>
<ref id="B46">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Gauss</surname>
<given-names>C.-F.</given-names>
</name>
</person-group> (<year>1823</year>). <source>Theoria combinations observationum erroribus minimis obnoxiae (Henricus Dieterich)</source>.</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gavrishchaka</surname>
<given-names>V. V.</given-names>
</name>
<name>
<surname>Banerjee</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Support vector machine as an efficient framework for stock market volatility forecasting</article-title>. <source>Comput. Manag. Sci.</source> <volume>3</volume>, <fpage>147</fpage>&#x2013;<lpage>160</lpage>. <pub-id pub-id-type="doi">10.1007/s10287-005-0005-5</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ghaderi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Foreman</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Nayebi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tipirneni</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Reddy</surname>
<given-names>C. K.</given-names>
</name>
<name>
<surname>Subbian</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A self-supervised learning-based approach to clustering multivariate time-series data with missing values (slac-time): an application to tbi phenotyping</article-title>. <source>J. Biomed. Inf.</source> <volume>143</volume>, <fpage>104401</fpage>. <pub-id pub-id-type="doi">10.1016/j.jbi.2023.104401</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ghahramani</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Hinton</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>1996</year>). <source>Parameter estimation for linear dynamical systems</source>.</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Godfrey</surname>
<given-names>L. B.</given-names>
</name>
<name>
<surname>Gashler</surname>
<given-names>M. S.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Neural decomposition of time-series data for effective generalization</article-title>. <source>IEEE Trans. neural Netw. Learn. Syst.</source> <volume>29</volume>, <fpage>2973</fpage>&#x2013;<lpage>2985</lpage>. <pub-id pub-id-type="doi">10.1109/TNNLS.2017.2709324</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Goldberger</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Amaral</surname>
<given-names>L. A. N.</given-names>
</name>
<name>
<surname>Glass</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Hausdorff</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Ivanov</surname>
<given-names>P. C.</given-names>
</name>
<name>
<surname>Mark</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2000</year>). <source>Physiobank, physiotoolkit, and physiome: components of a new research resource for complex physiologic signals</source>.</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gopakumar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tran</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Phung</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Venkatesh</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Forecasting daily patient outflow from a ward having no real-time clinical data</article-title>. <source>JMIR Med. Inf.</source> <volume>4</volume>, <fpage>e25</fpage>. <pub-id pub-id-type="doi">10.2196/medinform.5650</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gordon</surname>
<given-names>N. J.</given-names>
</name>
<name>
<surname>Salmond</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>A. F.</given-names>
</name>
</person-group> (<year>1993</year>). <article-title>Novel approach to nonlinear/non-Gaussian Bayesian state estimation</article-title>. <source>IEE Proc. F (radar signal processing) (IET)</source> <volume>140</volume>, <fpage>107</fpage>&#x2013;<lpage>113</lpage>. <pub-id pub-id-type="doi">10.1049/ip-f-2.1993.0015</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hamdi</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Ali</surname>
<given-names>J. B.</given-names>
</name>
<name>
<surname>Di Costanzo</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Fnaiech</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Moreau</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Ginoux</surname>
<given-names>J.-M.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Accurate prediction of continuous blood glucose based on support vector regression and differential evolution algorithm</article-title>. <source>Biocybern. Biomed. Eng.</source> <volume>38</volume>, <fpage>362</fpage>&#x2013;<lpage>372</lpage>. <pub-id pub-id-type="doi">10.1016/j.bbe.2018.02.005</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Harerimana</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>J. W.</given-names>
</name>
<name>
<surname>Jang</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A multi-headed transformer approach for predicting the patient&#x2019;s clinical time-series variables from charted vital signs</article-title>. <source>IEEE Access</source> <volume>10</volume>, <fpage>105993</fpage>&#x2013;<lpage>106004</lpage>. <pub-id pub-id-type="doi">10.1109/access.2022.3211334</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Haywood</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wilson</surname>
<given-names>G. T.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>A test for improved multi-step forecasting</article-title>. <source>J. Time Ser. Analysis</source> <volume>30</volume>, <fpage>682</fpage>&#x2013;<lpage>707</lpage>. <pub-id pub-id-type="doi">10.1111/j.1467-9892.2009.00634.x</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Helmini</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jihan</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Jayasinghe</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Perera</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Sales forecasting using multivariate long short term memory network models</article-title>. <source>PeerJ Prepr.</source> <volume>7</volume>, <fpage>e27712v1</fpage>. <pub-id pub-id-type="doi">10.7287/peerj.preprints.27712v1</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hernandez-Lorenzo</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ilundain</surname>
<given-names>I. S.</given-names>
</name>
<name>
<surname>Rodrigo</surname>
<given-names>J. L. A.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Timeseries biomarkers clustering for alzheimer&#x27;s disease progression</article-title>,&#x201d; in <source>2022 IEEE international conference on omni-layer intelligent systems (COINS)</source> (<publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>7</lpage>.</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hewamalage</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Bergmeir</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bandara</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Recurrent neural networks for time series forecasting: current status and future directions</article-title>. <source>Int. J. Forecast.</source> <volume>37</volume>, <fpage>388</fpage>&#x2013;<lpage>427</lpage>. <pub-id pub-id-type="doi">10.1016/j.ijforecast.2020.06.008</pub-id>
</citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hochreiter</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Schmidhuber</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>Long short-term memory</article-title>. <source>Neural Comput.</source> <volume>9</volume>, <fpage>1735</fpage>&#x2013;<lpage>1780</lpage>. <pub-id pub-id-type="doi">10.1162/neco.1997.9.8.1735</pub-id>
</citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Holt</surname>
<given-names>C. C.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Forecasting seasonals and trends by exponentially weighted moving averages</article-title>. <source>Int. J. Forecast.</source> <volume>20</volume>, <fpage>5</fpage>&#x2013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1016/j.ijforecast.2003.09.015</pub-id>
</citation>
</ref>
<ref id="B62">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hosseinzadeh</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Nassar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Boubrahimi</surname>
<given-names>S. F.</given-names>
</name>
<name>
<surname>Hamdi</surname>
<given-names>S. M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Ml-based streamflow prediction in the upper Colorado river basin using climate variables time series data</article-title>. <source>Hydrology</source> <volume>10</volume>, <fpage>29</fpage>. <pub-id pub-id-type="doi">10.3390/hydrology10020029</pub-id>
</citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ghalamsiah</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Patharkar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Pradhan</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Chu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>An entropy-based causality framework for cross-level faults diagnosis and isolation in building hvac systems</article-title>. <source>Energy Build.</source> <volume>317</volume>, <fpage>114378</fpage>. <pub-id pub-id-type="doi">10.1016/j.enbuild.2024.114378</pub-id>
</citation>
</ref>
<ref id="B64">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hyndman</surname>
<given-names>R. J.</given-names>
</name>
<name>
<surname>Ahmed</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Athanasopoulos</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Shang</surname>
<given-names>H. L.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Optimal combination forecasts for hierarchical time series</article-title>. <source>Comput. statistics and data analysis</source> <volume>55</volume>, <fpage>2579</fpage>&#x2013;<lpage>2589</lpage>. <pub-id pub-id-type="doi">10.1016/j.csda.2011.03.006</pub-id>
</citation>
</ref>
<ref id="B65">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Idriss</surname>
<given-names>T. E.</given-names>
</name>
<name>
<surname>Idri</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Abnane</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Bakkoury</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Predicting blood glucose using an lsm neural network</article-title>,&#x201d; in <source>2019 federated conference on computer science and information systems (FedCSIS)</source>, <fpage>35</fpage>&#x2013;<lpage>41</lpage>. <pub-id pub-id-type="doi">10.15439/2019F159</pub-id>
</citation>
</ref>
<ref id="B66">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Janacek</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Time series analysis forecasting and control</article-title>. <source>J. Time Ser. Analysis</source> <volume>31</volume>, <fpage>303</fpage>. <pub-id pub-id-type="doi">10.1111/j.1467-9892.2009.00643.x</pub-id>
</citation>
</ref>
<ref id="B67">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jeon</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Panagiotelis</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Petropoulos</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Probabilistic forecast reconciliation with applications to wind power and electric load</article-title>. <source>Eur. J. Operational Res.</source> <volume>279</volume>, <fpage>364</fpage>&#x2013;<lpage>379</lpage>. <pub-id pub-id-type="doi">10.1016/j.ejor.2019.05.020</pub-id>
</citation>
</ref>
<ref id="B68">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Jones</surname>
<given-names>P. W.</given-names>
</name>
<name>
<surname>Osipov</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Rokhlin</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2011</year>). &#x201c;<article-title>Randomized approximate nearest neighbors algorithm</article-title>,&#x201d; in <source>Proceedings of the national academy of sciences 108</source>, <fpage>15679</fpage>&#x2013;<lpage>15686</lpage>.</citation>
</ref>
<ref id="B69">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Julier</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Uhlmann</surname>
<given-names>J. K.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>New extension of the kalman filter to nonlinear systems</article-title>. <source>Signal Process. Sens. fusion, target Recognit. VI (Spie)</source> <volume>3068</volume>, <fpage>182</fpage>&#x2013;<lpage>193</lpage>. <pub-id pub-id-type="doi">10.1117/12.280797</pub-id>
</citation>
</ref>
<ref id="B70">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Jun</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Mulyadi</surname>
<given-names>A. W.</given-names>
</name>
<name>
<surname>Suk</surname>
<given-names>H.-I.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Stochastic imputation and uncertainty-aware attention to ehr for mortality prediction</article-title>,&#x201d; in <source>2019 international joint conference on neural networks (IJCNN)</source>, <fpage>1</fpage>&#x2013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1109/IJCNN.2019.8852132</pub-id>
</citation>
</ref>
<ref id="B71">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kalman</surname>
<given-names>R. E.</given-names>
</name>
</person-group> (<year>1963</year>). <article-title>Mathematical description of linear dynamical systems</article-title>. <source>J. Soc. Industrial Appl. Math. Ser. A Control</source> <volume>1</volume>, <fpage>152</fpage>&#x2013;<lpage>192</lpage>. <pub-id pub-id-type="doi">10.1137/0301010</pub-id>
</citation>
</ref>
<ref id="B72">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kandula</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yamana</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Pei</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Morita</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Shaman</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Evaluation of mechanistic and statistical methods in forecasting influenza-like illness</article-title>. <source>J. R. Soc. Interface</source> <volume>15</volume>, <fpage>20180174</fpage>. <pub-id pub-id-type="doi">10.1098/rsif.2018.0174</pub-id>
</citation>
</ref>
<ref id="B73">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kantz</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Schreiber</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2004</year>). <source>Nonlinear time series analysis</source>, <volume>7</volume>. <publisher-name>Cambridge University Press</publisher-name>. <pub-id pub-id-type="doi">10.1017/cbo9780511755798</pub-id>
</citation>
</ref>
<ref id="B74">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Katari</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Likith</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Sree</surname>
<given-names>M. P. S.</given-names>
</name>
<name>
<surname>Rachapudi</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Heart disease prediction using hybrid ml algorithms</article-title>,&#x201d; in <source>2023 international conference on sustainable computing and data communication systems (ICSCDS)</source> (<publisher-name>IEEE</publisher-name>), <fpage>121</fpage>&#x2013;<lpage>125</lpage>.</citation>
</ref>
<ref id="B75">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Katayama</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2005</year>). <source>Subspace methods for system identification</source>. <publisher-name>Springer</publisher-name>. <pub-id pub-id-type="doi">10.1007/1-84628-158-X</pub-id>
</citation>
</ref>
<ref id="B76">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khalique</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Khan</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Butt</surname>
<given-names>W. H.</given-names>
</name>
<name>
<surname>Matloob</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>An integrated approach for spatio-temporal cholera disease hotspot relation mining for public health management in Punjab, Pakistan</article-title>. <source>Int. J. Environ. Res. Public Health</source> <volume>17</volume>, <fpage>3763</fpage>. <pub-id pub-id-type="doi">10.3390/ijerph17113763</pub-id>
</citation>
</ref>
<ref id="B77">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Lai</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>W.-C.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Modeling long-and short-term temporal patterns with deep neural networks</article-title>,&#x201d; in <source>The 41st international ACM SIGIR conference on research and development in information retrieval</source>, <fpage>95</fpage>&#x2013;<lpage>104</lpage>. <pub-id pub-id-type="doi">10.1145/3209978.3210006</pub-id>
</citation>
</ref>
<ref id="B78">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Hauskrecht</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Modeling multivariate clinical event time-series with recurrent temporal mechanisms</article-title>. <source>Artif. Intell. Med.</source> <volume>112</volume>, <fpage>102021</fpage>. <pub-id pub-id-type="doi">10.1016/j.artmed.2021.102021</pub-id>
</citation>
</ref>
<ref id="B79">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Legendre</surname>
<given-names>A. M.</given-names>
</name>
</person-group> (<year>1806</year>). <source>Nouvelles m&#xe9;thodes pour la d&#xe9;termination des orbites des com&#xe8;tes: avec un suppl&#xe9;ment contenant divers perfectionnemens de ces m&#xe9;thodes et leur application aux deux com&#xe8;tes de 1805 (Courcier)</source>.</citation>
</ref>
<ref id="B80">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Lewis</surname>
<given-names>F. L.</given-names>
</name>
</person-group> (<year>1986</year>). <source>Optimal estimation: with an introduction to stochastic control theory</source>. <comment>
<italic>(No Title)</italic>
</comment>.</citation>
</ref>
<ref id="B81">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Hyndman</surname>
<given-names>R. J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Assessing mortality inequality in the us: what can be said about the future?</article-title> <source>Insur. Math. Econ.</source> <volume>99</volume>, <fpage>152</fpage>&#x2013;<lpage>162</lpage>. <pub-id pub-id-type="doi">10.1016/j.insmatheco.2021.03.014</pub-id>
</citation>
</ref>
<ref id="B82">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Panagiotelis</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Boosting domain-specific models with shrinkage: an application in mortality forecasting</article-title>. <source>Int. J. Forecast.</source> <pub-id pub-id-type="doi">10.1016/j.ijforecast.2024.05.001</pub-id>
</citation>
</ref>
<ref id="B83">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xuan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.-X.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Enhancing the locality and breaking the memory bottleneck of transformer on time series forecasting</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>32</volume>.</citation>
</ref>
<ref id="B84">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Application of sarima model in forecasting and analyzing inpatient cases of acute mountain sickness</article-title>. <source>BMC Public Health</source> <volume>23</volume>, <fpage>56</fpage>. <pub-id pub-id-type="doi">10.1186/s12889-023-14994-4</pub-id>
</citation>
</ref>
<ref id="B85">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2016</year>). <source>Time series modeling of irregularly sampled multivariate clinical data</source>. <publisher-name>University of Pittsburgh</publisher-name>. <comment>Ph.D. thesis</comment>.</citation>
</ref>
<ref id="B86">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Hauskrecht</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Clinical time series prediction: toward a hierarchical dynamical system framework</article-title>. <source>Artif. Intell. Med.</source> <volume>65</volume>, <fpage>5</fpage>&#x2013;<lpage>18</lpage>. <pub-id pub-id-type="doi">10.1016/j.artmed.2014.10.005</pub-id>
</citation>
</ref>
<ref id="B87">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Hauskrecht</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2013</year>). &#x201c;<article-title>Modeling clinical time series using Gaussian process sequences</article-title>,&#x201d; in <source>Proceedings of the 2013 SIAM international conference on data mining</source> (<publisher-name>SIAM</publisher-name>), <fpage>623</fpage>&#x2013;<lpage>631</lpage>.</citation>
</ref>
<ref id="B88">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Forecast methods for time series data: a survey</article-title>. <source>Ieee Access</source> <volume>9</volume>, <fpage>91896</fpage>&#x2013;<lpage>91912</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2021.3091162</pub-id>
</citation>
</ref>
<ref id="B89">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Luo</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Multivariate time series imputation with generative adversarial networks</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>31</volume>.</citation>
</ref>
<ref id="B90">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mahmudimanesh</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mirzaee</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Dehghan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bahrampour</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Forecasts of cardiac and respiratory mortality in tehran, Iran, using arimax and cnn-lstm models</article-title>. <source>Environ. Sci. Pollut. Res.</source> <volume>29</volume>, <fpage>28469</fpage>&#x2013;<lpage>28479</lpage>. <pub-id pub-id-type="doi">10.1007/s11356-021-18205-8</pub-id>
</citation>
</ref>
<ref id="B91">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Manaris</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Hughes</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Vassilandonakis</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2011</year>). &#x201c;<article-title>Monterey mirror: combining markov models, genetic algorithms, and power laws</article-title>,&#x201d; in <source>Computer science department, college of charleston, SC, USA, appearred in proceedings of 1st workshop in evolutionary music, 2011 IEEE congress on evolutionary computation (CEC 2011)</source> (<publisher-loc>New Orleans, LA, USA (Citeseer)</publisher-loc>), <fpage>33</fpage>&#x2013;<lpage>40</lpage>.</citation>
</ref>
<ref id="B92">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mart&#xed;nez</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Fr&#xed;as</surname>
<given-names>M. P.</given-names>
</name>
<name>
<surname>P&#xe9;rez</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>Rivera</surname>
<given-names>A. J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A methodology for applying k-nearest neighbor to time series forecasting</article-title>. <source>Artif. Intell. Rev.</source> <volume>52</volume>, <fpage>2019</fpage>&#x2013;<lpage>2037</lpage>. <pub-id pub-id-type="doi">10.1007/s10462-017-9593-z</pub-id>
</citation>
</ref>
<ref id="B93">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mitra</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ashraf</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Sepsis prediction and vital signs ranking in intensive care unit patients</article-title>. <source>arXiv Prepr. arXiv:1812.06686</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1812.06686</pub-id>
</citation>
</ref>
<ref id="B94">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mollura</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lehman</surname>
<given-names>L.-W. H.</given-names>
</name>
<name>
<surname>Mark</surname>
<given-names>R. G.</given-names>
</name>
<name>
<surname>Barbieri</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A novel artificial intelligence based intensive care unit monitoring system: using physiological waveforms to identify sepsis</article-title>. <source>Philosophical Trans. R. Soc. A</source> <volume>379</volume>, <fpage>20200252</fpage>. <pub-id pub-id-type="doi">10.1098/rsta.2020.0252</pub-id>
</citation>
</ref>
<ref id="B95">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Montgomery</surname>
<given-names>D. C.</given-names>
</name>
<name>
<surname>Jennings</surname>
<given-names>C. L.</given-names>
</name>
<name>
<surname>Kulahci</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2015</year>). <source>Introduction to time series analysis and forecasting</source>. <publisher-name>John Wiley and Sons</publisher-name>.</citation>
</ref>
<ref id="B96">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moon</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Son</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hwang</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Hybrid short-term load forecasting scheme using random forest and multilayer perceptron</article-title>. <source>Energies</source> <volume>11</volume>, <fpage>3283</fpage>. <pub-id pub-id-type="doi">10.3390/en11123283</pub-id>
</citation>
</ref>
<ref id="B97">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mulyadi</surname>
<given-names>A. W.</given-names>
</name>
<name>
<surname>Jun</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Suk</surname>
<given-names>H.-I.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Uncertainty-aware variational-recurrent imputation network for clinical time series</article-title>. <source>IEEE Trans. Cybern.</source> <volume>52</volume>, <fpage>9684</fpage>&#x2013;<lpage>9694</lpage>. <pub-id pub-id-type="doi">10.1109/TCYB.2021.3053599</pub-id>
</citation>
</ref>
<ref id="B98">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Naemi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mansourvar</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Schmidt</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wiil</surname>
<given-names>U. K.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Prediction of patients severity at emergency department using narx and ensemble learning</article-title>,&#x201d; in <source>
<italic>2020 IEEE international Conference on Bioinformatics and biomedicine (BIBM)</italic> (IEEE)</source>, <fpage>2793</fpage>&#x2013;<lpage>2799</lpage>. <pub-id pub-id-type="doi">10.1109/BIBM49941.2020.9313462</pub-id>
</citation>
</ref>
<ref id="B99">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Niu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Fusion of sequential visits and medical ontology for mortality prediction</article-title>. <source>J. Biomed. Inf.</source> <volume>127</volume>, <fpage>104012</fpage>. <pub-id pub-id-type="doi">10.1016/j.jbi.2022.104012</pub-id>
</citation>
</ref>
<ref id="B100">
<citation citation-type="book">
<comment>[Dataset]</comment> <person-group person-group-type="author">
<name>
<surname>Overchee</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Moor</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>1996</year>). <source>Subspace identification for linear system</source>.</citation>
</ref>
<ref id="B101">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Patharkar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Forzani</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Thomas</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Lind</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Eigen-entropy based time series signatures to support multivariate time series classification</article-title>. <source>Sci. Rep.</source> <volume>14</volume>, <fpage>16076</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-024-66953-7</pub-id>
</citation>
</ref>
<ref id="B102">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pearson</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>1901</year>). <article-title>Liii. on lines and planes of closest fit to systems of points in space</article-title>. <source>Lond. Edinb. Dublin philosophical Mag. J. Sci.</source> <volume>2</volume>, <fpage>559</fpage>&#x2013;<lpage>572</lpage>. <pub-id pub-id-type="doi">10.1080/14786440109462720</pub-id>
</citation>
</ref>
<ref id="B103">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pearson</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>1922</year>). <article-title>Francis galton 1822-1922: a centenary appreciation</article-title>, <fpage>1</fpage>&#x2013;<lpage>28</lpage>. <pub-id pub-id-type="doi">10.1037/11161-001</pub-id>
</citation>
</ref>
<ref id="B104">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Pearson</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>The life, letters and labours of francis galton</article-title>,&#x201d; in <source>Scientific and medical knowledge production, 1796-1918</source> (<publisher-loc>London, United Kingdom</publisher-loc>: <publisher-name>Routledge</publisher-name>), <fpage>311</fpage>&#x2013;<lpage>318</lpage>.</citation>
</ref>
<ref id="B105">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pillonetto</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Dinuzzo</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>De Nicolao</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Bayesian online multitask learning of Gaussian processes</article-title>. <source>IEEE Trans. Pattern Analysis Mach. Intell.</source> <volume>32</volume>, <fpage>193</fpage>&#x2013;<lpage>205</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2008.297</pub-id>
</citation>
</ref>
<ref id="B106">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pinto</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Valentim</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>da Silva</surname>
<given-names>L. F.</given-names>
</name>
<name>
<surname>de Souza</surname>
<given-names>G. F.</given-names>
</name>
<name>
<surname>de Moura Santos</surname>
<given-names>T. G. F.</given-names>
</name>
<name>
<surname>de Oliveira</surname>
<given-names>C. A. P.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Use of interrupted time series analysis in understanding the course of the congenital syphilis epidemic in Brazil</article-title>. <source>Lancet Regional Health&#x2013;Americas</source> <volume>7</volume>, <fpage>100163</fpage>. <pub-id pub-id-type="doi">10.1016/j.lana.2021.100163</pub-id>
</citation>
</ref>
<ref id="B107">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Plis</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Bunescu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Marling</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Shubrook</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Schwartz</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>A machine learning approach to predicting blood glucose levels for diabetes management</article-title>,&#x201d; in <source>Workshops at the Twenty-Eighth AAAI conference on artificial intelligence</source>.</citation>
</ref>
<ref id="B108">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Poloni</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Sbrana</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>A note on forecasting demand using the multivariate exponential smoothing framework</article-title>. <source>Int. J. Prod. Econ.</source> <volume>162</volume>, <fpage>143</fpage>&#x2013;<lpage>150</lpage>. <pub-id pub-id-type="doi">10.1016/j.ijpe.2015.01.017</pub-id>
</citation>
</ref>
<ref id="B109">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qi</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Sarfima model prediction for infectious diseases: application to hemorrhagic fever with renal syndrome and comparing with sarima</article-title>. <source>BMC Med. Res. Methodol.</source> <volume>20</volume>, <fpage>243</fpage>&#x2013;<lpage>247</lpage>. <pub-id pub-id-type="doi">10.1186/s12874-020-01130-8</pub-id>
</citation>
</ref>
<ref id="B110">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Quinonero-Candela</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Rasmussen</surname>
<given-names>C. E.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>A unifying view of sparse approximate Gaussian process regression</article-title>. <source>J. Mach. Learn. Res.</source> <volume>6</volume>, <fpage>1939</fpage>&#x2013;<lpage>1959</lpage>.</citation>
</ref>
<ref id="B111">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rabyk</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Schmid</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Ewma control charts for detecting changes in the mean of a long-memory process</article-title>. <source>Metrika</source> <volume>79</volume>, <fpage>267</fpage>&#x2013;<lpage>301</lpage>. <pub-id pub-id-type="doi">10.1007/s00184-015-0555-7</pub-id>
</citation>
</ref>
<ref id="B112">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rachmat</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Suhartono</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Comparative analysis of single exponential smoothing and holt&#x2019;s method for quality of hospital services forecasting in general hospital</article-title>. <source>Bull. Comput. Sci. Electr. Eng.</source> <volume>1</volume>, <fpage>80</fpage>&#x2013;<lpage>86</lpage>. <pub-id pub-id-type="doi">10.25008/bcsee.v1i2.8</pub-id>
</citation>
</ref>
<ref id="B113">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Reyna</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Josef</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Seyedi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jeter</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Shashikumar</surname>
<given-names>S. P.</given-names>
</name>
<name>
<surname>Westover</surname>
<given-names>M. B.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). &#x201c;<article-title>Early prediction of sepsis from clinical data: the physionet/computing in cardiology challenge 2019</article-title>,&#x201d; in <source>2019 computing in cardiology (cinc)</source> (<publisher-name>IEEE</publisher-name>).</citation>
</ref>
<ref id="B114">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reyna</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Josef</surname>
<given-names>C. S.</given-names>
</name>
<name>
<surname>Jeter</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Shashikumar</surname>
<given-names>S. P.</given-names>
</name>
<name>
<surname>Westover</surname>
<given-names>M. B.</given-names>
</name>
<name>
<surname>Nemati</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Early prediction of sepsis from clinical data: the physionet/computing in cardiology challenge 2019</article-title>. <source>Crit. care Med.</source> <volume>48</volume>, <fpage>210</fpage>&#x2013;<lpage>217</lpage>. <pub-id pub-id-type="doi">10.1097/CCM.0000000000004145</pub-id>
</citation>
</ref>
<ref id="B115">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Roberts</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>Control chart tests based on geometric moving averages</article-title>. <source>Technometrics</source> <volume>42</volume>, <fpage>97</fpage>&#x2013;<lpage>101</lpage>. <pub-id pub-id-type="doi">10.1080/00401706.2000.10485986</pub-id>
</citation>
</ref>
<ref id="B116">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Roberts</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Osborne</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ebden</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Reece</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gibson</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Aigrain</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Gaussian processes for time-series modelling</article-title>. <source>Philosophical Trans. R. Soc. A Math. Phys. Eng. Sci.</source> <volume>371</volume>, <fpage>20110550</fpage>. <pub-id pub-id-type="doi">10.1098/rsta.2011.0550</pub-id>
</citation>
</ref>
<ref id="B117">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rostami-Tabar</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Hyndman</surname>
<given-names>R. J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Hierarchical time series forecasting in emergency medical services</article-title>. <source>J. Serv. Res.</source>, <fpage>10946705241232169</fpage>. <pub-id pub-id-type="doi">10.1177/10946705241232169</pub-id>
</citation>
</ref>
<ref id="B118">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sagheer</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hamdoun</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Youness</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Deep lstm-based transfer learning approach for coherent forecasts in hierarchical time series</article-title>. <source>Sensors</source> <volume>21</volume>, <fpage>4379</fpage>. <pub-id pub-id-type="doi">10.3390/s21134379</pub-id>
</citation>
</ref>
<ref id="B119">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Said</surname>
<given-names>A. B.</given-names>
</name>
<name>
<surname>Erradi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Aly</surname>
<given-names>H. A.</given-names>
</name>
<name>
<surname>Mohamed</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Predicting covid-19 cases using bidirectional lstm on multivariate time series</article-title>. <source>Environ. Sci. Pollut. Res.</source> <volume>28</volume>, <fpage>56043</fpage>&#x2013;<lpage>56052</lpage>. <pub-id pub-id-type="doi">10.1007/s11356-021-14286-7</pub-id>
</citation>
</ref>
<ref id="B120">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Saif</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Niotis</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Scheyer</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Rahman</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Berkowitz</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Feasibility of using a wearable biosensor device in patients at risk for alzheimer&#x27;s disease dementia</article-title>. <source>J. Prev. Alzheimer&#x2019;s Dis.</source> <volume>7</volume>, <fpage>104</fpage>&#x2013;<lpage>111</lpage>. <pub-id pub-id-type="doi">10.14283/jpad.2019.39</pub-id>
</citation>
</ref>
<ref id="B121">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Samaee</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kobravi</surname>
<given-names>H. R.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Predicting the occurrence of wrist tremor based on electromyography using a hidden markov model and entropy based learning algorithm</article-title>. <source>Biomed. Signal Process. Control</source> <volume>57</volume>, <fpage>101739</fpage>. <pub-id pub-id-type="doi">10.1016/j.bspc.2019.101739</pub-id>
</citation>
</ref>
<ref id="B122">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shamout</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Clifton</surname>
<given-names>D. A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Machine learning for clinical outcome prediction</article-title>. <source>IEEE Rev. Biomed. Eng.</source> <volume>14</volume>, <fpage>116</fpage>&#x2013;<lpage>126</lpage>. <pub-id pub-id-type="doi">10.1109/RBME.2020.3007816</pub-id>
</citation>
</ref>
<ref id="B123">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shashikumar</surname>
<given-names>S. P.</given-names>
</name>
<name>
<surname>Stanley</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>Sadiq</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Holder</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Clifford</surname>
<given-names>G. D.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Early sepsis detection in critical care patients using multiscale blood pressure and heart rate dynamics</article-title>. <source>J. Electrocardiol.</source> <volume>50</volume>, <fpage>739</fpage>&#x2013;<lpage>743</lpage>. <pub-id pub-id-type="doi">10.1016/j.jelectrocard.2017.08.013</pub-id>
</citation>
</ref>
<ref id="B124">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shih</surname>
<given-names>S.-Y.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>F.-K.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>H.-y.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Temporal pattern attention for multivariate time series forecasting</article-title>. <source>Mach. Learn.</source> <volume>108</volume>, <fpage>1421</fpage>&#x2013;<lpage>1441</lpage>. <pub-id pub-id-type="doi">10.1007/s10994-019-05815-0</pub-id>
</citation>
</ref>
<ref id="B125">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Shukla</surname>
<given-names>S. N.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Estimation of blood pressure from non-invasive data</article-title>,&#x201d; in <source>2017 39th annual international conference of the IEEE engineering in medicine and biology society (EMBC)</source> (<publisher-name>IEEE</publisher-name>), <fpage>1772</fpage>&#x2013;<lpage>1775</lpage>. <pub-id pub-id-type="doi">10.1109/EMBC.2017.8037187</pub-id>
</citation>
</ref>
<ref id="B126">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Siami-Namini</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tavakoli</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Namin</surname>
<given-names>A. S.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>The performance of lstm and bilstm in forecasting time series</article-title>,&#x201d; in <source>2019 IEEE International conference on big data (Big Data)</source> (<publisher-name>IEEE</publisher-name>), <fpage>3285</fpage>&#x2013;<lpage>3292</lpage>.</citation>
</ref>
<ref id="B127">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sotoodeh</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ho</surname>
<given-names>J. C.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Improving length of stay prediction using a hidden markov model</article-title>. <source>AMIA Summits Transl. Sci. Proc.</source> <volume>2019</volume>, <fpage>425</fpage>&#x2013;<lpage>434</lpage>.</citation>
</ref>
<ref id="B128">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stegle</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Lippert</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Mooij</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Lawrence</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Borgwardt</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Efficient inference in matrix-variate Gaussian models with&#x5c;iid observation noise</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>24</volume>.</citation>
</ref>
<ref id="B129">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Suganya</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Arunadevi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Buhari</surname>
<given-names>S. M.</given-names>
</name>
</person-group> (<year>2020</year>). <source>Covid-19 forecasting using multivariate linear regression</source>.</citation>
</ref>
<ref id="B130">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Eeg classification with transformer-based models</article-title>,&#x201d; in <source>2021 ieee 3rd global conference on life sciences and technologies (lifetech)</source> (<publisher-name>IEEE</publisher-name>), <fpage>92</fpage>&#x2013;<lpage>93</lpage>. <pub-id pub-id-type="doi">10.1109/LifeTech52111.2021.9391844</pub-id>
</citation>
</ref>
<ref id="B131">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tandon</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ranjan</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Chakraborty</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Suhag</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Coronavirus (covid-19): arima-based time-series analysis to forecast near future and the effect of school reopening in India</article-title>. <source>J. Health Manag.</source> <volume>24</volume>, <fpage>373</fpage>&#x2013;<lpage>388</lpage>. <pub-id pub-id-type="doi">10.1177/09720634221109087</pub-id>
</citation>
</ref>
<ref id="B132">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tipirneni</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Reddy</surname>
<given-names>C. K.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Self-supervised transformer for sparse and irregularly sampled multivariate clinical time-series</article-title>. <source>ACM Trans. Knowl. Discov. Data (TKDD)</source> <volume>16</volume>, <fpage>1</fpage>&#x2013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.1145/3516367</pub-id>
</citation>
</ref>
<ref id="B133">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tyralis</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Papacharalampous</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Variable selection in time series forecasting using random forests</article-title>. <source>Algorithms</source> <volume>10</volume>, <fpage>114</fpage>. <pub-id pub-id-type="doi">10.3390/a10040114</pub-id>
</citation>
</ref>
<ref id="B134">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Valentim</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Caldeira-Silva</surname>
<given-names>G. J.</given-names>
</name>
<name>
<surname>Da Silva</surname>
<given-names>R. D.</given-names>
</name>
<name>
<surname>Albuquerque</surname>
<given-names>G. A.</given-names>
</name>
<name>
<surname>De Andrade</surname>
<given-names>I. G.</given-names>
</name>
<name>
<surname>Sales-Moioli</surname>
<given-names>A. I. L.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Stochastic petri net model describing the relationship between reported maternal and congenital syphilis cases in Brazil</article-title>. <source>BMC Med. Inf. Decis. Mak.</source> <volume>22</volume>, <fpage>40</fpage>. <pub-id pub-id-type="doi">10.1186/s12911-022-01773-1</pub-id>
</citation>
</ref>
<ref id="B135">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Vapnik</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>1999</year>). <source>The nature of statistical learning theory</source>. <publisher-name>Springer science and business media</publisher-name>.</citation>
</ref>
<ref id="B136">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vapnik</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Golowich</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Smola</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>1996</year>). <article-title>Support vector method for function approximation, regression estimation and signal processing</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>9</volume>.</citation>
</ref>
<ref id="B137">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vaswani</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shazeer</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Parmar</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Uszkoreit</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gomez</surname>
<given-names>A. N.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Attention is all you need</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>30</volume>.</citation>
</ref>
<ref id="B138">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vellido</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Societal issues concerning the application of artificial intelligence in medicine</article-title>. <source>Kidney Dis.</source> <volume>5</volume>, <fpage>11</fpage>&#x2013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.1159/000492428</pub-id>
</citation>
</ref>
<ref id="B139">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Veneri</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Pretegiani</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Rosini</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Federighi</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Federico</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Rufa</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Evaluating the human ongoing visual search performance by eye tracking application and sequencing tests</article-title>. <source>Comput. methods programs Biomed.</source> <volume>107</volume>, <fpage>468</fpage>&#x2013;<lpage>477</lpage>. <pub-id pub-id-type="doi">10.1016/j.cmpb.2011.02.006</pub-id>
</citation>
</ref>
<ref id="B140">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Molenaar</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Harsh</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Freeman</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gold</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Personalized state-space modeling of glucose dynamics for type 1 diabetes using continuously monitored glucose, insulin dose, and meal intake: an extended kalman filter approach</article-title>. <source>J. diabetes Sci. Technol.</source> <volume>8</volume>, <fpage>331</fpage>&#x2013;<lpage>345</lpage>. <pub-id pub-id-type="doi">10.1177/1932296814524080</pub-id>
</citation>
</ref>
<ref id="B141">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Hyndman</surname>
<given-names>R. J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Forecast combinations: an over 50-year review</article-title>. <source>Int. J. Forecast.</source> <volume>39</volume>, <fpage>1518</fpage>&#x2013;<lpage>1547</lpage>. <pub-id pub-id-type="doi">10.1016/j.ijforecast.2022.11.005</pub-id>
</citation>
</ref>
<ref id="B142">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Weerakody</surname>
<given-names>P. B.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>K. W.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Ela</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A review of irregular time series data handling with gated recurrent neural networks</article-title>. <source>Neurocomputing</source> <volume>441</volume>, <fpage>161</fpage>&#x2013;<lpage>178</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2021.02.046</pub-id>
</citation>
</ref>
<ref id="B143">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wickramasuriya</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Athanasopoulos</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Hyndman</surname>
<given-names>R. J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Optimal forecast reconciliation for hierarchical and grouped time series through trace minimization</article-title>. <source>J. Am. Stat. Assoc.</source> <volume>114</volume>, <fpage>804</fpage>&#x2013;<lpage>819</lpage>. <pub-id pub-id-type="doi">10.1080/01621459.2018.1448825</pub-id>
</citation>
</ref>
<ref id="B144">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Williams</surname>
<given-names>C. K.</given-names>
</name>
<name>
<surname>Rasmussen</surname>
<given-names>C. E.</given-names>
</name>
</person-group> (<year>2006</year>). <source>Gaussian processes for machine learning</source>. <publisher-loc>Cambridge, MA</publisher-loc>: <publisher-name>MIT press</publisher-name>.</citation>
</ref>
<ref id="B145">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Winters</surname>
<given-names>P. R.</given-names>
</name>
</person-group> (<year>1960</year>). <article-title>Forecasting sales by exponentially weighted moving averages</article-title>. <source>Manag. Sci.</source> <volume>6</volume>, <fpage>324</fpage>&#x2013;<lpage>342</lpage>. <pub-id pub-id-type="doi">10.1287/mnsc.6.3.324</pub-id>
</citation>
</ref>
<ref id="B146">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wold</surname>
<given-names>H. O.</given-names>
</name>
</person-group> (<year>1948</year>). <article-title>On prediction in stationary time series</article-title>. <source>Ann. Math. Statistics</source> <volume>19</volume>, <fpage>558</fpage>&#x2013;<lpage>567</lpage>. <pub-id pub-id-type="doi">10.1214/aoms/1177730151</pub-id>
</citation>
</ref>
<ref id="B147">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Covid-19 dynamics prediction by improved multi-polynomial regression model</article-title>,&#x201d; in <source>The 2nd international conference on computing and data science</source>, <fpage>1</fpage>&#x2013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1145/3448734.3450847</pub-id>
</citation>
</ref>
<ref id="B148">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xiao</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Choi</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Opportunities and challenges in developing deep learning models using electronic health records data: a systematic review</article-title>. <source>J. Am. Med. Inf. Assoc.</source> <volume>25</volume>, <fpage>1419</fpage>&#x2013;<lpage>1428</lpage>. <pub-id pub-id-type="doi">10.1093/jamia/ocy068</pub-id>
</citation>
</ref>
<ref id="B149">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xie</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ning</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ong</surname>
<given-names>M. E. H.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hsu</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Deep learning for temporal data representation in electronic health records: a systematic review of challenges and methodologies</article-title>. <source>J. Biomed. Inf.</source> <volume>126</volume>, <fpage>103980</fpage>. <pub-id pub-id-type="doi">10.1016/j.jbi.2021.103980</pub-id>
</citation>
</ref>
<ref id="B150">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>A multi-directional approach for missing value estimation in multivariate time series clinical data</article-title>. <source>J. Healthc. Inf. Res.</source> <volume>4</volume>, <fpage>365</fpage>&#x2013;<lpage>382</lpage>. <pub-id pub-id-type="doi">10.1007/s41666-020-00076-2</pub-id>
</citation>
</ref>
<ref id="B151">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yoon</surname>
<given-names>B.-J.</given-names>
</name>
<name>
<surname>Vaidyanathan</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Context-sensitive hidden markov models for modeling long-range dependencies in symbol sequences</article-title>. <source>IEEE Trans. Signal Process.</source> <volume>54</volume>, <fpage>4169</fpage>&#x2013;<lpage>4184</lpage>. <pub-id pub-id-type="doi">10.1109/TSP.2006.880252</pub-id>
</citation>
</ref>
<ref id="B152">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yule</surname>
<given-names>G. U.</given-names>
</name>
</person-group> (<year>1927</year>). <article-title>Vii. on a method of investigating periodicities disturbed series, with special reference to wolfer&#x2019;s sunspot numbers</article-title>. <source>Philosophical Trans. R. Soc. Lond. Ser. A, Contain. Pap. a Math. or Phys. Character</source> <volume>226</volume>, <fpage>267</fpage>&#x2013;<lpage>298</lpage>. <pub-id pub-id-type="doi">10.1098/rsta.1927.0007</pub-id>
</citation>
</ref>
<ref id="B153">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zeng</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Are transformers effective for time series forecasting?</article-title> <source>Proc. AAAI Conf. Artif. Intell.</source> <volume>37</volume>, <fpage>11121</fpage>&#x2013;<lpage>11128</lpage>. <pub-id pub-id-type="doi">10.1609/aaai.v37i9.26317</pub-id>
</citation>
</ref>
<ref id="B154">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Predicting blood pressure from physiological index data using the svr algorithm</article-title>. <source>BMC Bioinforma.</source> <volume>20</volume>, <fpage>109</fpage>&#x2013;<lpage>115</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-019-2667-y</pub-id>
</citation>
</ref>
<ref id="B155">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A temporal fusion transformer for short-term freeway traffic speed multistep prediction</article-title>. <source>Neurocomputing</source> <volume>500</volume>, <fpage>329</fpage>&#x2013;<lpage>340</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2022.05.083</pub-id>
</citation>
</ref>
<ref id="B156">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Flores</surname>
<given-names>K. B.</given-names>
</name>
<name>
<surname>Tran</surname>
<given-names>H. T.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Deep learning and regression approaches to forecasting blood glucose levels for type 1 diabetes</article-title>. <source>Biomed. Signal Process. Control</source> <volume>69</volume>, <fpage>102923</fpage>. <pub-id pub-id-type="doi">10.1016/j.bspc.2021.102923</pub-id>
</citation>
</ref>
<ref id="B157">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>McDermaid</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Predicting outcomes of chronic kidney disease from emr data based on random forest regression</article-title>. <source>Math. Biosci.</source> <volume>310</volume>, <fpage>24</fpage>&#x2013;<lpage>30</lpage>. <pub-id pub-id-type="doi">10.1016/j.mbs.2019.02.001</pub-id>
</citation>
</ref>
<ref id="B158">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hooi</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Ye</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Beatgan: anomalous rhythm detection using adversarially generated time series</article-title>. <source>IJCAI</source> <volume>2019</volume>, <fpage>4433</fpage>&#x2013;<lpage>4439</lpage>. <pub-id pub-id-type="doi">10.24963/ijcai.2019/616</pub-id>
</citation>
</ref>
<ref id="B159">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Herrero</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Georgiou</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Dilated recurrent neural networks for glucose forecasting in type 1 diabetes</article-title>. <source>J. Healthc. Inf. Res.</source> <volume>4</volume>, <fpage>308</fpage>&#x2013;<lpage>324</lpage>. <pub-id pub-id-type="doi">10.1007/s41666-020-00068-2</pub-id>
</citation>
</ref>
<ref id="B160">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Donner</surname>
<given-names>R. V.</given-names>
</name>
<name>
<surname>Marwan</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Donges</surname>
<given-names>J. F.</given-names>
</name>
<name>
<surname>Kurths</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Complex network approaches to nonlinear time series analysis</article-title>. <source>Phys. Rep.</source> <volume>787</volume>, <fpage>1</fpage>&#x2013;<lpage>97</lpage>. <pub-id pub-id-type="doi">10.1016/j.physrep.2018.10.005</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>