<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Med. Technol.</journal-id>
<journal-title>Frontiers in Medical Technology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Med. Technol.</abbrev-journal-title>
<issn pub-type="epub">2673-3129</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmedt.2025.1621158</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Medical Technology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Using machine learning methods to investigate the impact of comorbidities and clinical indicators on the mortality rate of COVID-19</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes"><name><surname>Hsieh</surname><given-names>Yueh-Chen</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="an1"><sup>&#x2020;</sup></xref><role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/><role content-type="https://credit.niso.org/contributor-roles/methodology/"/><role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/></contrib>
<contrib contrib-type="author" equal-contrib="yes"><name><surname>Chen</surname><given-names>Sin</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="an1"><sup>&#x2020;</sup></xref><uri xlink:href="https://loop.frontiersin.org/people/3198722/overview" /><role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/><role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/></contrib>
<contrib contrib-type="author"><name><surname>Tsao</surname><given-names>Shu-Yu</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref><uri xlink:href="https://loop.frontiersin.org/people/3053649/overview"/><role content-type="https://credit.niso.org/contributor-roles/software/"/><role content-type="https://credit.niso.org/contributor-roles/validation/"/><role content-type="https://credit.niso.org/contributor-roles/visualization/"/><role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/></contrib>
<contrib contrib-type="author"><name><surname>Hu</surname><given-names>Jiun-Ruey</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref><role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/></contrib>
<contrib contrib-type="author"><name><surname>Hsu</surname><given-names>Wan-Ting</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref><uri xlink:href="https://loop.frontiersin.org/people/1260259/overview" /><role content-type="https://credit.niso.org/contributor-roles/supervision/"/><role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/></contrib>
<contrib contrib-type="author" corresp="yes"><name><surname>Lee</surname><given-names>Chien-Chang</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff7"><sup>7</sup></xref>
<xref ref-type="corresp" rid="cor1">&#x002A;</xref><uri xlink:href="https://loop.frontiersin.org/people/913726/overview" /><role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/><role content-type="https://credit.niso.org/contributor-roles/methodology/"/><role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/></contrib>
</contrib-group>
<aff id="aff1"><label><sup>1</sup></label><institution>Department of Laboratory Medicine, National Taiwan University Hospital Yunlin Branch</institution>, <addr-line>Douliou</addr-line>, <country>Taiwan</country></aff>
<aff id="aff2"><label><sup>2</sup></label><institution>Department of Medicine, College of Medicine, National Taiwan University</institution>, <addr-line>Taipei</addr-line>, <country>Taiwan</country></aff>
<aff id="aff3"><label><sup>3</sup></label><institution>Department of Emergency Medicine, National Taiwan University Hospital</institution>, <addr-line>Taipei</addr-line>, <country>Taiwan</country></aff>
<aff id="aff4"><label><sup>4</sup></label><institution>Department of Cardiology, Smidt Heart Institute, Cedars-Sinai Medical Center</institution>, <addr-line>Los Angeles, CA</addr-line>, <country>United States</country></aff>
<aff id="aff5"><label><sup>5</sup></label><institution>Department of Epidemiology, Harvard T.H. Chan School of Public Health</institution>, <addr-line>Boston, MA</addr-line>, <country>United States</country></aff>
<aff id="aff6"><label><sup>6</sup></label><institution>Clinical AI Consulting Group</institution>, <addr-line>Brookline, MA</addr-line>, <country>United States</country></aff>
<aff id="aff7"><label><sup>7</sup></label><institution>Department of Information Management, Ministry of Health and Welfare</institution>, <addr-line>Taipei</addr-line>, <country>Taiwan</country></aff>
<author-notes>
<fn fn-type="edited-by"><p><bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/417299/overview">Pradeep Nair</ext-link>, Indo Pacific Studies Center, Australia</p></fn>
<fn fn-type="edited-by"><p><bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2197901/overview">Mihaela Dinsoreanu</ext-link>, Technical University of Cluj-Napoca, Romania</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1583750/overview">Zheng Yuan</ext-link>, China Academy of Chinese Medical Sciences, China</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3120011/overview">Balraj Preet Kaur</ext-link>, Chandigarh Group of Colleges Jhanjeri, India</p></fn>
<corresp id="cor1"><label>&#x002A;</label><bold>Correspondence:</bold> Chien-Chang Lee <email>hit3transparency@gmail.com</email>; <email>cclee100@gmail.com</email></corresp>
<fn fn-type="equal" id="an1"><label><sup>&#x2020;</sup></label><p>These authors have contributed equally to this work</p></fn>
</author-notes>
<pub-date pub-type="epub"><day>22</day><month>09</month><year>2025</year></pub-date>
<pub-date pub-type="collection"><year>2025</year></pub-date>
<volume>7</volume><elocation-id>1621158</elocation-id>
<history>
<date date-type="received"><day>06</day><month>05</month><year>2025</year></date>
<date date-type="accepted"><day>01</day><month>09</month><year>2025</year></date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2025 Hsieh, Chen, Tsao, Hu, Hsu and Lee.</copyright-statement>
<copyright-year>2025</copyright-year><copyright-holder>Hsieh, Chen, Tsao, Hu, Hsu and Lee</copyright-holder><license license-type="open-access" xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract><sec><title>Background</title>
<p>This study aims to develop a machine learning model to predict the 30-day mortality risk of hospitalized COVID-19 patients while leveraging federated learning to enhance data privacy and expand the model&#x0027;s applicability. Additionally, SHapley Additive exPlanations (SHAP) values were utilized to assess the impact of comorbidities on mortality.</p>
</sec><sec><title>Methods</title>
<p>A retrospective analysis was conducted on 6,321 clinical records of hospitalized COVID-19 patients between January 2021 and October 2022. After excluding cases involving patients under 18 years of age and non-Omicron infections, a total of 4,081 records were analyzed. Key features included three demographic data, six vital signs at admission, and 79 underlying comorbidities. Four machine learning models were compared, including Lasso, Random Forest, XGBoost, and TabNet, with XGBoost demonstrating superior performance. Federated learning was implemented to enable collaborative model training across multiple medical institutions while maintaining data security. SHAP values were applied to interpret the contribution of each comorbidity to the model&#x0027;s predictions.</p>
</sec><sec><title>Results</title>
<p>A subset of 2,156 records from the Taipei branch was used to evaluate model performance. XGBoost achieved the highest AUC of 0.96 and a sensitivity of 0.94. Two versions of the XGBoost model were trained: one incorporating vital signs, suitable for emergency room applications where patients come in with unstable vital signs, and another excluding vital signs, optimized for outpatient settings where we encounter patients with multiple comorbidities. After implementing federated learning, the AUC of the Taipei cohort decreased to 0.90, while the performance of other cohorts improved to meet the required standards. SHAP analysis identified comorbidities including diabetes mellitus, cerebrovascular disease, and chronic lung disease to have a neutral or even protective association with 30-day mortality.</p>
</sec><sec><title>Conclusion</title>
<p>XGBoost outperformed other models making it a viable tool for both emergency and outpatient settings. The study underscores the importance of chronic disease assessment in predicting COVID-19 mortality, revealing some comorbidities such as diabetes mellitus, cerebrovascular disease and chronic lung disease to have protective association with 30-day mortality. These findings suggest potential refinements in current treatment guidelines, particularly concerning high-risk conditions. The integration of federated learning further enhances the model&#x0027;s clinical applicability while preserving patient privacy.</p>
</sec>
</abstract>
<kwd-group>
<kwd>COVID-19</kwd>
<kwd>mortality</kwd>
<kwd>survival analysis</kwd>
<kwd>machine learning</kwd>
<kwd>federated learning</kwd>
</kwd-group><contract-num rid="cn001">NSTC 113-2314-B-002-178, MOST 110-2314-B-002-053-MY3</contract-num><contract-num rid="cn002">NTUHYL. 112-AI005, NTUHYL. 113-X023</contract-num><contract-sponsor id="cn001">National Science and Technology Council</contract-sponsor><contract-sponsor id="cn002">National Taiwan University Hospital Yunlin Branch</contract-sponsor><counts>
<fig-count count="4"/>
<table-count count="6"/><equation-count count="0"/><ref-count count="33"/><page-count count="11"/><word-count count="0"/></counts><custom-meta-wrap><custom-meta><meta-name>section-at-acceptance</meta-name><meta-value>Medtech Data Analytics</meta-value></custom-meta></custom-meta-wrap>
</article-meta>
</front>
<body><sec id="s1" sec-type="intro"><label>1</label><title>Introduction</title>
<p>Coronavirus disease 2019 (COVID-19) is a contagious disease caused by the virus SARS-CoV-2, which has had a profound impact on global economies, healthcare systems, and social norms (<xref ref-type="bibr" rid="B1">1</xref>). Since the initial case was identified in Wuhan, China, in December 2019 (<xref ref-type="bibr" rid="B2">2</xref>), over 777 million individuals have been infected, and more than 7 million have died worldwide (as of 9 February 2025) (<xref ref-type="bibr" rid="B3">3</xref>). The average time from exposure to symptom onset is five days, and approximately 5&#x0025; of patients with COVID-19 experience severe symptoms necessitating intensive care (<xref ref-type="bibr" rid="B4">4</xref>). While the diagnosis of COVID-19 is facilitated by the use of rapid antigen tests (RATs) (<xref ref-type="bibr" rid="B5">5</xref>) and polymerase chain reaction (PCR) (<xref ref-type="bibr" rid="B6">6</xref>) technology, the challenge lies in accurately assessing the severity of the disease based on clinical data and chest x-ray features (<xref ref-type="bibr" rid="B7">7</xref>). In 2020 WHO developed a Clinical Progression Scale, patients have been categorized as those with mild disease (ambulatory, not requiring supplemental oxygen), those with moderate disease (hospitalized, might requiring low-flow oxygen), and those with severe COVID-19 (on HFNC, NIV, IMV, or ECMO) (<xref ref-type="bibr" rid="B8">8</xref>). Nevertheless, accurate prediction of the prognosis of COVID-19 remains an elusive endeavor. Predicting COVID-19 mortality is important as it has significant implications for the selection of pharmacologic treatments, management strategies, and for family planning and goals of care discussions (<xref ref-type="bibr" rid="B9">9</xref>).</p>
<p>Previous clinical decision models have focused on common health data rather than comorbidities, which might have negative impact on accuracy since comorbidities and COVID-19 mortality are correlated (<xref ref-type="bibr" rid="B10">10</xref>). The 4C Mortality Score, which was identified as the most promising risk stratification model in numerous systematic reviews (<xref ref-type="bibr" rid="B11">11</xref>, <xref ref-type="bibr" rid="B12">12</xref>), is a risk stratification tool that predicts in-hospital mortality rate for hospitalized COVID-19 patients with eight parameters (age, sex, number of comorbidities, respiratory rate, peripheral oxygen saturation, level of consciousness, urea level, and C reactive protein) (<xref ref-type="bibr" rid="B13">13</xref>). Risk stratification tool is a method that predict one&#x0027;s risk based on its clinical histories and other factors. The 4C mortality score sum up the scores of the eight parameters and range from 1 to 21, each represent a certain mortality rate respectively. While the 4C Score trained with 35,463 patients showed moderate diagnostic accuracy for mortality with derivation cohort area under the receiver operating characteristic curve (AUC) of 0.79, it performs poorly on other cohorts, with AUC ranging from 0.63 to 0.73 (<xref ref-type="bibr" rid="B13">13</xref>). In the present study, we sought to increase the accuracy by incorporating the identity of the comorbidity, not just the number of comorbidities, to the prediction. We developed the Comorbidities and Clinical Indicators on the Mortality Model (CCIMM), a machine learning model that can accurately predict a patient&#x0027;s mortality rate within 30 days of hospitalization, using 79 comorbidities that are readily available at the time of admission.</p>
</sec>
<sec id="s2" sec-type="methods"><label>2</label><title>Methods</title>
<sec id="s2a"><label>2.1</label><title>Data collection</title>
<p>National Taiwan University Hospital (NTUH) operates three major branches in Taipei, Hsinchu, and Yu nlin, all of which are tertiary medical centers, with 2,600 beds, 1,500 beds, and 900 beds, respectively. This study collected clinical data retrospectively on patients hospitalized with Omicron variant COVID-19 from April 2022 to October 2022 across the three hospitals. Data included demographics, admission vital signs [e.g., temperature, breath rate, pulse rate, systolic blood pressure [SBP], diastolic blood pressure [DBP]], and underlying comorbidities. This research project was approved by the ethics committee of National Taiwan University Hospital Institutional Review Board. The study was conducted in accordance with the principles of the Declaration of Helsinki and the Good Clinical Practice Guidelines, and all the participants were informed consent. We have no access to information that could identify individual participants during or after data collection.</p>
</sec>
<sec id="s2b"><label>2.2</label><title>Outcomes</title>
<p>The primary outcomes were 30-day all-cause mortality. Mortality outcomes were verified by linking the database with the national death registry, enabling accurate determination of survival status after discharge. Institutional review boards of each respective hospital approved waivers of informed consent, and all data were deidentified.</p>
</sec>
<sec id="s2c"><label>2.3</label><title>Missing data</title>
<p>The <italic>missForest</italic> R package was used to impute missing values for continuous variables such as age, body mass index (BMI), temperature, respiratory rate, pulse rate, SBP, and DBP. This iterative method constructs random forest models for each variable with missing data, using observed values from other variables as predictors to estimate the missing values.</p>
</sec>
<sec id="s2d"><label>2.4</label><title>Input variables</title>
<p>Two sets of data were subjected to the establishment of the model: one comprising demographic information, vital signs upon admission, and underlying comorbidities; the other, comprising only demographic information and underlying comorbidities. The datasets include 79 comorbidities and 6 vital signs. A <italic>p</italic>-value&#x2009;&#x003C;&#x2009;0.05 in comorbidities and vital signs was considered to be statistically significant between COVID-19 survivors and non-survivors. A total of 6,321 patients were identified across the three hospitals, with 2,240 excluded due to patient age under 18 or the identification of non-Omicron variants of SARS-CoV-2. The discovery of Omicron variants was in South Africa in late November 2021 (<xref ref-type="bibr" rid="B14">14</xref>). In Taiwan, the Omicron variant became predominant in April 2022, causing the second epidemiological surge (<xref ref-type="bibr" rid="B15">15</xref>). We only use data after April 2022 to stratified Omicron variants. The Taipei cohort, comprising 2,156 patients, was employed for the model development. The Hsinchu and Yunlin cohorts were utilized in federated learning.</p>
</sec>
<sec id="s2e"><label>2.5</label><title>Machine learning methods</title>
<p>The dataset was divided into training and validation subsets, with 70&#x0025; allocated for training and 30&#x0025; for validation. To address the issue of class imbalance in case outcomes, the Synthetic Minority Oversampling Technique (SMOTE) was applied, enabling the oversampling of minority-class patients within the training dataset. Four machine learning models, including Lasso (<xref ref-type="bibr" rid="B16">16</xref>), Random Forest (<xref ref-type="bibr" rid="B17">17</xref>), TabNet (<xref ref-type="bibr" rid="B18">18</xref>), and Gradient-Boosted Tree (XGBoost) (<xref ref-type="bibr" rid="B19">19</xref>), were utilized. The models were implemented using the following Python libraries: <italic>sklearn</italic> (LogisticRegression and RandomForestClassifier), <italic>pytorch-tabnet</italic>, and <italic>xgboost</italic> (version 2.3.1).</p>
<p>Random Forest and XgBoost are ensemble learning methods that combine multiple weak decision trees to generate a strong decision tree. Random Forest uses the &#x201C;bagging&#x201D; method, where each tree is trained on a bootstrap sample of the data, and their predictions are aggregated by voting in classification tasks or averaging in regression tasks. The trees are trained independently and do not update each other, making the method relatively fast and robust against overfitting (<xref ref-type="bibr" rid="B17">17</xref>). In contrast, XGBoost is a gradient boosting method which builds trees sequentially, where each new tree focuses on correcting the residual errors of the previous trees using gradient information. The final prediction is obtained by summing the weighted outputs of all trees. This sequential learning allows for effective error correction and high accuracy but may increase the risk of overfitting if not properly regularized (<xref ref-type="bibr" rid="B19">19</xref>).</p>
<p>Lasso regression is a type of linear regression that adds a penalty term to reduce the size of the model&#x0027;s coefficients, helping to prevent overfitting. Although the model remains linear, the added penalty can shrink some coefficients exactly to zero, removing less useful features from the model. This results in a simpler, more interpretable model that often performs better on data with unrelated features (<xref ref-type="bibr" rid="B16">16</xref>).</p>
<p>Finally, TabNet is a method designed for tabular data. Its advantage lies in its ability to provide feature-level interpretability through attention masks. However, its performance can be sensitive to hyperparameter tuning, and training the model can be computationally demanding (<xref ref-type="bibr" rid="B18">18</xref>).</p>
<p>Shapley Additive Explanations (SHAP) were employed to enhance the interpretability of these models, allowing the quantification of each predictor&#x0027;s contribution to the model&#x0027;s predictions (<xref ref-type="bibr" rid="B20">20</xref>). Additionally, an unsupervised clustering approach based on Euclidean distance was used to group patients with similar SHAP profiles, aiding in the identification of phenotypic patterns within the study cohort.</p>
</sec>
<sec id="s2f"><label>2.6</label><title>Federated learning</title>
<p>Federated Learning (FL) was implemented to integrate the most effective model from the Taipei cohort with additional data from Hsinchu and Yunlin hospitals. FL is a decentralized and collaborative approach designed to address challenges related to data silos and sensitivity (<xref ref-type="bibr" rid="B21">21</xref>). Since the datasets from these hospitals shared identical features but varied in sample composition, horizontal alliance learning was employed. The process began with each participating hospital receiving the same initial model and parameters. Each hospital independently trained the model on its local data, computed gradient updates, and securely transmitted these updates to a central server. The server aggregated the encrypted gradients, updated the global model, and redistributed the improved model parameters back to the hospitals. This iterative process allowed the final model to generalize across datasets while preserving data privacy, enabling its application to a broader population.</p>
</sec>
<sec id="s2g"><label>2.7</label><title>Performance evaluation</title>
<p>The study aimed to predict 30-day mortality by employing a comprehensive set of evaluation metrics to assess model performance. The metrics included measures to capture predictive accuracy, precision, and reliability. Additionally, distinctions in vital pathogenic factors were evaluated across subgroups defined by gender, age, and body mass index (BMI). Gender was categorized into male and female groups, while age was segmented into elderly (&#x2265;65 years), middle-aged (40&#x2013;64 years), and young (18&#x2013;39 years) groups. BMI was classified into obese (&#x2265;30), overweight (25&#x2013;29), and normal (&#x003C;25) categories. These subgroup analyses provided insights into model performance variations across diverse patient demographics and clinical characteristics. All computational analyses were conducted using Python 3.10.8.</p>
</sec>
</sec>
<sec id="s3" sec-type="results"><label>3</label><title>Results</title>
<sec id="s3a"><label>3.1</label><title>Participants</title>
<p>The study included 4,081 hospitalized adult patients diagnosed with SARS-CoV-2 Omicron variant infections between April 2022 and October 2022, with the diagnosis confirmed by real-time polymerase chain reaction (RT-PCR). The initial registry consisted of 6,321 patients hospitalized for COVID-19 across three branches of NTUH&#x2014;Taipei, Hsinchu, and Yunlin&#x2014;from January 2021 to October 2022. <xref ref-type="fig" rid="F1">Figure&#x00A0;1</xref> illustrates the patient selection flowchart. <xref ref-type="table" rid="T1">Table&#x00A0;1</xref> summarizes the demographic information, vital signs upon admission, and underlying comorbidities stratified by primary outcome. Among the final cohort, 2,015 patients survived, while 141 patients succumbed to the disease. Non-survivors were significantly older than survivors, with a median age of 78 years compared to 69 years (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.001). Apart from temperature, vital signs differed significantly between survivors and non-survivors (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.001). Chronic kidney disease and metastatic cancer were significantly associated with mortality (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.001).</p>
<fig id="F1" position="float"><label>Figure 1</label>
<caption><p>Flowchart depicting construction of study cohort and external validation cohort from national Taiwan university hospital cohort.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fmedt-07-1621158-g001.tif"><alt-text content-type="machine-generated">Flowchart depicting the selection and analysis of patients hospitalized at National Taiwan University Hospital from January 2021 to October 2022. Out of 6,321 patients, 2,240 are excluded due to age and Delta cohort criteria, resulting in 4,081 COVID-19 patients. The Taipei cohort with 2,156 patients undergoes training and validation using Random Forest, Xgboost, LASSO, and TabNet. The optimal choice is Xgboost with vital signs data. Federated learning combines three cohorts into a final model. The Hsinchu and Yunlin cohort with 1,925 patients serves as the testing set.</alt-text>
</graphic>
</fig>
<table-wrap id="T1" position="float"><label>Table 1</label>
<caption><p>Characteristics of COVID patients in the Taipei cohort, stratified by 30-day survival status.</p></caption>
<table frame="hsides" rules="groups">
<colgroup>
<col align="left"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th valign="top" align="left">Characteristic</th>
<th valign="top" align="center">COVID-19 survivors (<italic>n</italic>&#x2009;&#x003D;&#x2009;2,015)</th>
<th valign="top" align="center">COVID-19 non-survivors (<italic>n</italic>&#x2009;&#x003D;&#x2009;141)</th>
<th valign="top" align="center">Total (<italic>n</italic>&#x2009;&#x003D;&#x2009;2,156)</th>
<th valign="top" align="center"><italic>P</italic>-value</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Age, mean (SD)</td>
<td valign="top" align="center">69 (19)</td>
<td valign="top" align="center">78 (16)</td>
<td valign="top" align="center">70 (19)</td>
<td valign="top" align="center">&#x003C; 0.001</td>
</tr>
<tr>
<td valign="top" align="left">Male gender, <italic>n</italic> (&#x0025;)</td>
<td valign="top" align="center">1,050 (52.1&#x0025;)</td>
<td valign="top" align="center">75 (53.2&#x0025;)</td>
<td valign="top" align="center">1,125 (52.2&#x0025;)</td>
<td valign="top" align="center">0.872</td>
</tr>
<tr>
<td valign="top" align="left">BMI, mean (SD)</td>
<td valign="top" align="center">22.81 (4.77)</td>
<td valign="top" align="center">22.83 (4.09)</td>
<td valign="top" align="center">22.81 (4.72)</td>
<td valign="top" align="center">0.967</td>
</tr>
<tr>
<td valign="top" align="left">BMI &#x003E;30, <italic>n</italic> (&#x0025;)</td>
<td valign="top" align="center">121 (6&#x0025;)</td>
<td valign="top" align="center">6 (4.3&#x0025;)</td>
<td valign="top" align="center">127 (5.9&#x0025;)</td>
<td valign="top" align="center">0.504</td>
</tr>
<tr>
<td valign="top" align="left" colspan="5">Vital sign</td>
</tr>
<tr>
<td valign="top" align="left">Temperature (SD)</td>
<td valign="top" align="center">36.7 (0.66)</td>
<td valign="top" align="center">36.5 (0.61)</td>
<td valign="top" align="center">36.7 (0.66)</td>
<td valign="top" align="center">0.001</td>
</tr>
<tr>
<td valign="top" align="left">Breath rate (SD)</td>
<td valign="top" align="center">19 (3.19)</td>
<td valign="top" align="center">21 (4.92)</td>
<td valign="top" align="center">19 (3.35)</td>
<td valign="top" align="center">&#x003C; 0.001</td>
</tr>
<tr>
<td valign="top" align="left">Pulse rate (SD)</td>
<td valign="top" align="center">88 (16.72)</td>
<td valign="top" align="center">94 (20.53)</td>
<td valign="top" align="center">88.24 (17.05)</td>
<td valign="top" align="center">&#x003C; 0.001</td>
</tr>
<tr>
<td valign="top" align="left">SBP (SD)</td>
<td valign="top" align="center">136 (22.56)</td>
<td valign="top" align="center">128 (29.21)</td>
<td valign="top" align="center">135 (23.03)</td>
<td valign="top" align="center">&#x003C; 0.001</td>
</tr>
<tr>
<td valign="top" align="left">DBP (SD)</td>
<td valign="top" align="center">77 (12.59)</td>
<td valign="top" align="center">73 (16.62)</td>
<td valign="top" align="center">76 (12.86)</td>
<td valign="top" align="center">&#x003C; 0.001</td>
</tr>
<tr>
<td valign="top" align="left">SpO2</td>
<td valign="top" align="center">97 (1.97)</td>
<td valign="top" align="center">96 (4.34)</td>
<td valign="top" align="center">97 (2.24)</td>
<td valign="top" align="center">&#x003C; 0.001</td>
</tr>
<tr>
<td valign="top" align="left" colspan="5">Underlying comorbidity</td>
</tr>
<tr>
<td valign="top" align="left">Hypertension (&#x0025;)</td>
<td valign="top" align="center">1,115 (55.3&#x0025;)</td>
<td valign="top" align="center">96 (68.1&#x0025;)</td>
<td valign="top" align="center">1,211 (56&#x0025;)</td>
<td valign="top" align="center">0.004</td>
</tr>
<tr>
<td valign="top" align="left">Chronic lung disease (&#x0025;)</td>
<td valign="top" align="center">223 (11.1&#x0025;)</td>
<td valign="top" align="center">9 (6.4&#x0025;)</td>
<td valign="top" align="center">32 (10.8&#x0025;)</td>
<td valign="top" align="center">0.111</td>
</tr>
<tr>
<td valign="top" align="left">Chronic kidney disease (&#x0025;)</td>
<td valign="top" align="center">526 (26.1&#x0025;)</td>
<td valign="top" align="center">63 (44.7&#x0025;)</td>
<td valign="top" align="center">589 (27.3&#x0025;)</td>
<td valign="top" align="center">&#x003C; 0.001</td>
</tr>
<tr>
<td valign="top" align="left">Cancer without metastasis (&#x0025;)</td>
<td valign="top" align="center">559 (27.7&#x0025;)</td>
<td valign="top" align="center">57 (40.4&#x0025;)</td>
<td valign="top" align="center">616 (28.6&#x0025;)</td>
<td valign="top" align="center">0.002</td>
</tr>
<tr>
<td valign="top" align="left">Metastatic cancer (&#x0025;)</td>
<td valign="top" align="center">193 (9.6&#x0025;)</td>
<td valign="top" align="center">31 (22&#x0025;)</td>
<td valign="top" align="center">224 (10.4&#x0025;)</td>
<td valign="top" align="center">&#x003C; 0.001</td>
</tr>
<tr>
<td valign="top" align="left">Diabetes mellitus (&#x0025;)</td>
<td valign="top" align="center">624 (31&#x0025;)</td>
<td valign="top" align="center">52 (36.9&#x0025;)</td>
<td valign="top" align="center">676 (31.4&#x0025;)</td>
<td valign="top" align="center">0.171</td>
</tr>
<tr>
<td valign="top" align="left">Peptic ulcer disease (&#x0025;)</td>
<td valign="top" align="center">173 (8.6&#x0025;)</td>
<td valign="top" align="center">16 (11.3&#x0025;)</td>
<td valign="top" align="center">189 (8.8&#x0025;)</td>
<td valign="top" align="center">0.333</td>
</tr>
<tr>
<td valign="top" align="left">Prior use of steroid (&#x0025;)</td>
<td valign="top" align="center">436 (21.6&#x0025;)</td>
<td valign="top" align="center">37 (26.2&#x0025;)</td>
<td valign="top" align="center">473 (21.9&#x0025;)</td>
<td valign="top" align="center">0.241</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="table-fn1"><p>CNS, central nervous system; SE, standard error.</p></fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3b"><label>3.2</label><title>Mortality prediction in machine learning models</title>
<p>The Taipei cohort was used to train and validate a machine learning model, while the Hsinchu and Yunlin branches implemented federated learning to evaluate generalizability. Demographic information, vital signs, and comorbidities were used to predict 30-day mortality. <xref ref-type="table" rid="T2">Table&#x00A0;2</xref> compares the performance of machine learning algorithms, with Xgboost achieving the highest area under the curve (AUC), area under the precision-recall curve (AUPRC), sensitivity, and negative predictive value (NPV) when specificity was set at 0.80. The DeLong test is a nonparametric statistical method used to compare AUCs of models that are tested on the same dataset, and it provides a <italic>p</italic>-value to determine whether the difference between the two AUCs is statistically significant (<xref ref-type="bibr" rid="B22">22</xref>). <xref ref-type="table" rid="T3">Table&#x00A0;3</xref> uses DeLong test to compare Xgboost with other algorithms. <italic>P</italic>-value&#x2009;&#x003C;&#x2009;0.05 means that Xgboost and that algorithms has significant difference, which also indicates that Xgboost performs better.</p>
<table-wrap id="T2" position="float"><label>Table 2</label>
<caption><p>Measures of model discrimination and accuracy in the validation dataset (NIS 2014)sss, including area under the curve (AUC) and its 95&#x0025; confidence intervals, sensitivity, specificity, positive predictive value (PPV), and negative predictive value (NPV).</p></caption>
<table frame="hsides" rules="groups">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th valign="top" align="left">Model</th>
<th valign="top" align="center">AUC</th>
<th valign="top" align="center">AUPRC</th>
<th valign="top" align="center">Sensitivity</th>
<th valign="top" align="center">Specificity</th>
<th valign="top" align="center">PPV</th>
<th valign="top" align="center">NPV</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Lasso</td>
<td valign="top" align="left">0.84 (95&#x0025; CI 0.829&#x2013;0.860)</td>
<td valign="top" align="center">0.70</td>
<td valign="top" align="center">0.71</td>
<td valign="top" align="center">0.80</td>
<td valign="top" align="center">0.52</td>
<td valign="top" align="center">0.91</td>
</tr>
<tr>
<td valign="top" align="left">Random forest</td>
<td valign="top" align="left">0.92 (95&#x0025; CI 0.919&#x2013;0.928)</td>
<td valign="top" align="center">0.83</td>
<td valign="top" align="center">0.81</td>
<td valign="top" align="center">0.80</td>
<td valign="top" align="center">0.67</td>
<td valign="top" align="center">0.94</td>
</tr>
<tr>
<td valign="top" align="left">Xgboost</td>
<td valign="top" align="left">0.96 (95&#x0025; CI 0.967&#x2013;0.974)</td>
<td valign="top" align="center">0.83</td>
<td valign="top" align="center">0.94</td>
<td valign="top" align="center">0.80</td>
<td valign="top" align="center">0.57</td>
<td valign="top" align="center">0.97</td>
</tr>
<tr>
<td valign="top" align="left">TabNet</td>
<td valign="top" align="left">0.78 (95&#x0025; CI 0.752&#x2013;0.804</td>
<td valign="top" align="center">0.68</td>
<td valign="top" align="center">0.60</td>
<td valign="top" align="center">0.80</td>
<td valign="top" align="center">0.46</td>
<td valign="top" align="center">0.88</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T3" position="float"><label>Table 3</label>
<caption><p>Delong test results comparing the AUC of XGBoost against Lasso, Random Forest, and TabNet models.</p></caption>
<table frame="hsides" rules="groups">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th valign="top" align="left">Model 1</th>
<th valign="top" align="center">Model 2</th>
<th valign="top" align="center">&#x0394;AUC</th>
<th valign="top" align="center"><italic>p</italic>-value</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">XGBoost</td>
<td valign="top" align="left">Lasso</td>
<td valign="top" align="center">0.12</td>
<td valign="top" align="center">&#x003C; 0.001</td>
</tr>
<tr>
<td valign="top" align="left">XGBoost</td>
<td valign="top" align="left">Random Forest</td>
<td valign="top" align="center">0.04</td>
<td valign="top" align="center">&#x003C; 0.001</td>
</tr>
<tr>
<td valign="top" align="left">XGBoost</td>
<td valign="top" align="left">TabNet</td>
<td valign="top" align="center">0.18</td>
<td valign="top" align="center">&#x003C; 0.001</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="table-fn2"><p>&#x0394;AUC represents the difference in AUC (XGBoost&#x2212;other model), and <italic>p</italic>-values indicate the statistical significance of the difference.</p></fn>
</table-wrap-foot>
</table-wrap>
<p>The inclusion of vital signs enhanced the Xgboost model&#x0027;s performance compared to using only demographic information and chronic illnesses. <xref ref-type="table" rid="T4">Table&#x00A0;4</xref> shows that incorporating vital signs improved all performance metrics, including AUC, sensitivity, and NPV. The results highlight the importance of vital signs in achieving superior predictive accuracy. In conclusion, the Xgboost model was the most effective prediction tool for 30-day mortality, offering a robust balance between sensitivity and specificity while maintaining a conservative approach.</p>
<table-wrap id="T4" position="float"><label>Table 4</label>
<caption><p>Xgboost model using different sets of features.</p></caption>
<table frame="hsides" rules="groups">
<colgroup>
<col align="left"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th valign="top" align="left">Model</th>
<th valign="top" align="center">AUC</th>
<th valign="top" align="center">AUPRC</th>
<th valign="top" align="center">Sensitivity</th>
<th valign="top" align="center">Specificity</th>
<th valign="top" align="center">PPV</th>
<th valign="top" align="center">NPV</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Vital signs and chronic illness</td>
<td valign="top" align="center">0.96</td>
<td valign="top" align="center">0.83</td>
<td valign="top" align="center">0.94</td>
<td valign="top" align="center">0.80</td>
<td valign="top" align="center">0.57</td>
<td valign="top" align="center">0.97</td>
</tr>
<tr>
<td valign="top" align="left">Chronic illness alone</td>
<td valign="top" align="center">0.93</td>
<td valign="top" align="center">0.72</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.80</td>
<td valign="top" align="center">0.54</td>
<td valign="top" align="center">0.96</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3c"><label>3.3</label><title>Feature importance and SHAP analysis</title>
<p>To provide a visual representation of the selected features, the SHAP approach was employed to illustrate their impact on 30-day mortality in the Xgboost model (<xref ref-type="bibr" rid="B23">23</xref>). The mean SHAP value plot (<xref ref-type="fig" rid="F2">Figure&#x00A0;2</xref>) ranks the top 20 features with the highest average absolute SHAP values, while the bee swarm plot (<xref ref-type="fig" rid="F3">Figure&#x00A0;3</xref>) presents the individual contributions of these features, offering insights into their stability and interpretation. In both figures, feature rankings indicate their importance to the predictive model, while SHAP values provide a unified measure of the influence of specific features. In the bee swarm plot, red dots represent high feature values, while blue dots indicate low values, enabling visualization of how each feature affects predictions.</p>
<fig id="F2" position="float"><label>Figure 2</label>
<caption><p>Variables of importance (20 most important variables) from xgboost, ranked by mean SHAP values. <bold>(A)</bold> Model using both vital and chronic illness. <bold>(B)</bold> Model using only chronic illness.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fmedt-07-1621158-g002.tif"><alt-text content-type="machine-generated">Bar charts titled \"A\" and \"B\" show the importance of various health factors using SHAP values. Chart \"A\" ranks age, BMI, and sex as top factors. Chart \"B\" also ranks age highest, followed by systolic blood pressure (SBP) and SpO2. Both charts list multiple health conditions with descending importance from top to bottom.</alt-text>
</graphic>
</fig>
<fig id="F3" position="float"><label>Figure 3</label>
<caption><p>Bee swarm plot valuing feature impact on predictions, where red and blue dots represent high and low feature values, respectively. Overlapping dots indicate instability without vital signs, highlighting their importance for stable and accurate predictions. <bold>(A)</bold> Model using both vital and chronic illness. <bold>(B)</bold> Model using only chronic illness.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fmedt-07-1621158-g003.tif"><alt-text content-type="machine-generated">Two SHAP value plots (A and B) display the impact of various features on model output. Both plots show features like age, BMI, and medical conditions, indicating their positive or negative contributions to the model. Red dots represent high feature values, while blue dots represent low values, with a color gradient in between. Plot A focuses on age, BMI, and common medical conditions, whereas Plot B additionally includes vital signs like blood pressure and pulse rate, emphasizing their influence by the SHAP value distribution.</alt-text>
</graphic>
</fig>
<p>Certain features, including age, pulse rate, and body mass index (BMI), are associated with an increased risk of mortality when their values are elevated. Conversely, lower values of systolic blood pressure (SBP), oxygen saturation (SpO&#x2082;), temperature, diastolic blood pressure (DBP), and respiratory rate are linked to higher mortality risk. Chronic illnesses such as diabetes mellitus without complications, cerebrovascular disease, and chronic lung disease are also significant predictors, but some of these features, surprisingly, are associated with a lower likelihood of mortality. This trend highlights different correlations between features(demographic information, vital signs, comorbidities) and mortality rate.</p>
<p>The mean SHAP value plot (<xref ref-type="fig" rid="F2">Figure&#x00A0;2</xref>) confirms the importance of both vital signs and chronic illnesses, with features such as age, BMI, and sex emerging as the most critical contributors. In the absence of vital signs, the bee swarm plot (<xref ref-type="fig" rid="F3">Figure&#x00A0;3B</xref>) shows overlapping red and blue dots, indicating instability, as the same feature value can exert different effects on prediction (<xref ref-type="bibr" rid="B23">23</xref>, <xref ref-type="bibr" rid="B24">24</xref>). Without vital signs, the SHAP values of chronic illnesses, including prior steroid use, cerebrovascular disease, and chronic kidney disease, dominate the model. However, the decline in SHAP values for most features suggests that the model is less robust and lacks sufficient influential factors (<xref ref-type="bibr" rid="B25">25</xref>, <xref ref-type="bibr" rid="B26">26</xref>).</p>
<p>The inclusion of vital signs in the model improves both stability and predictive accuracy, with six vital signs ranking among the top ten features. The absence of vital signs reduces model stability and precision, particularly for certain subgroups. For instance, stratified analysis (<xref ref-type="table" rid="T5">Table&#x00A0;5</xref>) reveals that predictions for individuals with a BMI of 30 or higher are less precise. Consequently, BMI&#x2009;&#x2265;&#x2009;30 is considered outside the scope of the model. These results underscore the critical role of vital signs in achieving reliable and stable predictions, while also identifying the chronic illnesses that have the most significant impact when vital signs are excluded.</p>
<table-wrap id="T5" position="float"><label>Table 5</label>
<caption><p>The area under the curve (AUC) in different subgroups gender (male), Age (elderly &#x003E;&#x003D; 65 y/o., middle age 40&#x2013;64 y/o, young 18&#x2013;39 y/o), BMI (obese &#x003E;&#x003D;30, overweight 25&#x2013;29, normal &#x003C;25).</p></caption>
<table frame="hsides" rules="groups">
<colgroup>
<col align="left"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th valign="top" align="left">Groupings</th>
<th valign="top" align="center">Subgroups</th>
<th valign="top" align="center">AUC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">All patients</td>
<td valign="top" align="center">All patients</td>
<td valign="top" align="center">0.96</td>
</tr>
<tr>
<td valign="top" align="left">Sex</td>
<td valign="top" align="center">Male</td>
<td valign="top" align="center">0.98</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="center">Female</td>
<td valign="top" align="center">0.97</td>
</tr>
<tr>
<td valign="top" align="left">Age</td>
<td valign="top" align="center">Age &#x2265;65</td>
<td valign="top" align="center">0.97</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="center">65&#x003E; Age &#x2265;40</td>
<td valign="top" align="center">1.00</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="center">39 &#x003E; Age &#x2265;18</td>
<td valign="top" align="center">1.00</td>
</tr>
<tr>
<td valign="top" align="left">BMI</td>
<td valign="top" align="center">BMI &#x2265;30</td>
<td valign="top" align="center">0.87</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="center">30 &#x003E; BMI &#x2265;25</td>
<td valign="top" align="center">0.97</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="center">BMI &#x003C;25</td>
<td valign="top" align="center">0.98</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3d"><label>3.4</label><title>Model performance and federated learning</title>
<p>Federated learning is a method where individuals, such as hospitals, can train their data on a local end and upload the parameters to the head server. This can prevent data leakage and transmission of large quantity data. We use Federated learning to combine training results from three hospital branches- Taipei, Hsinchu and Yunlin.</p>
<p>The Xgboost model was initially evaluated on the Taipei cohort, achieving high predictive performance with an area under the curve (AUC) of 0.96 (<xref ref-type="fig" rid="F4">Figure&#x00A0;4</xref> and <xref ref-type="table" rid="T6">Table&#x00A0;6</xref>). However, when the pre-federated learning model was applied to other cohorts, its performance declined. After implementing federated learning, the AUC of the Taipei cohort decreased to 0.90, while the performance of other cohorts improved to meet the required standards. The decrease in Taipei cohort result from combining training results from three cohort making the model less specific to individual cohort. Whereas the increase in Hsinchu and Yunlin cohorts is because after federated learning the model has more understanding of Hsinchu and Yunlin cohorts. These results (<xref ref-type="fig" rid="F4">Figure&#x00A0;4</xref> and <xref ref-type="table" rid="T6">Table&#x00A0;6</xref>) indicate that federated learning helps to enhance the generalizability of the model across diverse cohorts, while also protecting patient privacy by ensuring data is not centralized.</p>
<fig id="F4" position="float"><label>Figure 4</label>
<caption><p>Comparison of AUC of local and federated model in three hospitals, with xgboost model using both vital sign and chronic illness. Bold line represents AUC before FL and transparent line represents AUC after FL. Taipei, Hsinchu and Yunlin are in red, blue and orange respectively.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fmedt-07-1621158-g004.tif"><alt-text content-type="machine-generated">Line graph showing AUC versus epochs for different models. The AUC value rapidly increases from zero to about 0.9 within the first five epochs and then gradually levels off, maintaining a steady performance up to the 50th epoch. The bold line represents AUC before FL and the transparent line represents AUC after FL. The plot displays three colored lines, which Taipei, Hsinchu and Yunlin are in red, blue and orange respectively.</alt-text>
</graphic>
</fig>
<table-wrap id="T6" position="float"><label>Table 6</label>
<caption><p>Comparison of AUC before and after the application of federated learning, with xgboost model using both vital sign and chronic illness.</p></caption>
<table frame="hsides" rules="groups">
<colgroup>
<col align="left"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th valign="top" align="left">Models</th>
<th valign="top" align="center">Taipei</th>
<th valign="top" align="center">Hsinchu</th>
<th valign="top" align="center">Yunlin</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">AUC before FL</td>
<td valign="top" align="center">0.96</td>
<td valign="top" align="center">0.82</td>
<td valign="top" align="center">0.81</td>
</tr>
<tr>
<td valign="top" align="left">AUC of FL</td>
<td valign="top" align="center">0.90</td>
<td valign="top" align="center">0.93</td>
<td valign="top" align="center">0.90</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The inclusion of vital signs in conjunction with chronic illnesses improved overall model performance compared to models relying solely on chronic illness data. This demonstrates the significance of integrating diverse data types to achieve robust predictive capabilities in different patient populations.</p>
</sec>
</sec>
<sec id="s4" sec-type="discussion"><label>4</label><title>Discussion</title>
<p>Our findings demonstrate that XGBoost outperformed other algorithms tested in this study for predicting 30-day mortality among hospitalized COVID-19 patients. The inclusion of vital signs significantly enhanced model performance, achieving an AUC of 0.96 (<xref ref-type="table" rid="T3">Table&#x00A0;3</xref>). While the results confirmed the importance of dynamic predictors such as pulse rate, oxygen saturation, blood pressure, and respiratory rate, further discussion is warranted to explore broader implications and contextualize these findings within existing clinical and research frameworks.</p>
<p>Machine learning has increasingly been recognized as a transformative tool in addressing complex interactions in high-dimensional clinical datasets. Traditional statistical methods, such as logistic regression and LASSO, often lack the capacity to capture nonlinear relationships and interdependencies, limiting their predictive accuracy (<xref ref-type="bibr" rid="B27">27</xref>, <xref ref-type="bibr" rid="B28">28</xref>). Ensemble learning methods, including random forest and gradient-boosted decision trees (GBDT), have demonstrated superior performance in such scenarios (<xref ref-type="bibr" rid="B29">29</xref>). XGBoost, as a leading GBDT algorithm, leverages an iterative approach to optimize predictions and effectively handle missing or sparse data (<xref ref-type="bibr" rid="B30">30</xref>). This capability, combined with its robustness in incorporating both static and dynamic clinical variables, underscores its relevance for healthcare applications, particularly during a global health crisis like COVID-19.</p>
<p>A key strength of this study lies in its use of SHAP analysis to enhance model interpretability. One of the significant barriers to adopting machine learning in healthcare is the &#x201C;black-box&#x201D; nature of many algorithms. By quantifying the contributions of individual features to model predictions, SHAP analysis provides a transparent framework that bridges the gap between advanced computational methods and clinical decision-making. Identifying predictors such as chronic kidney disease, diabetes without complications, and hypertension allows clinicians to better understand the rationale behind predictions. This transparency not only fosters trust but also facilitates the integration of machine learning tools into routine clinical workflows, ensuring that decisions are informed and actionable.</p>
<p>In the early days, clinical diagnosis was purely based on physician&#x0027;s experiences with x-ray images and check-up data, lacking precision and efficiency (<xref ref-type="bibr" rid="B8">8</xref>). With the rise of ML, more models have been produced to estimate mortality and severity. According to systematic reviews in recent years (<xref ref-type="bibr" rid="B11">11</xref>, <xref ref-type="bibr" rid="B12">12</xref>), the 4C mortality score was identified as the most promising model. The 4C Mortality Score is a risk stratification tool that predicts in-hospital mortality rate for hospitalized COVID-19 patients with eight parameters routinely available at hospital admission: age, sex, number of comorbidities, respiratory rate, peripheral oxygen saturation, level of consciousness, urea level, and C reactive protein (<xref ref-type="bibr" rid="B13">13</xref>). Its dataset was stratified from 260 hospitals across England, Scotland, and Wales. The 4C Score showed moderate diagnostic accuracy for mortality with derivation cohort area under the receiver operating characteristic curve (AUC) 0.79. However, it performs poorly on other cohorts, with AUC ranging 0.63&#x223C;0.73. Therefore, it cannot be implemented worldwide.</p>
<p>Previous clinical decision models, such as the 4C Mortality Score, have focused on common health data rather than comorbidities, lacking accuracy in assessing severity and mortality. Our research bridges this gap by adding 79 comorbidities to the prediction. We developed the Comorbidities and Clinical Indicators on the Mortality Model (CCIMM), a machine learning model that can accurately predict a patient&#x0027;s mortality rate within 30 days of hospitalization. We came up with two models: one comprising demographic information, vital signs upon admission, and underlying comorbidities; the other, comprising only demographic information and underlying comorbidities. The former one, with vital signs, has the highest AUC of 0.96. It can be used in the emergency room, where patients have unstable vital signs. The latter, without vital signs, has an AUC of 0.93. It can be used in outpatient conditions, where vital signs are rather stable and patients often have chronic illnesses. This model is also helpful at understanding the association between comorbidities and 30-day mortality rate. In addition, both models used federated learning to integrate data from three hospitals, resulting in models suitable for broader population needs.</p>
<p>This research has allowed us to gain insight into the impact of comorbidities and clinical indicators on mortality in COVID-19. This is in contrast to the 4C score, which assesses comorbidities not by the identities of the comorbidities, but by the number of comorbidities, in a categorical fashion separated into three groups- 0, 1 or &#x2265;2. We use machine learning to deal with large amounts of data, testing different algorithms in hopes of getting the best results. CCIMrM can speed up the diagnostic agenda by being an accurate indicator of whether the patient needs antiviral medication. It can prevent hospitals from running out of medical supplies or isolation wards in devastating situations, and serve as an indicator of disease severity for patients and families. The model can also be applied to future epidemics, providing physicians with a quick and easy prediction. Furthermore, the best performing model was trained through federated learning. Our study extends the current understanding of mortality prediction for hospitalized COVID-19 patients by developing an accurate and explainable federated learning model based on comorbidities and clinical indicators.</p>
<p>Interestingly, the findings challenge certain assumptions in current clinical guidelines, particularly in the context of prescribing Paxlovid, which is an oral antiviral medication prescribe to patients with mild to moderate COVID-19. It can decrease the rate of severe COVID-19 or mortality with adjusted hazard ratios of 0.54 (<xref ref-type="bibr" rid="B31">31</xref>). While existing guidelines in Taiwan recommend Paxlovid for patients with conditions like diabetes mellitus, cerebrovascular disease, and chronic lung disease (<xref ref-type="bibr" rid="B12">12</xref>, <xref ref-type="bibr" rid="B32">32</xref>), our study found these comorbidities to have a neutral or even protective association with 30-day mortality. This discrepancy underscores the need for continuous refinement of evidence-based guidelines using real-world data and advanced analytical techniques. For instance, while chronic kidney disease and BMI &#x2265;30 remain high-priority risk factors, conditions like dementia and rheumatic diseases may warrant reevaluation based on the observed data trends. Additionally, the inconclusive findings regarding cancer highlight the necessity of individualized treatment plans rather than a one-size-fits-all approach to antiviral therapy.</p>
<p>Federated learning emerged as a critical tool in this study, addressing one of the main challenges in collaborative medical research&#x2014;data privacy. By enabling the training of machine learning models across decentralized datasets, federated learning ensures that patient data remains localized while still contributing to a robust, generalizable model. This approach proved particularly effective in improving performance across external cohorts, even though it led to a slight decline in the AUC for the Taipei cohort. Such trade-offs underscore the potential of federated learning to enhance the scalability and applicability of AI solutions in diverse healthcare settings, where data-sharing restrictions often impede collaborative advancements.</p>
</sec>
<sec id="s5"><label>5</label><title>Limitations</title>
<p>Despite its promising outcomes, the study has limitations that merit further discussion. First, the model&#x0027;s reliance on readily available clinical features, while advantageous for implementation, excludes other potentially valuable data sources, such as imaging and laboratory results. For example, chest x-rays and CT scans are often pivotal in assessing COVID-19 severity, and their integration into future models could significantly enhance predictive accuracy. Multimodal models that combine clinical, imaging, and laboratory data, as well as nuances about the severity of disease, would likely provide a more comprehensive understanding of patient risk profiles.</p>
<p>Second, the relatively small dataset focused exclusively on the Taiwanese population limits the generalizability of the findings. While federated learning mitigated some of these limitations by improving model performance in external cohorts, the demographic and clinical characteristics of the original dataset may still introduce biases. Expanding the dataset to include diverse populations from different geographic regions and healthcare systems would provide a more representative basis for training and validation. Additionally, the observed protective effects of certain comorbidities, such as diabetes and chronic lung disease, may reflect unique population-level characteristics rather than universal trends. Further studies are needed to validate these findings and investigate their underlying mechanisms.</p>
</sec>
<sec id="s6"><label>6</label><title>Future directions</title>
<p>The potential of integrating time-series data into predictive models represents another avenue for future research. COVID-19 is characterized by rapid disease progression, making real-time monitoring of vital signs crucial for early intervention. Incorporating continuous data streams from wearable devices or bedside monitors could enable dynamic risk assessment, allowing clinicians to adjust treatment plans in real time. In addition, as COVID variants evolve and clinical practices change with time, a real-world implementation would require frequent updates to maintain satisfactory performance, especially in a federated setting. Such advancements would move machine learning from static prediction models to dynamic, context-sensitive tools that align closely with the realities of patient care.</p>
<p>Another critical area for exploration is the role of interpretability in enhancing clinician adoption of machine learning tools. While SHAP analysis offers valuable insights into feature importance, further efforts are needed to ensure that these explanations are presented in a manner that is intuitive and clinically relevant. For example, integrating visualizations of SHAP values into electronic health record systems could provide clinicians with real-time feedback on patient risk factors, enabling more informed decision-making. Additionally, user-centered design principles should guide the development of interfaces that present machine learning outputs in ways that align with clinicians&#x0027; workflows and information needs.</p>
<p>The discrepancies observed between this study&#x0027;s findings and existing clinical guidelines also raise broader questions about how evidence is generated and translated into practice. Machine learning models offer the advantage of being data-driven, allowing for the identification of patterns that may not align with traditional clinical assumptions. However, these insights must be contextualized within the broader framework of clinical expertise and patient care priorities. For instance, while certain comorbidities may show a protective association with mortality in the study population, this does not necessarily negate their significance in other aspects of disease management. Collaborative efforts between data scientists, clinicians, and policymakers are essential to ensure that machine learning insights are translated into guidelines that are both evidence-based and clinically meaningful.</p>
</sec>
<sec id="s7" sec-type="conclusions"><label>7</label><title>Conclusions</title>
<p>In conclusion, this study highlights the transformative potential of machine learning, particularly XGBoost, in predicting 30-day mortality among hospitalized COVID-19 patients. While the inclusion of vital signs significantly enhanced predictive accuracy, federated learning demonstrated its value in improving generalizability across diverse cohorts. The integration of SHAP analysis provided a transparent framework for understanding model predictions, fostering clinician trust and facilitating practical application. Future research should focus on addressing the limitations of current models by incorporating multimodal data, expanding population diversity, and enhancing real-time monitoring capabilities. By aligning advanced computational techniques with clinical expertise, machine learning has the potential to revolutionize risk stratification and treatment optimization in COVID-19 care, paving the way for more personalized and effective healthcare solutions.</p>
</sec>
</body>
<back>
<sec id="s8" sec-type="data-availability"><title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found below: The dataset cannot be disclosed to public.</p>
</sec>
<sec id="s9" sec-type="ethics-statement"><title>Ethics statement</title>
<p>The studies involving humans were approved by National Taiwan University Hospital Institutional Review Board. The studies were conducted in accordance with the local legislation and institutional requirements. Written informed consent for participation was not required from the participants or the participants&#x2019; legal guardians/next of kin in accordance with the national legislation and institutional requirements.</p>
</sec>
<sec id="s10" sec-type="author-contributions"><title>Author contributions</title>
<p>Y-CH: Conceptualization, Methodology, Writing &#x2013; review &#x0026; editing. SC: Formal analysis, Writing &#x2013; original draft. S-YT: Software, Validation, Visualization, Writing &#x2013; review &#x0026; editing. J-RH: Writing &#x2013; review &#x0026; editing. W-TH: Supervision, Writing &#x2013; review &#x0026; editing. C-CL: Conceptualization, Methodology, Writing &#x2013; review &#x0026; editing.</p>
</sec>
<sec id="s11" sec-type="funding-information"><title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This study was funded by National Science and Technology Council grant NSTC 113-2314-B-002-178 and NSTC 114-2314-B-002 -102, and the research grants of National Taiwan University Hospital Yunlin Branch NTUHYL. 112-AI005 and NTUHYL. 113-X023. No funding bodies had any role in study design, data collection and analysis, decision to publish, or preparation of the manuscript.</p>
</sec>
<sec id="s12" sec-type="COI-statement"><title>Conflict of interest</title>
<p>W-TH was employed by Clinical AI Consulting Group.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s13" sec-type="ai-statement"><title>Generative AI statement</title>
<p>The author(s) declare that Generative AI was used in the creation of this manuscript. Generative AI was use to revise the wordings of every sections in the manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issue please contact us.</p>
</sec>
<sec id="s15" sec-type="disclaimer"><title>Publisher&#x0027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s14" sec-type="supplementary-material"><title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fmedt.2025.1621158/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fmedt.2025.1621158/full&#x0023;supplementary-material</ext-link></p>
<supplementary-material id="SD1" content-type="local-data">
<media mimetype="application" mime-subtype="vnd.openxmlformats-officedocument.wordprocessingml.document" xlink:href="Datasheet1.docx"/></supplementary-material>
</sec>
<ref-list><title>References</title>
<ref id="B1"><label>1.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Maison</surname><given-names>DP</given-names></name><name><surname>Tasissa</surname><given-names>H</given-names></name><name><surname>Deitchman</surname><given-names>A</given-names></name><name><surname>Peluso</surname><given-names>MJ</given-names></name><name><surname>Deng</surname><given-names>Y</given-names></name><name><surname>Miller</surname><given-names>FD</given-names></name><etal/></person-group> <article-title>COVID-19 clinical presentation, management, and epidemiology: a concise compendium</article-title>. <source>Front Public Health</source>. (<year>2025</year>) <volume>13</volume>:<fpage>1498445</fpage>. <pub-id pub-id-type="doi">10.3389/fpubh.2025.1498445</pub-id><pub-id pub-id-type="pmid">39957982</pub-id></citation></ref>
<ref id="B2"><label>2.</label><citation citation-type="other"><collab>COVID-19 Map</collab>. <article-title>In: Johns Hopkins Coronavirus Resour. Cent</article-title>. <comment>Available online at:</comment> <ext-link ext-link-type="uri" xlink:href="https://coronavirus.jhu.edu/map.html">https://coronavirus.jhu.edu/map.html</ext-link> <comment>(Accessed March 14, 2025)</comment>.</citation></ref>
<ref id="B3"><label>3.</label><citation citation-type="other"><article-title>COVID-19 cases &#x007C; WHO COVID-19 dashboard. In: datadot</article-title>. <comment>Available online at:</comment> <ext-link ext-link-type="uri" xlink:href="https://data.who.int/dashboards/covid19/cases">https://data.who.int/dashboards/covid19/cases</ext-link> <comment>(Accessed March 14, 2025</comment></citation></ref>
<ref id="B4"><label>4.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wiersinga</surname><given-names>WJ</given-names></name><name><surname>Rhodes</surname><given-names>A</given-names></name><name><surname>Cheng</surname><given-names>AC</given-names></name><name><surname>Peacock</surname><given-names>SJ</given-names></name><name><surname>Prescott</surname><given-names>HC</given-names></name></person-group>. <article-title>Pathophysiology, transmission, diagnosis, and treatment of coronavirus disease 2019 (COVID-19): a review</article-title>. <source>JAMA</source>. (<year>2020</year>) <volume>324</volume>:<fpage>782</fpage>&#x2013;<lpage>93</lpage>. <pub-id pub-id-type="doi">10.1001/jama.2020.12839</pub-id><pub-id pub-id-type="pmid">32648899</pub-id></citation></ref>
<ref id="B5"><label>5.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dinnes</surname><given-names>J</given-names></name><name><surname>Sharma</surname><given-names>P</given-names></name><name><surname>Berhane</surname><given-names>S</given-names></name><name><surname>van Wyk</surname><given-names>SS</given-names></name><name><surname>Nyaaba</surname><given-names>N</given-names></name><name><surname>Domen</surname><given-names>J</given-names></name><etal/></person-group> <article-title>Rapid, point-of-care antigen tests for diagnosis of SARS-CoV-2 infection</article-title>. <source>Cochrane Database Syst Rev</source>. (<year>2022</year>) <volume>2022</volume>:<fpage>CD013705</fpage>. <pub-id pub-id-type="doi">10.1002/14651858.CD013705.pub3</pub-id></citation></ref>
<ref id="B6"><label>6.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Binny</surname><given-names>RN</given-names></name><name><surname>Priest</surname><given-names>P</given-names></name><name><surname>French</surname><given-names>NP</given-names></name><name><surname>Parry</surname><given-names>M</given-names></name><name><surname>Lustig</surname><given-names>A</given-names></name><name><surname>Hendy</surname><given-names>SC</given-names></name><etal/></person-group> <article-title>Sensitivity of reverse transcription polymerase chain reaction tests for severe acute respiratory syndrome coronavirus 2 through time</article-title>. <source>J Infect Dis</source>. (<year>2022</year>) <volume>227</volume>:<fpage>9</fpage>&#x2013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.1093/infdis/jiac317</pub-id><pub-id pub-id-type="pmid">35876500</pub-id></citation></ref>
<ref id="B7"><label>7.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname><given-names>L</given-names></name><name><surname>Song</surname><given-names>W</given-names></name><name><surname>Patil</surname><given-names>N</given-names></name><name><surname>Sainlaire</surname><given-names>M</given-names></name><name><surname>Jasuja</surname><given-names>R</given-names></name><name><surname>Dykes</surname><given-names>PC</given-names></name></person-group>. <article-title>Predicting COVID-19 severity: challenges in reproducibility and deployment of machine learning methods</article-title>. <source>Int J Med Inf</source>. (<year>2023</year>) <volume>179</volume>:<fpage>105210</fpage>. <pub-id pub-id-type="doi">10.1016/j.ijmedinf.2023.105210</pub-id></citation></ref>
<ref id="B8"><label>8.</label><citation citation-type="journal"><collab>WHO Working Group on the Clinical Characterisation and Management of COVID-19 infection</collab>. <article-title>A minimal common outcome measure set for COVID-19 clinical research</article-title>. <source>Lancet Infect Dis</source>. (<year>2020</year>) <volume>20</volume>:<fpage>e192</fpage>&#x2013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1016/S1473-3099(20)30483-7</pub-id><pub-id pub-id-type="pmid">32539990</pub-id></citation></ref>
<ref id="B9"><label>9.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Attaway</surname><given-names>AH</given-names></name><name><surname>Scheraga</surname><given-names>RG</given-names></name><name><surname>Bhimraj</surname><given-names>A</given-names></name><name><surname>Biehl</surname><given-names>M</given-names></name><name><surname>Hatipo&#x011F;lu</surname><given-names>U</given-names></name></person-group>. <article-title>Severe COVID-19 pneumonia: pathogenesis and clinical management</article-title>. <source>Br Med J</source>. (<year>2021</year>) <volume>372</volume>:<fpage>n436</fpage>. <pub-id pub-id-type="doi">10.1136/bmj.n436</pub-id></citation></ref>
<ref id="B10"><label>10.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Adab</surname><given-names>P</given-names></name><name><surname>Haroon</surname><given-names>S</given-names></name><name><surname>O&#x0027;Hara</surname><given-names>ME</given-names></name><name><surname>Jordan</surname><given-names>RE</given-names></name></person-group>. <article-title>Comorbidities and COVID-19</article-title>. <source>BMJ</source>. (<year>2022</year>) <volume>377</volume>:<fpage>o1431</fpage>. <pub-id pub-id-type="doi">10.1136/bmj.o1431</pub-id><pub-id pub-id-type="pmid">35705219</pub-id></citation></ref>
<ref id="B11"><label>11.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Appel</surname><given-names>KS</given-names></name><name><surname>Geisler</surname><given-names>R</given-names></name><name><surname>Maier</surname><given-names>D</given-names></name><name><surname>Miljukov</surname><given-names>O</given-names></name><name><surname>Hopff</surname><given-names>SM</given-names></name><name><surname>Vehreschild</surname><given-names>JJ</given-names></name></person-group>. <article-title>A systematic review of predictor composition, outcomes, risk of bias, and validation of COVID-19 prognostic scores</article-title>. <source>Clin Infect Dis</source>. (<year>2024</year>) <volume>78</volume>:<fpage>889</fpage>&#x2013;<lpage>99</lpage>. <pub-id pub-id-type="doi">10.1093/cid/ciad618</pub-id><pub-id pub-id-type="pmid">37879096</pub-id></citation></ref>
<ref id="B12"><label>12.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wynants</surname><given-names>L</given-names></name><name><surname>Van Calster</surname><given-names>B</given-names></name><name><surname>Collins</surname><given-names>GS</given-names></name><name><surname>Riley</surname><given-names>RD</given-names></name><name><surname>Heinze</surname><given-names>G</given-names></name><name><surname>Schuit</surname><given-names>E</given-names></name><etal/></person-group> <article-title>Prediction models for diagnosis and prognosis of COVID-19: systematic review and critical appraisal</article-title>. <source>Br Med J</source>. (<year>2020</year>) <volume>369</volume>:<fpage>m1328</fpage>. <pub-id pub-id-type="doi">10.1136/bmj.m1328</pub-id></citation></ref>
<ref id="B13"><label>13.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Knight</surname><given-names>SR</given-names></name><name><surname>Ho</surname><given-names>A</given-names></name><name><surname>Pius</surname><given-names>R</given-names></name><name><surname>Buchan</surname><given-names>I</given-names></name><name><surname>Carson</surname><given-names>G</given-names></name><name><surname>Drake</surname><given-names>TM</given-names></name><etal/></person-group> <article-title>Risk stratification of patients admitted to hospital with COVID-19 using the ISARIC WHO clinical characterisation protocol: development and validation of the 4C mortality score</article-title>. <source>Br Med J</source>. (<year>2020</year>) <volume>370</volume>:<fpage>m3339</fpage>. <pub-id pub-id-type="doi">10.1136/bmj.m3339</pub-id></citation></ref>
<ref id="B14"><label>14.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Markov</surname><given-names>PV</given-names></name><name><surname>Ghafari</surname><given-names>M</given-names></name><name><surname>Beer</surname><given-names>M</given-names></name><name><surname>Lythgoe</surname><given-names>K</given-names></name><name><surname>Simmonds</surname><given-names>P</given-names></name><name><surname>Stilianakis</surname><given-names>NI</given-names></name><etal/></person-group> <article-title>The evolution of SARS-CoV-2</article-title>. <source>Nat Rev Microbiol</source>. (<year>2023</year>) <volume>21</volume>:<fpage>361</fpage>&#x2013;<lpage>79</lpage>. <pub-id pub-id-type="doi">10.1038/s41579-023-00878-2</pub-id><pub-id pub-id-type="pmid">37020110</pub-id></citation></ref>
<ref id="B15"><label>15.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kung</surname><given-names>Y-A</given-names></name><name><surname>Chuang</surname><given-names>C-H</given-names></name><name><surname>Chen</surname><given-names>Y-C</given-names></name><name><surname>Yang</surname><given-names>H-P</given-names></name><name><surname>Li</surname><given-names>H-C</given-names></name><name><surname>Chen</surname><given-names>C-L</given-names></name><etal/></person-group> <article-title>Worldwide SARS-CoV-2 omicron variant infection: emerging sub-variants and future vaccination perspectives</article-title>. <source>J Formos Med Assoc</source>. (<year>2025</year>) <volume>124</volume>:<fpage>592</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1016/j.jfma.2024.08.021</pub-id><pub-id pub-id-type="pmid">39179492</pub-id></citation></ref>
<ref id="B16"><label>16.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tibshirani</surname><given-names>R</given-names></name></person-group>. <article-title>Regression shrinkage and selection via the Lasso</article-title>. <source>J R Stat Soc Ser B Methodol</source>. (<year>1996</year>) <volume>58</volume>:<fpage>267</fpage>&#x2013;<lpage>88</lpage>. <pub-id pub-id-type="doi">10.1111/j.2517-6161.1996.tb02080.x</pub-id></citation></ref>
<ref id="B17"><label>17.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Breiman</surname><given-names>L</given-names></name></person-group>. <article-title>Random forests</article-title>. <source>Mach Learn</source>. (<year>2001</year>) <volume>45</volume>:<fpage>5</fpage>&#x2013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1023/A:1010933404324</pub-id></citation></ref>
<ref id="B18"><label>18.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Arik</surname><given-names>SO</given-names></name><name><surname>Pfister</surname><given-names>T</given-names></name></person-group>. <article-title>Tabnet: attentive interpretable tabular learning</article-title>. <source>Proc AAAI Conf Artif Intell</source>. (<year>2020</year>) <volume>45</volume>(<issue>8</issue>):<fpage>6679</fpage>&#x2013;<lpage>87</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1908.07442</pub-id></citation></ref>
<ref id="B19"><label>19.</label><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>T</given-names></name><name><surname>Guestrin</surname><given-names>C</given-names></name></person-group>. <article-title>XGBoost: a scalable tree boosting system</article-title>. <conf-name>KDD &#x0027;16: Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</conf-name> (<year>2016</year>). p. <fpage>785</fpage>&#x2013;<lpage>94</lpage></citation></ref>
<ref id="B20"><label>20.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Lundberg</surname><given-names>S</given-names></name><name><surname>Lee</surname><given-names>S-I</given-names></name></person-group>. <article-title>A Unified Approach to Interpreting Model Predictions</article-title>. (<year>2017</year>) <pub-id pub-id-type="doi">10.48550/arXiv.1705.07874</pub-id></citation></ref>
<ref id="B21"><label>21.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Yang</surname><given-names>Q</given-names></name><name><surname>Liu</surname><given-names>Y</given-names></name><name><surname>Chen</surname><given-names>T</given-names></name><name><surname>Tong</surname><given-names>Y</given-names></name></person-group>. <article-title>Federated Machine Learning: Concept and Applications</article-title>. (<year>2019</year>) <pub-id pub-id-type="doi">10.48550/arXiv.1902.04885</pub-id></citation></ref>
<ref id="B22"><label>22.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>DeLong</surname><given-names>ER</given-names></name><name><surname>DeLong</surname><given-names>DM</given-names></name><name><surname>Clarke-Pearson</surname><given-names>DL</given-names></name></person-group>. <article-title>Comparing the areas under two or more correlated receiver operating characteristic curves: a nonparametric approach</article-title>. <source>Biometrics</source>. (<year>1988</year>) <volume>44</volume>:<fpage>837</fpage>&#x2013;<lpage>45</lpage>. <pub-id pub-id-type="doi">10.2307/2531595</pub-id><pub-id pub-id-type="pmid">3203132</pub-id></citation></ref>
<ref id="B23"><label>23.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Datta</surname><given-names>D</given-names></name><name><surname>Ray</surname><given-names>S</given-names></name><name><surname>Martinez</surname><given-names>L</given-names></name><name><surname>Newman</surname><given-names>D</given-names></name><name><surname>Dalmida</surname><given-names>SG</given-names></name><name><surname>Hashemi</surname><given-names>J</given-names></name><etal/></person-group> <article-title>Feature identification using interpretability machine learning predicting risk factors for disease severity of in-patients with COVID-19 in south Florida</article-title>. <source>Diagn Basel Switz</source>. (<year>2024</year>) <volume>14</volume>:<fpage>1866</fpage>. <pub-id pub-id-type="doi">10.3390/diagnostics14171866</pub-id></citation></ref>
<ref id="B24"><label>24.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname><given-names>C</given-names></name><name><surname>Li</surname><given-names>L</given-names></name><name><surname>Huang</surname><given-names>W</given-names></name><name><surname>Wu</surname><given-names>T</given-names></name><name><surname>Xu</surname><given-names>Q</given-names></name><name><surname>Liu</surname><given-names>J</given-names></name><etal/></person-group> <article-title>Interpretable machine learning for early prediction of prognosis in sepsis: a discovery and validation study</article-title>. <source>Infect Dis Ther</source>. (<year>2022</year>) <volume>11</volume>:<fpage>1117</fpage>&#x2013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1007/s40121-022-00628-6</pub-id><pub-id pub-id-type="pmid">35399146</pub-id></citation></ref>
<ref id="B25"><label>25.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chieregato</surname><given-names>M</given-names></name><name><surname>Frangiamore</surname><given-names>F</given-names></name><name><surname>Morassi</surname><given-names>M</given-names></name><name><surname>Baresi</surname><given-names>C</given-names></name><name><surname>Nici</surname><given-names>S</given-names></name><name><surname>Bassetti</surname><given-names>C</given-names></name><etal/></person-group> <article-title>A hybrid machine learning/deep learning COVID-19 severity predictive model from CT images and clinical data</article-title>. <source>Sci Rep</source>. (<year>2022</year>) <volume>12</volume>:<fpage>4329</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-022-07890-1</pub-id><pub-id pub-id-type="pmid">35288579</pub-id></citation></ref>
<ref id="B26"><label>26.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qiu</surname><given-names>W</given-names></name><name><surname>Chen</surname><given-names>H</given-names></name><name><surname>Dincer</surname><given-names>AB</given-names></name><name><surname>Lundberg</surname><given-names>S</given-names></name><name><surname>Kaeberlein</surname><given-names>M</given-names></name><name><surname>Lee</surname><given-names>S-I</given-names></name></person-group>. <article-title>Interpretable machine learning prediction of all-cause mortality</article-title>. <source>Commun Med</source>. (<year>2022</year>) <volume>2</volume>:<fpage>125</fpage>. <pub-id pub-id-type="doi">10.1038/s43856-022-00180-x</pub-id><pub-id pub-id-type="pmid">36204043</pub-id></citation></ref>
<ref id="B27"><label>27.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ranganathan</surname><given-names>P</given-names></name><name><surname>Pramesh</surname><given-names>CS</given-names></name><name><surname>Aggarwal</surname><given-names>R</given-names></name></person-group>. <article-title>Common pitfalls in statistical analysis: logistic regression</article-title>. <source>Perspect Clin Res</source>. (<year>2017</year>) <volume>8</volume>:<fpage>148</fpage>&#x2013;<lpage>51</lpage>. <pub-id pub-id-type="doi">10.4103/picr.PICR_87_17</pub-id><pub-id pub-id-type="pmid">28828311</pub-id></citation></ref>
<ref id="B28"><label>28.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Freijeiro-Gonz&#x00E1;lez</surname><given-names>L</given-names></name><name><surname>Febrero-Bande</surname><given-names>M</given-names></name><name><surname>Gonz&#x00E1;lez-Manteiga</surname><given-names>W</given-names></name></person-group>. <article-title>A critical review of LASSO and its derivatives for variable selection under dependence among covariates</article-title>. (<year>2020</year>). <pub-id pub-id-type="doi">10.48550/arXiv.2012.11470</pub-id></citation></ref>
<ref id="B29"><label>29.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mohammed</surname><given-names>A</given-names></name><name><surname>Kora</surname><given-names>R</given-names></name></person-group>. <article-title>A comprehensive review on ensemble deep learning: opportunities and challenges</article-title>. <source>J King Saud Univ - Comput Inf Sci</source>. (<year>2023</year>) <volume>35</volume>:<fpage>757</fpage>&#x2013;<lpage>74</lpage>. <pub-id pub-id-type="doi">10.1016/j.jksuci.2023.01.014</pub-id></citation></ref>
<ref id="B30"><label>30.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hong</surname><given-names>W</given-names></name><name><surname>Zhou</surname><given-names>X</given-names></name><name><surname>Jin</surname><given-names>S</given-names></name><name><surname>Lu</surname><given-names>Y</given-names></name><name><surname>Pan</surname><given-names>J</given-names></name><name><surname>Lin</surname><given-names>Q</given-names></name><etal/></person-group> <article-title>A comparison of XGBoost, random forest, and nomograph for the prediction of disease severity in patients with COVID-19 pneumonia: implications of cytokine and immune cell profile</article-title>. <source>Front Cell Infect Microbiol</source>. (<year>2022</year>) <volume>12</volume>:<fpage>819267</fpage>. <pub-id pub-id-type="doi">10.3389/fcimb.2022.819267</pub-id><pub-id pub-id-type="pmid">35493729</pub-id></citation></ref>
<ref id="B31"><label>31.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Najjar-Debbiny</surname><given-names>R</given-names></name><name><surname>Gronich</surname><given-names>N</given-names></name><name><surname>Weber</surname><given-names>G</given-names></name><name><surname>Khoury</surname><given-names>J</given-names></name><name><surname>Amar</surname><given-names>M</given-names></name><name><surname>Stein</surname><given-names>N</given-names></name><etal/></person-group> <article-title>Effectiveness of paxlovid in reducing severe coronavirus disease 2019 and mortality in high-risk patients</article-title>. <source>Clin Infect Dis</source>. (<year>2023</year>) <volume>76</volume>:<fpage>e342</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1093/cid/ciac443</pub-id><pub-id pub-id-type="pmid">35653428</pub-id></citation></ref>
<ref id="B32"><label>32.</label><citation citation-type="book"><person-group person-group-type="author"><name><surname>Ahr</surname><given-names>H</given-names></name></person-group>. <source>Guidelines for the Clinical Management of Novel Coronavirus SARS-CoV-2 Infection</source>. <edition>27th ed</edition>. <publisher-loc>Taipei</publisher-loc>: <publisher-name>Taiwan Centers for Disease Control (Taiwan CDC)</publisher-name> (<year>2024</year>). <comment>Available online at:</comment> <ext-link ext-link-type="uri" xlink:href="https://www.cdc.gov.tw/File/Get/nj5FzVB1d0lqoMfrZFBP4g">https://www.cdc.gov.tw/File/Get/nj5FzVB1d0lqoMfrZFBP4g</ext-link> <comment>(Accessed March 14, 2025)</comment>.</citation></ref></ref-list>
</back>
</article>