<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Digit. Health</journal-id>
<journal-title>Frontiers in Digital Health</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Digit. Health</abbrev-journal-title>
<issn pub-type="epub">2673-253X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fdgth.2022.848599</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Digital Health</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Sepsis Prediction for the General Ward Setting</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Yu</surname> <given-names>Sean C.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1556817/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Gupta</surname> <given-names>Aditi</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1648186/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Betthauser</surname> <given-names>Kevin D.</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Lyons</surname> <given-names>Patrick G.</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1651683/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Lai</surname> <given-names>Albert M.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Kollef</surname> <given-names>Marin H.</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/698037/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Payne</surname> <given-names>Philip R. O.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1450028/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Michelson</surname> <given-names>Andrew P.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Institute for Informatics, Washington University School of Medicine in St. Louis</institution>, <addr-line>St. Louis, MO</addr-line>, <country>United States</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Biomedical Engineering, Washington University in St. Louis</institution>, <addr-line>St. Louis, MO</addr-line>, <country>United States</country></aff>
<aff id="aff3"><sup>3</sup><institution>Department of Pharmacy, Barnes-Jewish Hospital</institution>, <addr-line>St. Louis, MO</addr-line>, <country>United States</country></aff>
<aff id="aff4"><sup>4</sup><institution>Division of Pulmonary and Critical Care, Washington University School of Medicine in St. Louis</institution>, <addr-line>St. Louis, MO</addr-line>, <country>United States</country></aff>
<aff id="aff5"><sup>5</sup><institution>Healthcare Innovation Lab, BJC HealthCare, and Washington University in St. Louis School of Medicine</institution>, <addr-line>St. Louis, MO</addr-line>, <country>United States</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Lina F. Soualmia, Universit&#x000E9; de Rouen, France</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Seth Russell, University of Colorado Anschutz Medical Campus, United States; Tyler John Loftus, University of Florida, United States</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Sean C. Yu <email>sean.yu&#x00040;wustl.edu</email></corresp>
<fn fn-type="other" id="fn001"><p>This article was submitted to Health Informatics, a section of the journal Frontiers in Digital Health</p></fn></author-notes>
<pub-date pub-type="epub">
<day>08</day>
<month>03</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>4</volume>
<elocation-id>848599</elocation-id>
<history>
<date date-type="received">
<day>04</day>
<month>01</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>28</day>
<month>01</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2022 Yu, Gupta, Betthauser, Lyons, Lai, Kollef, Payne and Michelson.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Yu, Gupta, Betthauser, Lyons, Lai, Kollef, Payne and Michelson</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license> </permissions>
<abstract>
<sec>
<title>Objective</title>
<p>To develop and evaluate a sepsis prediction model for the general ward setting and extend the evaluation through a novel pseudo-prospective trial design.</p></sec>
<sec>
<title>Design</title>
<p>Retrospective analysis of data extracted from electronic health records (EHR).</p></sec>
<sec>
<title>Setting</title>
<p>Single, tertiary-care academic medical center in St. Louis, MO, USA.</p></sec>
<sec>
<title>Patients</title>
<p>Adult, non-surgical inpatients admitted between January 1, 2012 and June 1, 2019.</p></sec>
<sec>
<title>Interventions</title>
<p>None.</p></sec>
<sec>
<title>Measurements and Main Results</title>
<p>Of the 70,034 included patient encounters, 3.1% were septic based on the Sepsis-3 criteria. Features were generated from the EHR data and were used to develop a machine learning model to predict sepsis 6-h ahead of onset. The best performing model had an Area Under the Receiver Operating Characteristic curve (AUROC or c-statistic) of 0.862 &#x000B1; 0.011 and Area Under the Precision-Recall Curve (AUPRC) of 0.294 &#x000B1; 0.021 compared to that of Logistic Regression (0.857 &#x000B1; 0.008 and 0.256 &#x000B1; 0.024) and NEWS 2 (0.699 &#x000B1; 0.012 and 0.092 &#x000B1; 0.009). In the pseudo-prospective trial, 388 (69.7%) septic patients were alerted on with a specificity of 81.4%. Within 24 h of crossing the alert threshold, 20.9% had a sepsis-related event occur.</p></sec>
<sec>
<title>Conclusions</title>
<p>A machine learning model capable of predicting sepsis in the general ward setting was developed using the EHR data. The pseudo-prospective trial provided a more realistic estimation of implemented performance and demonstrated a 29.1% Positive Predictive Value (PPV) for sepsis-related intervention or outcome within 48 h.</p></sec></abstract>
<kwd-group>
<kwd>sepsis</kwd>
<kwd>electronic health records</kwd>
<kwd>machine learning</kwd>
<kwd>prediction</kwd>
<kwd>general ward</kwd>
</kwd-group>
<counts>
<fig-count count="3"/>
<table-count count="2"/>
<equation-count count="0"/>
<ref-count count="30"/>
<page-count count="8"/>
<word-count count="5446"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>Introduction</title>
<p>Sepsis is defined as a life-threatening organ dysfunction caused by a dysregulated host response to infection (<xref ref-type="bibr" rid="B1">1</xref>). In 2017, sepsis was responsible for 5.8% of all hospital stays and $38.2 billion in hospital costs (<xref ref-type="bibr" rid="B2">2</xref>). Moreover, sepsis has a high mortality rate and was found to be implicated in about one in every three inpatient deaths (<xref ref-type="bibr" rid="B3">3</xref>).</p>
<p>Early and effective therapy is critical in the management of patients with sepsis, as prolonged recognition and delayed treatment increase mortality (<xref ref-type="bibr" rid="B4">4</xref>, <xref ref-type="bibr" rid="B5">5</xref>). As a result, there is an abundance of literature focusing on the early detection and prediction of sepsis through traditional or newly developed scoring systems such as the Systemic Inflammatory Response Syndrome (SIRS) score, National Early Warning Score (NEWS), or quick Sequential Organ Failure Assessment (qSOFA) score; or more recently through the use of machine learning models (<xref ref-type="bibr" rid="B6">6</xref>&#x02013;<xref ref-type="bibr" rid="B8">8</xref>). Most of these efforts focus on the Emergency Department (ED) or Intensive Care Unit (ICU) settings which are data-rich and have a higher prevalence of sepsis compared to the general ward setting (<xref ref-type="bibr" rid="B9">9</xref>&#x02013;<xref ref-type="bibr" rid="B11">11</xref>). However, patients who develop sepsis in the general ward setting have worse outcomes compared to those who develop sepsis in the ED or ICU (<xref ref-type="bibr" rid="B12">12</xref>). Because general ward patients are observed less closely than in the ED or ICU setting with fewer vital signs documented and laboratory tests performed, they represent a proportionally more vulnerable population that could benefit more from an augmented sepsis early warning system. Therefore, the objective of this study was to develop a machine learning model for predicting sepsis in the general ward setting, compare its performance to commonly used instruments for sepsis surveillance such as SIRS and NEWS, and extend the model evaluation using a novel simulated pseudo-prospective trial (<xref ref-type="bibr" rid="B13">13</xref>, <xref ref-type="bibr" rid="B14">14</xref>).</p>
</sec>
<sec sec-type="materials and methods" id="s2">
<title>Materials and Methods</title>
<sec>
<title>Study Design, Data Sources, and Population</title>
<p>The model was developed and validated using the Electronic Health Record (EHR) data from Barnes-Jewish Hospital / Washington University School of Medicine in St. Louis, a large, academic, tertiary-care academic medical center. All patients &#x02265;18 years of age that were admitted to the hospital between January 1, 2012 and June 1, 2019 were eligible for inclusion. Patients were excluded if they were admitted to the Psychiatry or Obstetrics services, due to highly variable rates of physiologic data collection. Encounters were excluded if there were no billing code, vital signs, laboratory, service, room, or medication data to indicate a complete patient stay. Encounters were also excluded if the total length of stay was below 12 h or exceeded 30 days. After the assignment of index time and prediction time, further exclusion criteria were applied based on that index time (<xref ref-type="supplementary-material" rid="SM1">Supplementary Methods 3</xref>). To focus on patients most likely to benefit from a risk prediction model, the following populations were excluded: patients who had cultures procured or received antibiotics within 48 h prior to prediction time; patients who had sepsis present on admission (by admission International Classification of Diseases (ICD) code); patients who were in the ICU within 24 h prior to the prediction time. To avoid the conflation of post-surgical care with sepsis care, patients were ineligible if they had surgery within 72 h prior to the prediction time. To avoid predicting on patients with excessive missingness, encounters were also required to have at least 3 of each vital sign and at least one complete blood count, and one basic or comprehensive metabolic panel test within 24 h prior to the prediction time (<xref ref-type="supplementary-material" rid="SM1">Supplementary Figure 1</xref>).</p>
<p>This project was approved with a waiver of informed consent by the Washington University in St. Louis Institutional Review Board (IRB &#x00023;201804121).</p>
</sec>
<sec>
<title>Sepsis Definition</title>
<p>Sepsis was defined using the Sepsis-3 implementation based on Suspicion of Infection (SOI) determined by concomitant antibiotics and cultures, Sequential Organ Failure Assessment (SOFA) score in the ICU setting, and qSOFA elsewhere (<xref ref-type="supplementary-material" rid="SM1">Supplementary Methods 3</xref>) (<xref ref-type="bibr" rid="B8">8</xref>, <xref ref-type="bibr" rid="B15">15</xref>). The anti-infectives for SOI was limited to intravenous anti-infectives except oral vancomycin and metronidazole. In accordance with the Sepsis-3 criteria, SOI required either having antibiotics within 72 h of culture collection or culture collection within 24 h of having antibiotics (<xref ref-type="bibr" rid="B8">8</xref>). The time of Suspicion of Infection (T<sub>SOI</sub>) was the time either before antibiotic order start time or culture collection time (<xref ref-type="bibr" rid="B15">15</xref>). To meet sepsis criteria, the patient must have had a SOFA or qSOFA score &#x02265; 2, depending on the location, between 48 h prior to and 24 h after T<sub>SOI</sub>. For sepsis cases, the time of sepsis onset (T<sub>Sepsis</sub>) was the same as T<sub>SOI</sub>.</p>
<p>To facilitate the model development, each encounter was assigned an index time (T<sub>Index</sub>), which for sepsis encounters was T<sub>Sepsis</sub>, and for non-sepsis encounters was 6-h prior to the maximum of either (1) the midpoint between admission and discharge, or (2) 12 h into admission. The time of prediction (T<sub>Prediction</sub>) was 6-h prior to the index time (<bold>eMethods 3</bold>).</p>
</sec>
<sec>
<title>Feature Generation and Engineering</title>
<p>Features were generated from the demographics, locations, medications, vital signs, and laboratory data available until the time of prediction (<xref ref-type="supplementary-material" rid="SM1">Supplementary Methods 2</xref>). Medications were mapped to classes and subcategories of the Multum MediSource Lexicon by Cerner (Denver, CO). Comorbidities were determined using ICD codes only from prior admissions and were mapped using the Elixhauser comorbidity system (<xref ref-type="bibr" rid="B16">16</xref>). The time series data were summarized as various univariate statistics (max, mean, etc.) over multiple time horizons (3 h, 6 h, etc.). Measures of variance such as SD were only computed if there were at least 4 measurements within the time horizon. Missing values, especially results of non-routine lab tests, were likely not missing at random but as a result of clinical judgment, thus, were kept as is (<xref ref-type="supplementary-material" rid="SM1">Supplementary Table 2</xref>). Features with &#x0003E;75% missingness, however, were excluded as they are unlikely to improve performance. For models that required fully non-null input, mean-imputation was used.</p>
</sec>
<sec>
<title>Model Development</title>
<p>Patient encounters were split at the patient level to avoid &#x0201C;identity confounding&#x0201D; into the train (75%) and test sets (25%) (<xref ref-type="bibr" rid="B17">17</xref>). Data transformation parameters were generated based on the training set, then applied to both sets. Random search with repeated cross-validation on the training set was used to tune the hyperparameters of an eXtreme Gradient Boosting (XGBoost) model, and the optimal combination was used for training on the full training set (<xref ref-type="supplementary-material" rid="SM1">Supplementary Methods 4</xref>) (<xref ref-type="bibr" rid="B18">18</xref>). The feature importance for the optimized XGBoost model (<bold>XGB opt</bold>) was estimated using the well-validated SHAP approach, a method of credit attribution based on coalitional game theory with useful properties such as additivity and the ability to provide explanations for individual predictions (<xref ref-type="bibr" rid="B19">19</xref>). To condense the model into one that is easier to transport and implement, a &#x0201C;lite&#x0201D; version of the XGBoost model (<bold>XGB lite</bold>) was created using a small subset of features based on the sum of the absolute SHapley Additive exPlanations (SHAP) values across the training set (<xref ref-type="supplementary-material" rid="SM1">Supplementary Figure 2</xref>). For comparison, an XGBoost model with default parameters (<bold>XGB unopt</bold>) was trained, as was a logistic regression model with l2 regularization (<bold>LogReg</bold>; <xref ref-type="supplementary-material" rid="SM1">Supplementary Figure 3</xref>).</p>
</sec>
<sec>
<title>Model Performance</title>
<p>The trained models were compared against the <bold>SIRS</bold> score, National Early Warning Score 2 (<bold>NEWS2</bold>), and <bold>qSOFA</bold> score (<xref ref-type="bibr" rid="B6">6</xref>&#x02013;<xref ref-type="bibr" rid="B8">8</xref>, <xref ref-type="bibr" rid="B20">20</xref>). Using the data from within the 24-h time window preceding prediction time, SIRS was calculated as the highest score occurring within a 1-h sliding window; NEWS2 was calculated using the last available measurements; qSOFA was calculated using the most abnormal measurements. For SIRS, NEWS2, and qSOFA, the lack of measurements was interpreted as normal.</p>
<p>The performance of the model was evaluated on bootstrap samples of the test set. Evaluated metrics include the Area Under the Receiver Operating Characteristic curve (<bold>AUROC</bold>) and the Area Under the Precision-Recall Curve (<bold>AUPRC</bold>). Model calibration was assessed by binning the test set into deciles of predicted risk and comparing their predicted probability of sepsis with the actual proportion of sepsis cases. The impact of threshold selection was visualized by plotting performance metrics (specificity, sensitivity, etc.) against the probability threshold.</p>
</sec>
<sec>
<title>Pseudo-Prospective Trial</title>
<p>While the model was trained and evaluated on a single time point per encounter, real-world implementation would likely involve continuous risk prediction throughout patient encounters. To better understand the implemented performance of the best performing sepsis prediction algorithm, the model was applied hourly to patient encounters in the test set spanning full admission duration. Patients whose model prediction crossed the threshold maximizing F1 score (harmonic mean of precision and recall) will hereby be referred to as having been &#x0201C;alerted on,&#x0201D; and for those, &#x0201C;alert time&#x0201D; was defined as the first alert instance for the encounter. For each patient hour, time-sensitive exclusion criteria (e.g., not in the general ward or already on anti-infectives) were applied again to remove inappropriate alerts. First, the cross-tabulation of sepsis status and alert status was generated. Then, among those who were alerted, we assessed the proportion of encounters with the following sepsis-related interventions and outcomes: sepsis-relevant culture collection, sepsis-relevant anti-infective administration, ventilator initiation, ICU transfer, sepsis onset, or death.</p>
</sec>
<sec>
<title>Statistical Analysis</title>
<p>Variables were summarized using frequencies and proportions for categorical data or medians and interquartile ranges (IQR) for continuous data. Statistical comparisons were performed using the Chi-square and Mann&#x02013;Whitney <italic>U</italic> tests where appropriate. A <italic>p</italic>-value &#x0003C; 0.01 was considered statistically significant. Analysis and figure generation were performed with Python version 3.7.1 (Python Software Foundation, Beaverton, OR) using the following packages: scipy, numpy, pandas, matplotlib, sklearn, xgboost, and shap (<xref ref-type="bibr" rid="B18">18</xref>, <xref ref-type="bibr" rid="B21">21</xref>&#x02013;<xref ref-type="bibr" rid="B26">26</xref>).</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<sec>
<title>Patient Population</title>
<p>From the initial inpatient population of 401,235 encounters, 331,201 met exclusion criteria, leaving 70,034 encounters in the final cohort (<xref ref-type="supplementary-material" rid="SM1">Supplementary Figure 1</xref>). Application of the Sepsis-3 criteria identified 2,206 (3.1%) patient with sepsis encounters. Patients with sepsis were slightly older [65.6 (56.3&#x02013;74.3) vs. 60.8 (49.4&#x02013;71.2), <italic>p</italic> &#x0003C; 0.01], more likely to be white (71.3 vs. 61.8%, <italic>p</italic> &#x0003C; 0.01), had a higher Elixhauser comorbidity score [19 (10&#x02013;29) vs. 9 (1&#x02013;17), <italic>p</italic> &#x0003C; 0.01], a longer length of stay [12.9 (8.0&#x02013;19.3) vs. 3.9 (2.3&#x02013;6.7), <italic>p</italic> &#x0003C; 0.01], and higher inpatient mortality (16.6% vs. 0.8%, <italic>p</italic> &#x0003C; 0.01) (<xref ref-type="table" rid="T1">Table 1</xref>, <xref ref-type="supplementary-material" rid="SM1">Supplementary Table 3</xref>).</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Cohort characteristics.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>Variable</bold></th>
<th valign="top" align="center"><bold>Total</bold><break/> <bold>[<italic>n</italic> = 70,034</bold><break/> <bold>(100.0%)]</bold></th>
<th valign="top" align="center"><bold>Sepsis</bold><break/> <bold>(<italic>n</italic> = 2,206 [3.1%])</bold></th>
<th valign="top" align="center"><bold>Non-sepsis</bold><break/> <bold>(<italic>n</italic> = 67,828 [96.9%])</bold></th>
<th valign="top" align="center"><bold>p<xref ref-type="table-fn" rid="TN1"><sup>a</sup></xref></bold></th>
</tr>
</thead>
<tbody>
<tr>
<td/>
<td/>
<td/>
<td/>
<td valign="top" align="center">&#x0003C;0.01<bold>&#x0002A;</bold></td>
</tr>
<tr>
<td valign="top" align="left">Age (years), median (IQR)</td>
<td valign="top" align="center">61.0 (49.6&#x02013;71.3)</td>
<td valign="top" align="center">65.5 (56.3&#x02013;74.3)</td>
<td valign="top" align="center">60.8 (49.4&#x02013;71.2)</td>
<td valign="top" align="center">&#x0003C;0.01<bold>&#x0002A;</bold> </td>
</tr>
<tr>
<td valign="top" align="left">Sex (female), <italic>n</italic> (%)</td>
<td valign="top" align="center">32,751 (46.8%)</td>
<td valign="top" align="center">992 (45.0%)</td>
<td valign="top" align="center">31,759 (46.8%)</td>
<td valign="top" align="center">0.090</td>
</tr>
<tr>
<td valign="top" align="left">Race, <italic>n</italic> (%)</td>
<td/>
<td/>
<td/>
<td valign="top" align="center">&#x0003C;0.01<bold>&#x0002A;</bold></td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0; White, <italic>n</italic> (%)</td>
<td valign="top" align="center">43,516 (62.1%)</td>
<td valign="top" align="center">1,573 (71.3%)</td>
<td valign="top" align="center">41,943 (61.8%)</td>
<td valign="top" align="center">&#x0003C;0.01<bold>&#x0002A;</bold></td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0; Other/unknown, <italic>n</italic> (%)</td>
<td valign="top" align="center">3,787 (5.4%)</td>
<td valign="top" align="center">129 (5.8%)</td>
<td valign="top" align="center">3,658 (5.4%)</td>
<td valign="top" align="center">0.378</td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0; Black, <italic>n</italic> (%)</td>
<td valign="top" align="center">22,285 (31.8%)</td>
<td valign="top" align="center">487 (22.1%)</td>
<td valign="top" align="center">21,798 (32.1%)</td>
<td valign="top" align="center">&#x0003C;0.01<bold>&#x0002A;</bold></td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0; Asian, <italic>n</italic> (%)</td>
<td valign="top" align="center">446 (0.6%)</td>
<td valign="top" align="center">17 (0.8%)</td>
<td valign="top" align="center">429 (0.6%)</td>
<td valign="top" align="center">0.505</td>
</tr>
<tr>
<td valign="top" align="left">BMI, median (IQR)</td>
<td valign="top" align="center">27.6 (23.5&#x02013;33.0)</td>
<td valign="top" align="center">27.2 (23.1&#x02013;33.4)</td>
<td valign="top" align="center">27.6 (23.5&#x02013;33.0)</td>
<td valign="top" align="center">0.252</td>
</tr>
<tr>
<td valign="top" align="left">Admitted through ED, <italic>n</italic> (%)</td>
<td valign="top" align="center">33,364 (47.6%)</td>
<td valign="top" align="center">747 (33.9%)</td>
<td valign="top" align="center">32,617 (48.1%)</td>
<td valign="top" align="center">&#x0003C;0.01<bold>&#x0002A;</bold></td>
</tr>
<tr>
<td valign="top" align="left">LOS (days), median (IQR)</td>
<td valign="top" align="center">3.9 (2.4&#x02013;7.0)</td>
<td valign="top" align="center">12.9 (8.0&#x02013;19.3)</td>
<td valign="top" align="center">3.9 (2.3&#x02013;6.7)</td>
<td valign="top" align="center">&#x0003C;0.01<bold>&#x0002A;</bold></td>
</tr>
<tr>
<td valign="top" align="left">Discharge disposition</td>
<td/>
<td/>
<td/>
<td valign="top" align="center">&#x0003C;0.01<bold>&#x0002A;</bold></td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0; Home, <italic>n</italic> (%)</td>
<td valign="top" align="center">59,367 (84.8%)</td>
<td valign="top" align="center">1,185 (53.7%)</td>
<td valign="top" align="center">58,182 (85.8%)</td>
<td valign="top" align="center">&#x0003C;0.01<bold>&#x0002A;</bold></td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0; Hospice, <italic>n</italic> (%)</td>
<td valign="top" align="center">854 (1.2%)</td>
<td valign="top" align="center">88 (4.0%)</td>
<td valign="top" align="center">766 (1.1%)</td>
<td valign="top" align="center">&#x0003C;0.01<bold>&#x0002A;</bold></td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0; Acute care facility, <italic>n</italic> (%)</td>
<td valign="top" align="center">436 (0.6%)</td>
<td valign="top" align="center">17 (0.8%)</td>
<td valign="top" align="center">419 (0.6%)</td>
<td valign="top" align="center">0.447</td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0; Nonacute care facility, <italic>n</italic> (%)</td>
<td valign="top" align="center">8,234 (11.8%)</td>
<td valign="top" align="center">539 (24.4%)</td>
<td valign="top" align="center">7,695 (11.3%)</td>
<td valign="top" align="center">&#x0003C;0.01<bold>&#x0002A;</bold></td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0; In-hospital death, <italic>n</italic> (%)</td>
<td valign="top" align="center">889 (1.3%)</td>
<td valign="top" align="center">367 (16.6%)</td>
<td valign="top" align="center">522 (0.8%)</td>
<td valign="top" align="center">&#x0003C;0.01<bold>&#x0002A;</bold></td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0; Other, <italic>n</italic> (%)</td>
<td valign="top" align="center">254 (0.4%)</td>
<td valign="top" align="center">10 (0.5%)</td>
<td valign="top" align="center">244 (0.4%)</td>
<td valign="top" align="center">0.589</td>
</tr>
<tr>
<td valign="top" align="left">Sepsis discharge ICD code<xref ref-type="table-fn" rid="TN2"><sup>b</sup></xref></td>
<td/>
<td/>
<td/>
<td valign="top" align="center">&#x0003C;0.01<bold>&#x0002A;</bold></td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0; Sepsis, <italic>n</italic> (%)</td>
<td valign="top" align="center">1,049 (1.5%)</td>
<td valign="top" align="center">543 (24.6%)</td>
<td valign="top" align="center">506 (0.7%)</td>
<td valign="top" align="center">&#x0003C;0.01<bold>&#x0002A;</bold></td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0; Severe sepsis, <italic>n</italic> (%)</td>
<td valign="top" align="center">510 (0.7%)</td>
<td valign="top" align="center">358 (16.2%)</td>
<td valign="top" align="center">152 (0.2%)</td>
<td valign="top" align="center">&#x0003C;0.01<bold>&#x0002A;</bold></td>
</tr>
<tr>
<td valign="top" align="left">&#x000A0; Septic shock, <italic>n</italic> (%)</td>
<td valign="top" align="center">378 (0.5%)</td>
<td valign="top" align="center">293 (13.3%)</td>
<td valign="top" align="center">85 (0.1%)</td>
<td valign="top" align="center">&#x0003C;0.01<bold>&#x0002A;</bold></td>
</tr>
<tr>
<td valign="top" align="left">30-day readmission, <italic>n</italic> (%)</td>
<td valign="top" align="center">14,817 (21.2%)</td>
<td valign="top" align="center">440 (19.9%)</td>
<td valign="top" align="center">14,377 (21.2%)</td>
<td valign="top" align="center">0.165</td>
</tr>
<tr>
<td valign="top" align="left">Elixhauser comorbidity score, median (IQR)<xref ref-type="table-fn" rid="TN3"><sup>c</sup></xref></td>
<td valign="top" align="center">9 (1&#x02013;18)</td>
<td valign="top" align="center">19 (10&#x02013;29)</td>
<td valign="top" align="center">9 (1&#x02013;17)</td>
<td valign="top" align="center">&#x0003C;0.01<bold>&#x0002A;</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><italic>BMI, body mass index; ED, emergency department; LOS, length of stay; ICD, International Classification of Diseases</italic>.</p>
<fn id="TN1">
<label>a</label>
<p><italic>Comparison of variables between sepsis and non-sepsis cohort was performed using Mann&#x02013;Whitney U test for continuous variables, and &#x003C7;<sup>2</sup> for categorical variables. Statistical significance (p &#x0003C; 0.01) is denoted by &#x0002A;</italic>.</p></fn>
<fn id="TN2">
<label>b</label>
<p><italic>Based on sepsis discharge ICD code list from (<xref ref-type="bibr" rid="B27">27</xref>)</italic>.</p></fn>
<fn id="TN3">
<label>c</label>
<p><italic>Based on Elixhauser comorbidity weights from (<xref ref-type="bibr" rid="B28">28</xref>)</italic>.</p></fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec>
<title>Model Performance</title>
<p>The optimized XGBoost model (<bold>XGB opt</bold>) using all 1,071 features had the highest AUROC (0.862 &#x000B1; 0.011) and AUPRC (0.294 &#x000B1; 0.021), compared to the unoptimized XGBoost model (<bold>XGB unopt</bold>), logistic regression (<bold>LogReg</bold>), and the lite XGBoost model (<bold>XGB lite</bold>), all of which had similar performances only slightly worse than XGB opt (<xref ref-type="fig" rid="F1">Figure 1</xref>, <xref ref-type="supplementary-material" rid="SM1">Supplementary Table 4</xref>). The scoring systems, however, had a significantly lower performance with a loss in AUROC over 0.150.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Model performance: Receiver Operating Characteristic curve and Precision-Recall curve. The solid lines represent the 50th percentile curves based on 20 bootstraps (full resampling with replacement) iterations of the test dataset, and the shaded regions represent the area between the 25th and 75th percentiles. AUROC, area under receiver operating characteristic curve; AUPRC, area under precision recall curve; XGB opt, optimized XGBoost model; XGB lite, simple XGBoost model; XGB unopt, unoptimized, out-of-the-box XGBoost model; LogReg, logistic regression; NEWS2, National Early Warning Score 2; qSOFA, quick Sequential Organ Failure Assessment; SIRS, Systemic Inflammatory Response Syndrome.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fdgth-04-848599-g0001.tif"/>
</fig>
<p>The top five most impactful features for the optimized XGBoost model were found to be: time from admission to prediction time, NEWS2 score, age, qSOFA score, and maximum respiratory rate within 48 h prior to prediction time (<xref ref-type="fig" rid="F2">Figure 2</xref>). The calibration curve yielded an <italic>r</italic><sup>2</sup> value of 0.837 (<xref ref-type="supplementary-material" rid="SM1">Supplementary Figure 4</xref>). The threshold plot demonstrates the tradeoff between precision and recall and revealed the highest F1 score (0.346) to be at a threshold of around 0.137 (<xref ref-type="fig" rid="F3">Figure 3</xref>).</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>SHapley Additive exPlanations (SHAP) feature importance. Comparison of variables between the sepsis and non-sepsis cohort was performed using the Mann&#x02013;Whitney <italic>U</italic> test for continuous variables, and &#x003C7;<sup>2</sup> for categorical variables. Statistical significance (<italic>p</italic> &#x0003C; 0.01) is denoted by &#x0002A;. qSOFA, quick sequential organ failure assessment; NEWS2, national early warning system 2; SBP, systolic blood pressure; WBC, white blood cell count; MAP, mean arterial pressure.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fdgth-04-848599-g0002.tif"/>
</fig>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Threshold plot for the optimized XGBoost model. The test set was bootstrapped (full resampling with replacement) 20 times and various performance metrics (recall, precision, specificity, and F1) were plotted against the threshold value. For each metric, the line and shaded area represent the median and IQR. A vertical black line was drawn at the threshold maximizing the F1 score.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fdgth-04-848599-g0003.tif"/>
</fig>
</sec>
<sec>
<title>Pseudo-Prospective Trial</title>
<p>The EHR data of the 17,441 encounters in the test set (557 sepsis encounters and 16,884 non-sepsis encounters) were binned hourly into 2,387,482 patient hours. After exclusions, 3,532 encounters were alerted upon, of which 388 met the sepsis criteria (11.0% PPV) (<xref ref-type="supplementary-material" rid="SM1">Supplementary Table 4</xref>). Of the 557 sepsis encounters, 388 were alerted upon (69.7% sensitivity). Of the 13,740 non-sepsis encounters, 3,144 were alerted upon (81.4% specificity).</p>
<p>Of the 3,532 alerted encounters, from the time of the first alert, within 48 h, 23.9% had sepsis-relevant cultures drawn, 13.2% received sepsis-relevant anti-infectives, 2.5% had ventilator initiated, 6.9% experienced sepsis onset, 4.7% were transferred to ICU, and 0.6% died (<xref ref-type="table" rid="T2">Table 2</xref>, <xref ref-type="supplementary-material" rid="SM1">Supplementary Table 5</xref>). Altogether, 29.1% experienced a sepsis-related intervention or outcome within 48 h of the first alert.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Pseudoprospective trial, outcomes for alerted subjects.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>Intervention or outcome</bold></th>
<th valign="top" align="center"><bold>Within 24 h</bold></th>
<th valign="top" align="center"><bold>Within 48 h</bold></th>
<th valign="top" align="center"><bold>Within 72 h</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Sepsis-relevant cultures</td>
<td valign="top" align="center">600 (17.0%)</td>
<td valign="top" align="center">843 (23.9%)</td>
<td valign="top" align="center">1,018 (28.8%)</td>
</tr>
<tr>
<td valign="top" align="left">Sepsis-relevant anti-infectives</td>
<td valign="top" align="center">286 (8.1%)</td>
<td valign="top" align="center">466 (13.2%)</td>
<td valign="top" align="center">591 (16.7%)</td>
</tr>
<tr>
<td valign="top" align="left">Ventilator initiation</td>
<td valign="top" align="center">51 (1.4%)</td>
<td valign="top" align="center">87 (2.5%)</td>
<td valign="top" align="center">119 (3.4%)</td>
</tr>
<tr>
<td valign="top" align="left">Sepsis onset</td>
<td valign="top" align="center">182 (5.2%)</td>
<td valign="top" align="center">245 (6.9%)</td>
<td valign="top" align="center">291 (8.2%)</td>
</tr>
<tr>
<td valign="top" align="left">ICU transfer</td>
<td valign="top" align="center">112 (3.2%)</td>
<td valign="top" align="center">167 (4.7%)</td>
<td valign="top" align="center">209 (5.9%)</td>
</tr>
<tr>
<td valign="top" align="left">Death</td>
<td valign="top" align="center">8 (0.2%)</td>
<td valign="top" align="center">21 (0.6%)</td>
<td valign="top" align="center">36 (1.0%)</td>
</tr>
<tr>
<td valign="top" align="left">Total</td>
<td valign="top" align="center">739 (20.9%)</td>
<td valign="top" align="center">1,028 (29.1%)</td>
<td valign="top" align="center">1,237 (35.0%)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><italic>Of the patients who crossed the set threshold in the pseudoprospective trial, and of those who were not already suspected of or being treated for sepsis, sepsis-related interventions and outcomes within various time horizons were identified</italic>.</p>
</table-wrap-foot>
</table-wrap>
<p>Visualizations of the sample patient trajectories alongside hourly predicted sepsis risk scores facilitated inspections of model successes and failures (<xref ref-type="supplementary-material" rid="SM1">Supplementary Figure 5</xref>).</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<p>The objective of this study was to develop a machine learning model capable of predicting sepsis 6-h ahead of clinical onset using one of the largest inpatient EHR datasets. Unlike most sepsis prediction studies which focus on the data-rich ICU or ED setting, this study focused on the general ward setting where the prediction task is made especially challenging due to the sparsity of data and low prevalence (<xref ref-type="bibr" rid="B29">29</xref>). Moreover, the cohort criteria excluded patients who were already suspected of, or were being treated for sepsis, as a clinical prediction model is unlikely to benefit these patients. The resultant cohort represents patients who were not captured by clinical judgment and thus could benefit from clinical decision support. Further, this study provides a novel way of better estimating real-world performance through the assessment of a pseudo-prospective trial.</p>
<p>Excluding patients who were suspected of or were already being treated for sepsis, alongside several other exclusion criteria, resulted in the elimination of the majority of inpatients from the initial population (<xref ref-type="supplementary-material" rid="SM1">Supplementary Figure 1</xref>). As a result, the retained sepsis cohort are likely cases of hospital-acquired sepsis or community-acquired sepsis with delayed recognition. Though the restrictive exclusion criteria may limit generalizability, the resultant cohort is more likely to benefit from an automated warning system.</p>
<p>We compared the performance of several machine learning models as well as traditional scoring systems and found the optimized XGBoost to have the best AUROC and AUPRC for detecting sepsis ahead of meeting traditional diagnostic criteria. The &#x0201C;lite&#x0201D; model was used with 25 features and had a similar diagnostic performance.</p>
<p>Of the important features, as determined by SHAP, time from admission to prediction time was the most important, indicating that prolonged length of stay is both a risk factor and outcome for sepsis. The qSOFA and NEWS2 scores were also important predictors, demonstrating the utility of these scores as features though deficient on their own. Admission through the ED was associated with a lower probability of sepsis, likely due to the emphasis on sepsis screening in the ED setting. Interestingly, while most medication information was not important for the model, anticonvulsants had a surprisingly high SHAP value with sepsis patients receiving &#x0201C;anticonvulsants&#x0201D; about 10% more frequently than non-sepsis patients (46.6 vs. 36.6%, <xref ref-type="fig" rid="F2">Figure 2</xref>). However, the Multum classification for anticonvulsants included medications such as magnesium sulfate and lorazepam which are not always used as anticonvulsants, thus, more work is needed on automated feature generation from medication data. Another unexpectedly important feature was the Coombs test, which is unlikely to be related to sepsis, but had noticeably different rates of missingness between sepsis and non-sepsis patients (60.1% for sepsis vs. 70.3% for non-sepsis, <xref ref-type="fig" rid="F2">Figure 2</xref>). Comorbidities from prior admissions were noticeably absent from the list of important features, likely because 46.3% of all encounters were first encounters and did not have any prior admissions. It&#x00027;s possible that the importance of comorbidities as features may rise with time, with larger populations with longer histories being collected in the electronic health record and the ability to retrieve information cross-sites.</p>
<p>The pseudo-prospective trial demonstrated a novel approach to better estimating real-world model performance and showed that 29.1% of the alerted patients required sepsis-related intervention or had a sepsis-related outcome within 48 h (<xref ref-type="table" rid="T2">Table 2</xref>). While the algorithm was capable of identifying patients who ultimately required cultures (39.0%) and anti-infectives (28.1%), the actual incidence of Sepsis-3 onset after the patients were alerted on was relatively low (11.0% at any point after and 6.9% within 48 h). This may be due to problems in labeling&#x02014;despite our attempt to exclude surgical patients from the cohort, they are not capable of being excluded on a prospective basis and frequently meet the sepsis criteria. Moreover, alerted patients may be critically ill and treated for sepsis but not meet the Sepsis-3 criteria. Also, many patients who are in the ED have higher scores which improve through interventions, but then have scores that rise again later during their stay at the hospital. Since this only evaluated the first time a patient crossed the sepsis threshold, the subsequent and potentially more important clinical changes would be missed. The pseudo-prospective trial highlights some of the anticipated challenges of translating a diagnostic scoring method from a retrospective data set to a prospective population, which necessitates further investigation.</p>
<p>Impressively, the unoptimized XGBoost solution had a median AUROC just 5% lower than the optimized version, and similar performance to the optimized logistic regression model and the lite XGBoost model. The relatively small benefit conferred by the more complex model compared to logistic regression is consistent with prior literature (<xref ref-type="bibr" rid="B30">30</xref>). If the added complexity is problematic&#x02014;for interpretability, debugging, or implementation&#x02014;then it could be argued that the simpler logistic regression model is preferred despite the performance loss. Though NEWS2 and qSOFA were very important features in XGBoost, the gap between traditional scoring systems and machine learning models was noticeable with the worst ML model conferring a 15.1% AUROC improvement over the best traditional scoring system.</p>
<p>This study has limited generalizability as a single institution study. The study used an interpretation of the Sepsis-3 definition and is likely to generalize poorly to sites using alternate definitions (<xref ref-type="bibr" rid="B1">1</xref>, <xref ref-type="bibr" rid="B8">8</xref>, <xref ref-type="bibr" rid="B15">15</xref>). By design, the study was focused on the general ward setting, and the results are not applicable to other settings. Many of the excluded subpopulations (children, surgical, etc.) warrant further investigation. While a pseudo-prospective trial was performed, a true prospective study is needed to gauge real-world performance. The pseudo-prospective trial could be further improved by investigating repeated alerts, incorporating alert lock-out periods, accounting for measurement-to-documentation time gap, etc. For the pseudo-prospective trial, a threshold was assigned to maximize the F1 score. However, further work is necessary to define an operationally meaningful threshold. For the calculation of the qSOFA score, GCS was missing in our dataset and assumed normal, which may negatively impact the sepsis label assignment process. However, Seymour et al. found that the lack of GCS in the VA dataset did not significantly reduce the predictive validity of qSOFA (<xref ref-type="bibr" rid="B8">8</xref>). As is typical of studies using electronic health records data, there were and likely remain problems concerning missingness and accuracy of clinical data.</p>
</sec>
<sec sec-type="conclusions" id="s5">
<title>Conclusion</title>
<p>A machine learning model designed to predict sepsis 6-h ahead of meeting diagnostic criteria yielded an AUROC of 0.862 &#x000B1; 0.011 and an AUPRC of 0.294 &#x000B1; 0.021. Pseudo-prospective evaluation of the model revealed relatively good clinical performance, despite a large class imbalance.</p>
</sec>
<sec sec-type="data-availability" id="s6">
<title>Data Availability Statement</title>
<p>The data analyzed in this study is subject to the following licenses/restrictions: The dataset includes protected health information, specifically electronic health records with patient identifiers, thus cannot be shared. Requests to access these datasets should be directed to i2help&#x00040;wustl.edu.</p>
</sec>
<sec id="s7">
<title>Ethics Statement</title>
<p>The studies involving human participants were reviewed and approved by Washington University Institutional Review Board &#x00023;201804121. The Ethics Committee waived the requirement of written informed consent for participation.</p>
</sec>
<sec id="s8">
<title>Author Contributions</title>
<p>AM, AG, and SY contributed to study conception and design as well as data collection. Primary analysis of data was performed by SY, the results of which were discussed with AM, KB, and AG. The manuscript was drafted by SY and AM. The manuscript was reviewed and approved by all authors.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>This work was supported by the Big Ideas Grant <bold>(</bold>Washington University School of Medicine Institute for Informatics and Healthcare Innovation Lab).</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x00027;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<sec sec-type="supplementary-material" id="s11">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fdgth.2022.848599/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fdgth.2022.848599/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Singer</surname> <given-names>M</given-names></name> <name><surname>Deutschman</surname> <given-names>CS</given-names></name> <name><surname>Seymour</surname> <given-names>CW</given-names></name> <name><surname>Shankar-Hari</surname> <given-names>M</given-names></name> <name><surname>Annane</surname> <given-names>D</given-names></name> <name><surname>Bauer</surname> <given-names>M</given-names></name> <etal/></person-group>. <article-title>The third international consensus definitions for sepsis and septic shock (Sepsis-3)</article-title>. <source>JAMA</source>. (<year>2016</year>) <volume>315</volume>:<fpage>801</fpage>&#x02013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1001/jama.2016.0287</pub-id><pub-id pub-id-type="pmid">26903338</pub-id></citation></ref>
<ref id="B2">
<label>2.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liang</surname> <given-names>L</given-names></name> <name><surname>Moore</surname> <given-names>B</given-names></name> <name><surname>Soni</surname> <given-names>A</given-names></name></person-group>. <source>National Inpatient Hospital Costs: The Most Expensive Conditions by Payer, 2017: Statistical Brief&#x00023; 261</source>. Healthcare Cost and Utilization Project (HCUP) Statistical Briefs (<year>2006</year>).<pub-id pub-id-type="pmid">32833416</pub-id></citation></ref>
<ref id="B3">
<label>3.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>V</given-names></name> <name><surname>Escobar</surname> <given-names>GJ</given-names></name> <name><surname>Greene</surname> <given-names>JD</given-names></name> <name><surname>Soule</surname> <given-names>J</given-names></name> <name><surname>Whippy</surname> <given-names>A</given-names></name> <name><surname>Angus</surname> <given-names>DC</given-names></name> <etal/></person-group>. <article-title>Hospital deaths in patients with sepsis from 2 independent cohorts</article-title>. <source>JAMA</source>. (<year>2014</year>) <volume>312</volume>:<fpage>90</fpage>&#x02013;<lpage>2</lpage>. <pub-id pub-id-type="doi">10.1001/jama.2014.5804</pub-id><pub-id pub-id-type="pmid">24838355</pub-id></citation></ref>
<ref id="B4">
<label>4.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Seymour</surname> <given-names>CW</given-names></name> <name><surname>Gesten</surname> <given-names>F</given-names></name> <name><surname>Prescott</surname> <given-names>HC</given-names></name> <name><surname>Friedrich</surname> <given-names>ME</given-names></name> <name><surname>Iwashyna</surname> <given-names>TJ</given-names></name> <name><surname>Phillips</surname> <given-names>GS</given-names></name> <etal/></person-group>. <article-title>Time to treatment and mortality during mandated emergency care for sepsis</article-title>. <source>N Engl J Med</source>. (<year>2017</year>) <volume>376</volume>:<fpage>2235</fpage>&#x02013;<lpage>44</lpage>. <pub-id pub-id-type="doi">10.1056/NEJMoa1703058</pub-id><pub-id pub-id-type="pmid">28528569</pub-id></citation></ref>
<ref id="B5">
<label>5.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kumar</surname> <given-names>A</given-names></name> <name><surname>Roberts</surname> <given-names>D</given-names></name> <name><surname>Wood</surname> <given-names>KE</given-names></name> <name><surname>Light</surname> <given-names>B</given-names></name> <name><surname>Parrillo</surname> <given-names>JE</given-names></name> <name><surname>Sharma</surname> <given-names>S</given-names></name> <etal/></person-group>. <article-title>Duration of hypotension before initiation of effective antimicrobial therapy is the critical determinant of survival in human septic shock</article-title>. <source>Crit Care Med</source>. (<year>2006</year>) <volume>34</volume>:<fpage>1589</fpage>&#x02013;<lpage>96</lpage>. <pub-id pub-id-type="doi">10.1097/01.CCM.0000217961.75225.E9</pub-id><pub-id pub-id-type="pmid">16625125</pub-id></citation></ref>
<ref id="B6">
<label>6.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bone</surname> <given-names>RC</given-names></name> <name><surname>Balk</surname> <given-names>RA</given-names></name> <name><surname>Cerra</surname> <given-names>FB</given-names></name> <name><surname>Dellinger</surname> <given-names>RP</given-names></name> <name><surname>Fein</surname> <given-names>AM</given-names></name> <name><surname>Knaus</surname> <given-names>WA</given-names></name> <etal/></person-group>. <article-title>Definitions for sepsis and organ failure and guidelines for the use of innovative therapies in sepsis</article-title>. <source>Chest</source>. (<year>1992</year>) <volume>101</volume>:<fpage>1644</fpage>&#x02013;<lpage>55</lpage>. <pub-id pub-id-type="doi">10.1378/chest.101.6.1644</pub-id><pub-id pub-id-type="pmid">1597042</pub-id></citation></ref>
<ref id="B7">
<label>7.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pimentel</surname> <given-names>MA</given-names></name> <name><surname>Redfern</surname> <given-names>OC</given-names></name> <name><surname>Gerry</surname> <given-names>S</given-names></name> <name><surname>Collins</surname> <given-names>GS</given-names></name> <name><surname>Malycha</surname> <given-names>J</given-names></name> <name><surname>Prytherch</surname> <given-names>D</given-names></name> <etal/></person-group>. <article-title>A comparison of the ability of the National Early Warning Score and the National Early Warning Score 2 to identify patients at risk of in-hospital mortality: a multi-centre database study</article-title>. <source>Resuscitation</source>. (<year>2019</year>) <volume>134</volume>:<fpage>147</fpage>&#x02013;<lpage>56</lpage>. <pub-id pub-id-type="doi">10.1016/j.resuscitation.2018.09.026</pub-id><pub-id pub-id-type="pmid">30287355</pub-id></citation></ref>
<ref id="B8">
<label>8.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Seymour</surname> <given-names>CW</given-names></name> <name><surname>Liu</surname> <given-names>VX</given-names></name> <name><surname>Iwashyna</surname> <given-names>TJ</given-names></name> <name><surname>Brunkhorst</surname> <given-names>FM</given-names></name> <name><surname>Rea</surname> <given-names>TD</given-names></name> <name><surname>Scherag</surname> <given-names>A</given-names></name> <etal/></person-group>. <article-title>Assessment of clinical criteria for sepsis: for the Third International Consensus Definitions for Sepsis and Septic Shock (Sepsis-3)</article-title>. <source>JAMA</source>. (<year>2016</year>) <volume>315</volume>:<fpage>762</fpage>&#x02013;<lpage>74</lpage>. <pub-id pub-id-type="doi">10.1001/jama.2016.0288</pub-id><pub-id pub-id-type="pmid">26903335</pub-id></citation></ref>
<ref id="B9">
<label>9.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nemati</surname> <given-names>S</given-names></name> <name><surname>Holder</surname> <given-names>A</given-names></name> <name><surname>Razmi</surname> <given-names>F</given-names></name> <name><surname>Stanley</surname> <given-names>MD</given-names></name> <name><surname>Clifford</surname> <given-names>GD</given-names></name> <name><surname>Buchman</surname> <given-names>TG</given-names></name></person-group>. <article-title>An interpretable machine learning model for accurate prediction of sepsis in the ICU</article-title>. <source>Crit Care Med</source>. (<year>2018</year>) <volume>46</volume>:<fpage>547</fpage>. <pub-id pub-id-type="doi">10.1097/CCM.0000000000002936</pub-id><pub-id pub-id-type="pmid">29286945</pub-id></citation></ref>
<ref id="B10">
<label>10.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Desautels</surname> <given-names>T</given-names></name> <name><surname>Calvert</surname> <given-names>J</given-names></name> <name><surname>Hoffman</surname> <given-names>J</given-names></name> <name><surname>Jay</surname> <given-names>M</given-names></name> <name><surname>Kerem</surname> <given-names>Y</given-names></name> <name><surname>Shieh</surname> <given-names>L</given-names></name> <etal/></person-group>. <article-title>Prediction of sepsis in the intensive care unit with minimal electronic health record data: a machine learning approach</article-title>. <source>JMIR Med Inform</source>. (<year>2016</year>) <volume>4</volume>:<fpage>e5909</fpage>. <pub-id pub-id-type="doi">10.2196/medinform.5909</pub-id><pub-id pub-id-type="pmid">27694098</pub-id></citation></ref>
<ref id="B11">
<label>11.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Reyna</surname> <given-names>MA</given-names></name> <name><surname>Josef</surname> <given-names>C</given-names></name> <name><surname>Seyedi</surname> <given-names>S</given-names></name> <name><surname>Jeter</surname> <given-names>R</given-names></name> <name><surname>Shashikumar</surname> <given-names>SP</given-names></name> <name><surname>Westover</surname> <given-names>MB</given-names></name> <etal/></person-group>. <article-title>Early prediction of sepsis from clinical data: the PhysioNet/Computing in Cardiology Challenge 2019</article-title>. <source>Comput. Cardiol.</source> (<year>2019</year>) <volume>4</volume>:<fpage>1</fpage>&#x02013;<lpage>4</lpage>. <pub-id pub-id-type="doi">10.22489/CinC.2019.412</pub-id><pub-id pub-id-type="pmid">31939789</pub-id></citation></ref>
<ref id="B12">
<label>12.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Levy</surname> <given-names>MM</given-names></name> <name><surname>Dellinger</surname> <given-names>RP</given-names></name> <name><surname>Townsend</surname> <given-names>SR</given-names></name> <name><surname>Linde-Zwirble</surname> <given-names>WT</given-names></name> <name><surname>Marshall</surname> <given-names>JC</given-names></name> <name><surname>Bion</surname> <given-names>J</given-names></name> <etal/></person-group>. <article-title>The Surviving Sepsis Campaign: results of an international guideline-based performance improvement program targeting severe sepsis</article-title>. <source>Intensive Care Med</source>. (<year>2010</year>) <volume>36</volume>:<fpage>222</fpage>&#x02013;<lpage>31</lpage>. <pub-id pub-id-type="doi">10.1007/s00134-009-1738-3</pub-id><pub-id pub-id-type="pmid">20069275</pub-id></citation></ref>
<ref id="B13">
<label>13.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shah</surname> <given-names>PK</given-names></name> <name><surname>Ginestra</surname> <given-names>JC</given-names></name> <name><surname>Ungar</surname> <given-names>LH</given-names></name> <name><surname>Junker</surname> <given-names>P</given-names></name> <name><surname>Rohrbach</surname> <given-names>JI</given-names></name> <name><surname>Fishman</surname> <given-names>NO</given-names></name> <etal/></person-group>. <article-title>A simulated prospective evaluation of a deep learning model for real-time prediction of clinical deterioration among ward patients</article-title>. <source>Crit Care Med</source>. (<year>2021</year>) <volume>49</volume>:<fpage>1312</fpage>&#x02013;<lpage>21</lpage>. <pub-id pub-id-type="doi">10.1097/CCM.0000000000004966</pub-id><pub-id pub-id-type="pmid">33711001</pub-id></citation></ref>
<ref id="B14">
<label>14.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hyland</surname> <given-names>SL</given-names></name> <name><surname>Faltys</surname> <given-names>M</given-names></name> <name><surname>H&#x000FC;ser</surname> <given-names>M</given-names></name> <name><surname>Lyu</surname> <given-names>X</given-names></name> <name><surname>Gumbsch</surname> <given-names>T</given-names></name> <name><surname>Esteban</surname> <given-names>C</given-names></name> <etal/></person-group>. <article-title>Early prediction of circulatory failure in the intensive care unit using machine learning</article-title>. <source>Nat Med</source>. (<year>2020</year>) <volume>26</volume>:<fpage>364</fpage>&#x02013;<lpage>73</lpage>. <pub-id pub-id-type="doi">10.1038/s41591-020-0789-4</pub-id><pub-id pub-id-type="pmid">32152583</pub-id></citation></ref>
<ref id="B15">
<label>15.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>S</given-names></name> <name><surname>Betthauser</surname> <given-names>KD</given-names></name> <name><surname>Gupta</surname> <given-names>A</given-names></name> <name><surname>Lyons</surname> <given-names>PG</given-names></name> <name><surname>Lai</surname> <given-names>AM</given-names></name> <name><surname>Kollef</surname> <given-names>MH</given-names></name> <etal/></person-group>. <article-title>Comparison of sepsis definitions as automated criteria</article-title>. <source>Crit Care Med</source>. (<year>2021</year>) <volume>49</volume>:<fpage>e433</fpage>&#x02013;<lpage>43</lpage>. <pub-id pub-id-type="doi">10.1097/CCM.0000000000004875</pub-id><pub-id pub-id-type="pmid">33591014</pub-id></citation></ref>
<ref id="B16">
<label>16.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Quan</surname> <given-names>H</given-names></name> <name><surname>Sundararajan</surname> <given-names>V</given-names></name> <name><surname>Halfon</surname> <given-names>P</given-names></name> <name><surname>Fong</surname> <given-names>A</given-names></name> <name><surname>Burnand</surname> <given-names>B</given-names></name> <name><surname>Luthi</surname> <given-names>J-C</given-names></name> <etal/></person-group>. <article-title>Coding algorithms for defining comorbidities in ICD-9-CM and ICD-10 administrative data</article-title>. <source>Med Care</source>. (<year>2005</year>) <volume>43</volume>:<fpage>1130</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1097/01.mlr.0000182534.19832.83</pub-id><pub-id pub-id-type="pmid">16224307</pub-id></citation></ref>
<ref id="B17">
<label>17.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Neto</surname> <given-names>EC</given-names></name> <name><surname>Pratap</surname> <given-names>A</given-names></name> <name><surname>Perumal</surname> <given-names>TM</given-names></name> <name><surname>Tummalacherla</surname> <given-names>M</given-names></name> <name><surname>Snyder</surname> <given-names>P</given-names></name> <name><surname>Bot</surname> <given-names>BM</given-names></name> <etal/></person-group>. <article-title>Detecting the impact of subject characteristics on machine learning-based diagnostic applications</article-title>. <source>NPJ Digital Med</source>. (<year>2019</year>) <volume>2</volume>:<fpage>1</fpage>&#x02013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1038/s41746-019-0178-x</pub-id><pub-id pub-id-type="pmid">31633058</pub-id></citation></ref>
<ref id="B18">
<label>18.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>T</given-names></name> <name><surname>Guestrin</surname> <given-names>C</given-names></name></person-group>. <article-title>XG boost: A scalable tree boosting system</article-title>. In: <source>Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</source>. <publisher-loc>New York, NY</publisher-loc>: <publisher-name>ACM</publisher-name> (<year>2016</year>). p. <fpage>785</fpage>&#x02013;<lpage>94</lpage>. <pub-id pub-id-type="doi">10.1145/2939672.2939785</pub-id></citation>
</ref>
<ref id="B19">
<label>19.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lundberg</surname> <given-names>SM</given-names></name> <name><surname>Lee</surname> <given-names>S-I</given-names></name></person-group>editors. <article-title>A unified approach to interpreting model predictions</article-title>. In: <source>Proceedings of the 31st International Conference On Neural Information Processing Systems</source> (<year>2017</year>).</citation>
</ref>
<ref id="B20">
<label>20.</label>
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>SC</given-names></name> <name><surname>Shivakumar</surname> <given-names>N</given-names></name> <name><surname>Betthauser</surname> <given-names>K</given-names></name> <name><surname>Gupta</surname> <given-names>A</given-names></name> <name><surname>Lai</surname> <given-names>AM</given-names></name> <name><surname>Kollef</surname> <given-names>MH</given-names></name> <etal/></person-group>. <article-title>Performance of early warning scores for sepsis identification in the general ward setting</article-title>. <source>JAMIA Open</source>. (<year>2021</year>) <volume>4</volume>:<fpage>ooab062</fpage>. <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/jamiaopen/ooab062">https://doi.org/10.1093/jamiaopen/ooab062</ext-link>.</citation>
</ref>
<ref id="B21">
<label>21.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Virtanen</surname> <given-names>P</given-names></name> <name><surname>Gommers</surname> <given-names>R</given-names></name> <name><surname>Oliphant</surname> <given-names>TE</given-names></name> <name><surname>Haberland</surname> <given-names>M</given-names></name> <name><surname>Reddy</surname> <given-names>T</given-names></name> <name><surname>Cournapeau</surname> <given-names>D</given-names></name> <etal/></person-group>. <article-title>SciPy 1.0: fundamental algorithms for scientific computing in Python</article-title>. <source>Nat Methods</source>. (<year>2020</year>) <volume>17</volume>:<fpage>261</fpage>&#x02013;<lpage>72</lpage>. <pub-id pub-id-type="doi">10.1038/s41592-020-0772-5</pub-id><pub-id pub-id-type="pmid">32094914</pub-id></citation></ref>
<ref id="B22">
<label>22.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Harris</surname> <given-names>CR</given-names></name> <name><surname>Millman</surname> <given-names>KJ</given-names></name> <name><surname>van der Walt</surname> <given-names>SJ</given-names></name> <name><surname>Gommers</surname> <given-names>R</given-names></name> <name><surname>Virtanen</surname> <given-names>P</given-names></name> <name><surname>Cournapeau</surname> <given-names>D</given-names></name> <etal/></person-group>. <article-title>Array programming with NumPy</article-title>. <source>Nature</source>. (<year>2020</year>) <volume>585</volume>:<fpage>357</fpage>&#x02013;<lpage>62</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-020-2649-2</pub-id><pub-id pub-id-type="pmid">32939066</pub-id></citation></ref>
<ref id="B23">
<label>23.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>McKinney</surname> <given-names>W</given-names></name> <name><surname>editor Data structures for statistical computing in</surname> <given-names>python</given-names></name></person-group>. In: <source>Proceedings of the 9th Python in Science Conferenc</source>. <publisher-loc>Austin, TX</publisher-loc> (<year>2010</year>). <pub-id pub-id-type="doi">10.25080/Majora-92bf1922-00a</pub-id></citation>
</ref>
<ref id="B24">
<label>24.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hunter</surname> <given-names>JD</given-names></name></person-group>. <article-title>Matplotlib: a 2D graphics environment</article-title>. <source>IEEE Ann Hist Comput</source>. (<year>2007</year>) <volume>9</volume>:<fpage>90</fpage>&#x02013;<lpage>5</lpage>. <pub-id pub-id-type="doi">10.1109/MCSE.2007.55</pub-id></citation>
</ref>
<ref id="B25">
<label>25.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pedregosa</surname> <given-names>F</given-names></name> <name><surname>Varoquaux</surname> <given-names>G</given-names></name> <name><surname>Gramfort</surname> <given-names>A</given-names></name> <name><surname>Michel</surname> <given-names>V</given-names></name> <name><surname>Thirion</surname> <given-names>B</given-names></name> <name><surname>Grisel</surname> <given-names>O</given-names></name> <etal/></person-group>. <article-title>Scikit-learn: machine learning in Python</article-title>. <source>J Mach Learn Res</source>. (<year>2011</year>) <volume>12</volume>:<fpage>2825</fpage>&#x02013;<lpage>30</lpage>. <pub-id pub-id-type="doi">10.5555/1953048.2078195</pub-id></citation>
</ref>
<ref id="B26">
<label>26.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lundberg</surname> <given-names>SM</given-names></name> <name><surname>Erion</surname> <given-names>G</given-names></name> <name><surname>Chen</surname> <given-names>H</given-names></name> <name><surname>DeGrave</surname> <given-names>A</given-names></name> <name><surname>Prutkin</surname> <given-names>JM</given-names></name> <name><surname>Nair</surname> <given-names>B</given-names></name> <etal/></person-group>. <article-title>From local explanations to global understanding with explainable AI for trees</article-title>. <source>Nat Mach Intellig</source>. (<year>2020</year>) <volume>2</volume>:<fpage>56</fpage>&#x02013;<lpage>67</lpage>. <pub-id pub-id-type="doi">10.1038/s42256-019-0138-9</pub-id><pub-id pub-id-type="pmid">32607472</pub-id></citation></ref>
<ref id="B27">
<label>27.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Buchman</surname> <given-names>TG</given-names></name> <name><surname>Simpson</surname> <given-names>SQ</given-names></name> <name><surname>Sciarretta</surname> <given-names>KL</given-names></name> <name><surname>Finne</surname> <given-names>KP</given-names></name> <name><surname>Sowers</surname> <given-names>N</given-names></name> <name><surname>Collier</surname> <given-names>M</given-names></name> <etal/></person-group>. <article-title>Sepsis among medicare beneficiaries: 1. The burdens of sepsis, 20122018</article-title>. <source>Critic Care Med</source>. (<year>2020</year>) <volume>48</volume>:<fpage>276</fpage>&#x02013;<lpage>88</lpage>. <pub-id pub-id-type="doi">10.1097/CCM.0000000000004224</pub-id><pub-id pub-id-type="pmid">32058366</pub-id></citation></ref>
<ref id="B28">
<label>28.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Moore</surname> <given-names>BJ</given-names></name> <name><surname>White</surname> <given-names>S</given-names></name> <name><surname>Washington</surname> <given-names>R</given-names></name> <name><surname>Coenen</surname> <given-names>N</given-names></name> <name><surname>Elixhauser</surname> <given-names>A</given-names></name></person-group>. <article-title>Identifying increased risk of readmission and in-hospital mortality using hospital administrative data</article-title>. <source>Med Care</source>. (<year>2017</year>) <volume>55</volume>:<fpage>698</fpage>&#x02013;<lpage>705</lpage>. <pub-id pub-id-type="doi">10.1097/MLR.0000000000000735</pub-id><pub-id pub-id-type="pmid">28857964</pub-id></citation></ref>
<ref id="B29">
<label>29.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fleuren</surname> <given-names>LM</given-names></name> <name><surname>Klausch</surname> <given-names>TL</given-names></name> <name><surname>Zwager</surname> <given-names>CL</given-names></name> <name><surname>Schoonmade</surname> <given-names>LJ</given-names></name> <name><surname>Guo</surname> <given-names>T</given-names></name> <name><surname>Roggeveen</surname> <given-names>LF</given-names></name> <etal/></person-group>. <article-title>Machine learning for the prediction of sepsis: a systematic review and meta-analysis of diagnostic test accuracy</article-title>. <source>Intens Care Med</source>. (<year>2020</year>) <volume>46</volume>:<fpage>383</fpage>&#x02013;<lpage>400</lpage>. <pub-id pub-id-type="doi">10.1007/s00134-019-05872-y</pub-id><pub-id pub-id-type="pmid">31965266</pub-id></citation></ref>
<ref id="B30">
<label>30.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rajkomar</surname> <given-names>A</given-names></name> <name><surname>Oren</surname> <given-names>E</given-names></name> <name><surname>Chen</surname> <given-names>K</given-names></name> <name><surname>Dai</surname> <given-names>AM</given-names></name> <name><surname>Hajaj</surname> <given-names>N</given-names></name> <name><surname>Hardt</surname> <given-names>M</given-names></name> <etal/></person-group>. <article-title>Scalable and accurate deep learning with electronic health records</article-title>. <source>NPJ Digit Med</source>. (<year>2018</year>) <volume>1</volume>:<fpage>1</fpage>&#x02013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1038/s41746-018-0029-1</pub-id><pub-id pub-id-type="pmid">31304302</pub-id></citation></ref>
</ref-list> 
</back>
</article> 