<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Drug Saf. Regul.</journal-id>
<journal-title>Frontiers in Drug Safety and Regulation</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Drug Saf. Regul.</abbrev-journal-title>
<issn pub-type="epub">2674-0869</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1267623</article-id>
<article-id pub-id-type="doi">10.3389/fdsfr.2023.1267623</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Drug Safety and Regulation</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Identification of small cell lung cancer patients who are at risk of developing common serious adverse event groups with machine learning</article-title>
<alt-title alt-title-type="left-running-head">Wanika et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fdsfr.2023.1267623">10.3389/fdsfr.2023.1267623</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Wanika</surname>
<given-names>Linda</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/2345417/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Evans</surname>
<given-names>Neil D.</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/377577/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Chappell</surname>
<given-names>Michael J.</given-names>
</name>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/302390/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
</contrib-group>
<aff>
<institution>School of Engineering</institution>, <institution>University of Warwick</institution>, <addr-line>Coventry</addr-line>, <country>United Kingdom</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/633890/overview">Assaf Gottlieb</ext-link>, University of Texas Health Science Center at Houston, United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> Xiaoqian Jiang, University of Texas Health Science Center at Houston, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/460057/overview">Pantelis Natsiavas</ext-link>, Centre for Research and Technology Hellas (INAB&#x7c;CERTH), Greece</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Michael J. Chappell, <email>m.j.chappell@warwick.ac.uk</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>15</day>
<month>09</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>3</volume>
<elocation-id>1267623</elocation-id>
<history>
<date date-type="received">
<day>26</day>
<month>07</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>28</day>
<month>08</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Wanika, Evans and Chappell.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Wanika, Evans and Chappell</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>
<bold>Introduction:</bold> Across multiple studies, the most common serious adverse event groups that Small Cell Lung Cancer (SCLC) patients experience, whilst undergoing chemotherapy treatment, are: Blood and Lymphatic Disorders, Infections and Infestations together with Metabolism and Nutrition Disorders. The majority of the research that investigates the relationship between adverse events and SCLC patients, focuses on specific adverse events such as neutropenia and thrombocytopenia.</p>
<p>
<bold>Aim:</bold> This study aims to utilise machine learning in order to identify those patients who are at risk of developing common serious adverse event groups, as well as their specific adverse event classification grade.</p>
<p>
<bold>Methods:</bold> Data from five clinical trial studies were analysed and 12 analysis groups were formed based on the serious adverse event group and grade.</p>
<p>
<bold>Results:</bold> The best test runs for each of the models were able to produce an area under the curve (AUC) score of at least 0.714. The best model was the Blood and Lymphatic Disorder group, SAE grade 0 vs. grade 3 (best AUC &#x3d; 1, sensitivity rate &#x3d; 0.84, specificity rate &#x3d; 0.96).</p>
<p>
<bold>Conclusion:</bold> The top features that contributed to this prediction were total bilirubin, alkaline phosphatase, and age. Future work should investigate the relationship between these features and common SAE groups.</p>
</abstract>
<kwd-group>
<kwd>small cell lung cancer</kwd>
<kwd>serious adverse events</kwd>
<kwd>grade classification</kwd>
<kwd>chemotherapy</kwd>
<kwd>machine learning</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Advanced Methods in Pharmacovigilance and Pharmacoepidemiology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Lung cancer is one of the most common cancers worldwide. Approximately 7.5% of people are at risk of developing lung cancer (<xref ref-type="bibr" rid="B7">Cancer Research UK, 2023</xref>; <xref ref-type="bibr" rid="B56">World Cancer Research Fund International, 2023</xref>). Small Cell Lung Cancer (SCLC) accounts for 15% of lung cancer cases (<xref ref-type="bibr" rid="B26">Kahnert et al., 2016</xref>). While the majority of lung cancer diagnoses are Non-Small Cell Lung Cancer cases, SCLC patients in general have a higher metastasise rate (<xref ref-type="bibr" rid="B31">Krohn et al., 2014</xref>). Treatment options for SCLC aim to simply manage the disease (<xref ref-type="bibr" rid="B19">Deneka et al., 2019</xref>). Chemotherapy is one of the main treatment options for SCLC with the goal of reducing the spread of the tumour through the disruption of the tumour DNA replication and cell division process (<xref ref-type="bibr" rid="B29">Kitao et al., 2017</xref>). Common examples of such chemotherapy include carboplatin and cisplatin (<xref ref-type="bibr" rid="B43">Oun et al., 2018</xref>; <xref ref-type="bibr" rid="B3">Azab et al., 2019</xref>).</p>
<p>As with all medications, patients may experience adverse events whilst undergoing chemotherapy treatment. Adverse events are unintended effects in response to a treatment therapy. Adverse events can be classified into different grades according to the common terminology criteria for adverse events (CTCAE) (<xref ref-type="bibr" rid="B52">Trotti et al., 2003</xref>; <xref ref-type="bibr" rid="B8">Cancer Therapy Evaluation Program, 2023</xref>). Adverse events that are grades 1 or 2 tend to be relatively mild to moderate and often do not require serious medical intervention. In the context of this paper, serious adverse events (SAEs) are adverse events that are classed as grade 3 or higher. For grade 3 events, patients may require hospitalisation and their quality of life may begin to be severely affected. In grade 4, the SAE is considered to be life threatening and the patient is in need of urgent medical care. SAE grade 5 is death caused by an adverse event (<xref ref-type="bibr" rid="B8">Cancer Therapy Evaluation Program, 2023</xref>). SAEs remain a critical challenge as many patients who experience SAEs, may be suspended from their medical treatment. This unfortunately increases the likelihood that their tumour may begin to thrive, proliferate and has the potential to metastasise. There are also potential ongoing costs for other medications that need to be used. Some SAEs also have no defined mechanism which is a challenge in terms of being able to predict which patients are at risk (<xref ref-type="bibr" rid="B21">Duncan et al., 2015</xref>). There is a need to identify which patients are likely to develop SAEs as this can aid in specialised monitoring of at-risk patients while they commence and maintain their prescribed treatment.</p>
<p>One of the most commonly experienced SAEs in patients who develop SCLC is neutropenia. This SAE belongs to the Blood and Lymphatic system Disorders adverse event group (<xref ref-type="bibr" rid="B28">Kishida et al., 2009</xref>; <xref ref-type="bibr" rid="B8">Cancer Therapy Evaluation Program, 2023</xref>). Neutropenia is a term used to describe low neutrophil levels. Patients with neutropenia are generally more susceptible to infections and sepsis (<xref ref-type="bibr" rid="B42">Nesher and Rolston, 2013</xref>; <xref ref-type="bibr" rid="B30">Kochanek et al., 2019</xref>). There are many studies which have identified risk factors for neutropenia through the use of machine learning (<xref ref-type="bibr" rid="B13">Cho et al., 2020</xref>; <xref ref-type="bibr" rid="B53">Ven&#xe4;l&#xe4;inen et al., 2021</xref>; <xref ref-type="bibr" rid="B55">Wiberg et al., 2021</xref>). Machine learning uses algorithms in order to uncover possible relationships between variables in a dataset. Supervised classification machine learning can be used to assess the relationship between different input features in order to predict a particular response, such as whether or not a patient may experience neutropenia (<xref ref-type="bibr" rid="B40">Nasteski, 2017</xref>). Example risk factors for neutropenia include age and low blood cell count (<xref ref-type="bibr" rid="B35">Lyman et al., 2014</xref>).</p>
<p>While the vast majority of SAEs experienced by SCLC patients who are treated with chemotherapy agents is neutropenia, there are other SAEs from different adverse event groups that patients may also be at risk of developing (<xref ref-type="bibr" rid="B34">Ludwig et al., 2014</xref>; <xref ref-type="bibr" rid="B38">McQuade et al., 2020</xref>). Moreover, many of the machine learning studies that are published for adverse events in general focus more on the comparison between patients who do not experience an adverse event (the control group) and those patients who do indeed develop the adverse event. Many machine learning classification algorithms are primarily used to determine two possible outcomes. However, SAEs can potentially have 4 different classifications, no SAE or grade 0, SAE grade 3, SAE grade 4 and SAE grade 5. Note, that there is no SAE grade 1 or 2 in order to avoid confusion with adverse events grade 1 and 2 which are not SAEs. The identification of not only which SAE group a patient is likely to be a member of, but also the grade, before a patient has commenced their chemotherapy treatment, would offer many benefits. For example, the patients who are identified as being at risk of developing a particular SAE group may receive closer monitoring and they could potentially be prescribed other medications to help combat the potential onset of a particular SAE group. For patients who are identified as at risk of developing a SAE grade 5, these patients may be given an alternative cancer treatment or have their dosages reduced, in order to mitigate the risk of more serious consequences, including death, due to the SAE.</p>
<p>The aims of this investigation are to identify those patients who are at risk of developing SAEs from commonly occurring SAE groups, as well as their classifications. Moreover, this study aims to highlight any predictive features that may make a patient susceptible to developing a particular SAE group.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<sec id="s2-1">
<title>2.1 Clinical data collection</title>
<p>Access to SCLC clinical trial data was obtained through Project Data Sphere (<xref ref-type="bibr" rid="B45">Project Data Sphere, 2023</xref>). Data from the following five clinical trial studies were used: NCT02499770, NCT00143455, NCT01439568, NCT00119613 and NCT00363415 (<xref ref-type="bibr" rid="B18">ClinicalTrials.gov:NCT02499770, 2020</xref>; <xref ref-type="bibr" rid="B15">ClinicalTrials.gov:NCT00143455, 2010</xref>; <xref ref-type="bibr" rid="B17">ClinicalTrials.gov:NCT01439568, 2019</xref>; <xref ref-type="bibr" rid="B14">ClinicalTrials.gov:NCT00119613, 2008</xref>; <xref ref-type="bibr" rid="B16">ClinicalTrials.gov:NCT00363415, 2009</xref>). These studies were selected based on the accessibility of the data and the inclusion of laboratory data. All of these studies had patients who were assigned to start a chemotherapy treatment. The data from these studies were merged together in order to analyse the occurrence of SAEs in SCLC patients. A total of 1,043 patients were eligible for the analysis. <xref ref-type="table" rid="T1">Table 1</xref> provides a summary of the baseline characteristics that were included in the analysis.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Summary baseline features that were included in the analysis.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Features</th>
<th align="left">Total N: 1,043</th>
<th align="left">No SAE N: 289</th>
<th align="left">Yes SAE N: 754</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td colspan="4" align="left">Demographic</td>
</tr>
<tr>
<td align="left">Age: below 45 Yrs (count (%))</td>
<td align="left">26 (2.5)</td>
<td align="left">6 (2.1)</td>
<td align="left">20 (2.7)</td>
</tr>
<tr>
<td align="left">Age: 45 to 49 Yrs (count (%))</td>
<td align="left">59 (5.7)</td>
<td align="left">28 (9.7)</td>
<td align="left">31 (4.1)</td>
</tr>
<tr>
<td align="left">Age: 50 to 54 Yrs (count (%))</td>
<td align="left">147 (14.1)</td>
<td align="left">49 (17)</td>
<td align="left">98 (13)</td>
</tr>
<tr>
<td align="left">Age: 55 to 59 Yrs (count (%))</td>
<td align="left">193 (18.5)</td>
<td align="left">58 (20.1)</td>
<td align="left">135 (17.9)</td>
</tr>
<tr>
<td align="left">Age: 60 to 64 Yrs (count (%))</td>
<td align="left">217 (20.8)</td>
<td align="left">55 (19)</td>
<td align="left">162 (21.5)</td>
</tr>
<tr>
<td align="left">Age: 65 to 69 Yrs (count (%))</td>
<td align="left">205 (19.7)</td>
<td align="left">54 (18.7)</td>
<td align="left">151 (20)</td>
</tr>
<tr>
<td align="left">Age: 70 to 74 Yrs (count (%))</td>
<td align="left">122 (11.7)</td>
<td align="left">26 (9)</td>
<td align="left">96 (12.7)</td>
</tr>
<tr>
<td align="left">Age: 75 to 79 Yrs (count (%))</td>
<td align="left">56 (5.4)</td>
<td align="left">9 (3.1)</td>
<td align="left">47 (6.2)</td>
</tr>
<tr>
<td align="left">Age: 80 or above Yrs (count (%))</td>
<td align="left">18 (1.7)</td>
<td align="left">4 (1.4)</td>
<td align="left">14 (1.9)</td>
</tr>
<tr>
<td align="left">Sex: Female (count (%))</td>
<td align="left">308 (29.5)</td>
<td align="left">88 (30.4)</td>
<td align="left">220 (29.2)</td>
</tr>
<tr>
<td align="left">Sex: Male (count (%))</td>
<td align="left">735 (70.5)</td>
<td align="left">201 (69.6)</td>
<td align="left">534 (70.8)</td>
</tr>
<tr>
<td align="left">Race: White (count (%))</td>
<td align="left">716 (68.6)</td>
<td align="left">232 (80.3)</td>
<td align="left">484 (64.2)</td>
</tr>
<tr>
<td align="left">Race: Black (count (%))</td>
<td align="left">11 (1.1)</td>
<td align="left">1 (0.3)</td>
<td align="left">10 (1.3)</td>
</tr>
<tr>
<td align="left">Race: Asian (count (%))</td>
<td align="left">62 (5.9)</td>
<td align="left">9 (3.1)</td>
<td align="left">53 (7)</td>
</tr>
<tr>
<td align="left">Race: Other (count (%))</td>
<td align="left">10 (1)</td>
<td align="left">4 (1.4)</td>
<td align="left">6 (0.8)</td>
</tr>
<tr>
<td align="left">Time since first diagnosis (Days) (mean, 95%CI)</td>
<td align="left">16.8 (15.9&#x2013;17.8)</td>
<td align="left">16.6 (14.9&#x2013;18.4)</td>
<td align="left">16.9 (15.8&#x2013;18.1)</td>
</tr>
<tr>
<td colspan="4" align="left">Laboratory Findings</td>
</tr>
<tr>
<td align="left">Haemoglobin (G/L) (mean, 95%CI)</td>
<td align="left">99.5 (96.1&#x2013;102.9)</td>
<td align="left">110.1 (104.5&#x2013;115.7)</td>
<td align="left">95.4 (91.2&#x2013;99.5)</td>
</tr>
<tr>
<td align="left">Neutrophils (10<sup>9</sup>/L) (mean, 95%CI)</td>
<td align="left">6.4 (6.1&#x2013;6.6)</td>
<td align="left">6.5 (6.1&#x2013;6.9)</td>
<td align="left">6.3 (6&#x2013;6.6)</td>
</tr>
<tr>
<td align="left">Platelets (10<sup>9</sup>/L) (mean, 95%CI)</td>
<td align="left">318.5 (310.5&#x2013;326.5)</td>
<td align="left">329.8 (314.4&#x2013;345.2)</td>
<td align="left">314.1 (304.7&#x2013;323.5)</td>
</tr>
<tr>
<td align="left">Leukocytes (10<sup>9</sup>/L) (mean, 95%CI)</td>
<td align="left">9.3 (9&#x2013;9.5)</td>
<td align="left">9.5 (9.2&#x2013;9.9)</td>
<td align="left">9.2 (8.9&#x2013;9.5)</td>
</tr>
<tr>
<td align="left">Creatinine (&#xb5;Mol/L) (mean, 95%CI)</td>
<td align="left">77.5 (76.1&#x2013;78.9)</td>
<td align="left">76.6 (74&#x2013;79.1)</td>
<td align="left">77.9 (76.3&#x2013;79.6)</td>
</tr>
<tr>
<td align="left">Lactate Dehydrogenase (U/L) (mean, 95%CI)</td>
<td align="left">598.1 (537.1&#x2013;659.1)</td>
<td align="left">630.7 (507.2&#x2013;754.2)</td>
<td align="left">585.1 (515.1&#x2013;655.1)</td>
</tr>
<tr>
<td align="left">Sodium (mMol/L) (mean, 95%CI)</td>
<td align="left">137.9 (137.4&#x2013;138.3)</td>
<td align="left">137.7 (136.9&#x2013;138.6)</td>
<td align="left">137.9 (137.4&#x2013;138.4)</td>
</tr>
<tr>
<td align="left">Total Bilirubin (&#xb5;Mol/L) (mean, 95%CI)</td>
<td align="left">8.8 (8.4&#x2013;9.2)</td>
<td align="left">8.6 (7.8&#x2013;9.3)</td>
<td align="left">8.9 (8.5&#x2013;9.4)</td>
</tr>
<tr>
<td align="left">Albumin (G/L) (mean, 95%CI)</td>
<td align="left">37.5 (37&#x2013;38)</td>
<td align="left">38 (37.1&#x2013;39)</td>
<td align="left">37.3 (36.7&#x2013;37.9)</td>
</tr>
<tr>
<td align="left">Alkaline Phosphatase (U/L) (mean, 95%CI)</td>
<td align="left">148.9 (138.9&#x2013;158.8)</td>
<td align="left">147.7 (133.2&#x2013;162.2)</td>
<td align="left">149.3 (136.7&#x2013;162)</td>
</tr>
<tr>
<td align="left">Aspartate Aminotransferase (U/L) (mean, 95%CI)</td>
<td align="left">35.3 (32.9&#x2013;37.8)</td>
<td align="left">34.5 (30&#x2013;39)</td>
<td align="left">35.7 (32.7&#x2013;38.7)</td>
</tr>
<tr>
<td align="left">Alanine Aminotransferase (U/L) (mean, 95%CI)</td>
<td align="left">34.9 (32.7&#x2013;37.2)</td>
<td align="left">34.7 (30.4&#x2013;39)</td>
<td align="left">35 (32.3&#x2013;37.7)</td>
</tr>
<tr>
<td colspan="4" align="left">Concomitant Medications</td>
</tr>
<tr>
<td align="left">Analgesic (count (%))</td>
<td align="left">353 (33.8)</td>
<td align="left">78 (27)</td>
<td align="left">275 (36.5)</td>
</tr>
<tr>
<td align="left">Blood Agents (count (%))</td>
<td align="left">98 (9.4)</td>
<td align="left">14 (4.8)</td>
<td align="left">84 (11.1)</td>
</tr>
<tr>
<td align="left">Anti-Inflammatory (count (%))</td>
<td align="left">322 (30.9)</td>
<td align="left">73 (25.3)</td>
<td align="left">249 (33)</td>
</tr>
<tr>
<td align="left">GI Tract (count (%))</td>
<td align="left">332 (31.8)</td>
<td align="left">76 (26.3)</td>
<td align="left">256 (34)</td>
</tr>
<tr>
<td align="left">Hypertension (count (%))</td>
<td align="left">215 (20.6)</td>
<td align="left">42 (14.5)</td>
<td align="left">173 (22.9)</td>
</tr>
<tr>
<td align="left">Respiratory (count (%))</td>
<td align="left">240 (23)</td>
<td align="left">51 (17.6)</td>
<td align="left">189 (25.1)</td>
</tr>
<tr>
<td align="left">Nitrate (count (%))</td>
<td align="left">23 (2.2)</td>
<td align="left">2 (0.7)</td>
<td align="left">21 (2.8)</td>
</tr>
<tr>
<td align="left">Diabetes (count (%))</td>
<td align="left">50 (4.8)</td>
<td align="left">12 (4.2)</td>
<td align="left">38 (5)</td>
</tr>
<tr>
<td align="left">Vaso acting (count (%))</td>
<td align="left">14 (1.3)</td>
<td align="left">4 (1.4)</td>
<td align="left">10 (1.3)</td>
</tr>
<tr>
<td align="left">Osteoporosis (count (%))</td>
<td align="left">27 (2.6)</td>
<td align="left">7 (2.4)</td>
<td align="left">20 (2.7)</td>
</tr>
<tr>
<td align="left">Brain and Mind (count (%))</td>
<td align="left">234 (22.4)</td>
<td align="left">50 (17.3)</td>
<td align="left">184 (24.4)</td>
</tr>
<tr>
<td align="left">Statin (count (%))</td>
<td align="left">63 (6)</td>
<td align="left">12 (4.2)</td>
<td align="left">51 (6.8)</td>
</tr>
<tr>
<td align="left">Gout (count (%))</td>
<td align="left">32 (3.1)</td>
<td align="left">11 (3.8)</td>
<td align="left">21 (2.8)</td>
</tr>
<tr>
<td align="left">Infections (count (%))</td>
<td align="left">100 (9.6)</td>
<td align="left">26 (9)</td>
<td align="left">74 (9.8)</td>
</tr>
<tr>
<td align="left">Cardiac (count (%))</td>
<td align="left">25 (2.4)</td>
<td align="left">3 (1)</td>
<td align="left">22 (2.9)</td>
</tr>
<tr>
<td align="left">Thyroid (count (%))</td>
<td align="left">29 (2.8)</td>
<td align="left">2 (0.7)</td>
<td align="left">27 (3.6)</td>
</tr>
<tr>
<td align="left">Cancer (count (%))</td>
<td align="left">21 (2)</td>
<td align="left">3 (1)</td>
<td align="left">18 (2.4)</td>
</tr>
<tr>
<td align="left">Muscle relaxant (count (%))</td>
<td align="left">11 (1.1)</td>
<td align="left">2 (0.7)</td>
<td align="left">9 (1.2)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="Tfn1">
<p>95 % CI, refers to the 95% confidence intervals. For continuous features the values inside the brackets are the 95% CIs. For the categorical features, the values inside the brackets are the number of patients who fall into a specific group, expressed as a percentage.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Only baseline features were included for the analysis in order to assess whether the model is able identify patients who are at risk of developing a particular SAE group and grade before they commence their treatment. Typically, age is often displayed as a continuous feature. However, for some of the clinical trials only an estimated age range was provided, thus in order to preserve as much data as possible all ages were based on ranges. For the concomitant medications, the groups were based on the main indication for each of the medications that were supplied. Many of the concomitant entries for the patients had missing entries for the intended indication of the concomitant medication. Where medications had multiple indications, they were placed in a separate group. An example of this is the nitrates group which refers to medications such as glyceryl trinitrate which can be used for cardiac therapy as well as blood pressure (<xref ref-type="bibr" rid="B25">Hashimoto and Kobayashi, 2003</xref>). Vaso acting medications such as pentoxifylline refers to treatments that also can be used to treat both blood pressure and cardiac therapy, however, they may have a different mechanism when compared to the nitrates class (<xref ref-type="bibr" rid="B25">Hashimoto and Kobayashi, 2003</xref>; <xref ref-type="bibr" rid="B44">Prasad and Lee, 2007</xref>). Features that had more than 80% of entries missing were excluded from the analysis. The correlation values for each of the features when compared to the onset of any SAE can be found in <xref ref-type="sec" rid="s11">Supplementary Table S1</xref> of the <xref ref-type="sec" rid="s11">Supplementary Materials</xref>.</p>
</sec>
<sec id="s2-2">
<title>2.2 SAE occurrence and common SAE groups</title>
<p>As mentioned in the introduction, SAE refers to an adverse event that is grade 3 or higher. Out of 1,043 patients, 754 patients experienced a SAE (<xref ref-type="table" rid="T1">Table 1</xref>), the most common SAE was neutropenia, which accounted for 37% of all the SAE occurrences during the prescribed chemotherapy treatment. The three most common SAE groups were Blood and Lymphatic Disorders (59.4% of entries), Infections and Infestations (7.3% of entries) and Metabolism and Nutrition Disorders (5.6% of entries).</p>
</sec>
<sec id="s2-3">
<title>2.3 Development of machine learning models</title>
<sec id="s2-3-1">
<title>2.3.1 Preparation of data for machine learning</title>
<p>Different analysis groups were formed based on the three common SAE groups. All of the analysis groups contained SAE information as well as the features (variables) that are presented in <xref ref-type="table" rid="T1">Table 1</xref>. In order to handle the multiclassification of different SAE grades and groups, a 1 vs. 1 approach was used. An example of the 1 vs. 1 approach would be patients who experienced a Blood and Lymphatic Disorder SAE grade 3 vs. patients who experienced a Blood and Lymphatic Disorder SAE grade 4. The model would then predict whether patients experienced grade 3 or grade 4. From the three common SAE groups a total of 12 analysis groups were formed. Analysis groups which resulted in less than 100 patients in total were excluded as it was deemed that there would be insufficient information available for the machine learning to make robust predictions. An example of one of the analysis groups that was excluded was Blood and Lymphatic Disorder SAE grade 5 vs. Infections and Infestations SAE grade 5.</p>
<p>While the Blood and Lymphatic SAE group accounted for 54.7% of SAE entries, other SAE groups (as well as higher grades) are likely to have fewer individuals who experienced that particular SAE group and grade. This would result in an imbalanced training set which could decrease the model&#x2019;s ability and performance. Many models that train on an imbalanced dataset will most likely predict the majority class as there are more instances present in the data set than for the minority class. There are several methods for handling imbalanced datasets, such as the inclusion of weights, costs, and sampling techniques (<xref ref-type="bibr" rid="B5">Blagus and Lusa, 2013</xref>; <xref ref-type="bibr" rid="B20">Dubey et al., 2014</xref>; <xref ref-type="bibr" rid="B50">Tao et al., 2019</xref>). The synthetic minority oversampling technique (SMOTE) was applied to the training set in order to balance the number of cases for both the majority and minority class. SMOTE creates new instances of the minority class, as well as reducing the number of entries in the majority class (<xref ref-type="bibr" rid="B5">Blagus and Lusa, 2013</xref>). The term &#x201c;class&#x201d; simply refers to the target variable which for all the analysis groups will be the SAE groups and grades that are compared to each other using the 1 vs. 1 approach. Other techniques were attempted such as using weights, costs, up-sampling, and down-sampling. Models that were trained with SMOTE, in this analysis always achieved higher performance scores when compared to the other techniques.</p>
<p>
<xref ref-type="table" rid="T2">Table 2</xref> shows the different subdivisions of the data that were used in order to develop the machine learning models.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Summary of the different analysis groups used to build the machine learning models.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Analysis group</th>
<th align="left">Original training set (80%)</th>
<th align="left">SMOTE training set</th>
<th align="left">Testing set (20%)</th>
<th align="left">Positive class % in SMOTE training set</th>
<th align="left">Positive class % in testing set</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Blood 0 vs. 3</td>
<td align="left">461</td>
<td align="left">1,596</td>
<td align="left">115</td>
<td align="left">57</td>
<td align="left">47</td>
</tr>
<tr>
<td align="left">Blood 0 vs. 4</td>
<td align="left">443</td>
<td align="left">1,442</td>
<td align="left">110</td>
<td align="left">43</td>
<td align="left">53</td>
</tr>
<tr>
<td align="left">Blood 3 vs. 4</td>
<td align="left">441</td>
<td align="left">1,463</td>
<td align="left">110</td>
<td align="left">43</td>
<td align="left">50</td>
</tr>
<tr>
<td align="left">Infec 0 vs. 3</td>
<td align="left">295</td>
<td align="left">455</td>
<td align="left">73</td>
<td align="left">43</td>
<td align="left">19</td>
</tr>
<tr>
<td align="left">Infec 0 vs. 4</td>
<td align="left">254</td>
<td align="left">140</td>
<td align="left">63</td>
<td align="left">43</td>
<td align="left">13</td>
</tr>
<tr>
<td align="left">Infec 0 vs. 5</td>
<td align="left">244</td>
<td align="left">77</td>
<td align="left">60</td>
<td align="left">43</td>
<td align="left">7</td>
</tr>
<tr>
<td align="left">Infec 3 vs. 4</td>
<td align="left">86</td>
<td align="left">154</td>
<td align="left">21</td>
<td align="left">43</td>
<td align="left">29</td>
</tr>
<tr>
<td align="left">Metab 0 vs. 3</td>
<td align="left">272</td>
<td align="left">322</td>
<td align="left">68</td>
<td align="left">43</td>
<td align="left">7</td>
</tr>
<tr>
<td align="left">Metab 0 vs. 4</td>
<td align="left">245</td>
<td align="left">77</td>
<td align="left">61</td>
<td align="left">43</td>
<td align="left">10</td>
</tr>
<tr>
<td align="left">Blood vs. Infec</td>
<td align="left">244</td>
<td align="left">238</td>
<td align="left">61</td>
<td align="left">43</td>
<td align="left">7</td>
</tr>
<tr>
<td align="left">Blood vs. Metab</td>
<td align="left">240</td>
<td align="left">119</td>
<td align="left">60</td>
<td align="left">43</td>
<td align="left">3</td>
</tr>
<tr>
<td align="left">Infect vs. Metab</td>
<td align="left">90</td>
<td align="left">245</td>
<td align="left">22</td>
<td align="left">43</td>
<td align="left">32</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Blood: Blood and Lymphatic Disorder group, Infec: Infections and Infestations SAE, group. Metab: Metabolism and Nutrition Disorder group. The numbers in the analysis group refer to the SAE, grade. Note that for SAE, 0, this refers to individuals who developed no SAE during the trial. For the last three analysis group, e.g., blood vs. infec, the SAE, grade was 3, for instance Blood group grade 3 vs. Infec group grade 3. SMOTE: synthetic minority oversampling technique. Positive class refers to the group that is on the right side of the vs., for instance in the first group the positive group is Blood group with SAE, grade 3.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The analysis groups were then split into an 80:20 split for the training and testing datasets. This decision was made so that the model was able to learn as much as possible from the training set. After splitting the data and transforming the training data through SMOTE, the K nearest neighbours (KNN) algorithm was then applied in order to predict values for any missing data entries. In short the KNN algorithm predicted the missing value entry based on the values of its corresponding neighbours (<xref ref-type="bibr" rid="B36">Malarvizhi and Thanamani, 2012</xref>). The features were then centred and normalised in order to minimise the likelihood that the model will favour particular features because they seem larger in absolute value when compared other features. Once these implementations were completed, a machine learning algorithm can be applied to the training data, in order to learn any intricate patterns between the features and the target variable.</p>
</sec>
<sec id="s2-3-2">
<title>2.3.2 AI implementation: extreme gradient boosting</title>
<p>In this analysis, the algorithm of choice used for the machine learning was the extreme gradient boosting (XGBOOST) algorithm. Other algorithms were tested such as random forest, decision trees and neural networks. However, XGBOOST yielded the best results in terms of resulting values for the area under the curve (AUC) and sensitivity rates. XGBOOST is a sequential gradient boosting algorithm developed by Tianqi Chen. This algorithm can be used for both classification and regression problems (<xref ref-type="bibr" rid="B10">Chen and Guestrin, 2016</xref>). In classification problems, the target variable is often presented as a 0 or a 1. A simplified example of how XGBOOST predicts the target variable is presented in the following.</p>
<p>In the example training data, the target variable is Blood and Lymphatic Disorders SAE grade 0 vs. grade 3. A &#x201c;0&#x201d; in the target variable would indicate patient did not develop any SAE whereas &#x201c;1&#x201d; would indicate a patient did develop a Blood and Lymphatic Disorder SAE grade 3. There are five patients in this example, three of them did develop Blood and Lymphatic Disorders SAE grade 3 and two of the patients did not. Therefore, three of the patients have &#x201c;1&#x201d; as their target label and the other patients have &#x201c;0&#x201d; as their target label.</p>
<p>The base XGBOOST model for binary classification will predict 0.5 for all patients. Based on this, residuals can be calculated in order to take into account the difference between the base predictions and the true target label. As an example, the residuals for the patients who did not develop SAE grade 3 would be &#x2212;0.5 (0&#x2013;0.5). Once the residuals were calculated then the similarity score can be obtained. This equation is given by:<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>S</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>p</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where, in this case, the previous probability refers to the base probability. The regularisation parameter &#x3bb; can be used to determine whether more branches (splits) should be developed for this model (tree pruning). Assuming that &#x3bb; is set to 1, the similarity score in this example is 0.111.</p>
<p>This similarity score is then compared to a new similarity score which has been formed based on the addition of a feature. For this example, the feature that was implemented was age. Patients who were younger than 50 years old fell into one group whereas those who were aged 50 or older were placed into a different group. The similarity scores for both of these groups were calculated. For simplicity, the similarity score for patients who are younger than 50 years old was 0.8 and the similarity score for the other group was 1. These two similarity scores were then compared to the previous score in order to calculate the gain.<disp-formula id="e2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>m</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>S</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>a</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>s</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>m</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>S</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>b</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>s</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
</p>
<p>The gains here are 0.689 and 0.889 respectively. As these values are positive this split is feasible. If the maximum number of splits has not been achieved then the model can continue to branch out and incorporate different features. However, in this example the maximum split has been achieved and thus the new predictions can now be formed. First the output value is calculated which is given by:<disp-formula id="e3">
<mml:math id="m3">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>V</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>E</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>p</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
<p>Note that the output values are now based on the age groups as well as the initial target variable. The equations for the new predictions and prediction probabilities are given below:<disp-formula id="e4">
<mml:math id="m4">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>w</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>g</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>O</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>v</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
<disp-formula id="e5">
<mml:math id="m5">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>w</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>s</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>w</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>w</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>where the learning rate (which impacts on the step size of each iteration) is often given by 0.3. With the new predictions any new iterations that are made will build upon the probabilities from the previous iterations with the aim of reducing the residuals to 0.</p>
<p>In this analysis, the negative log likelihood of each of the iterations is used to measure the performance of each iteration. The negative log likelihood takes into account the new prediction probabilities and the actual true labels (0 and 1). Therefore, by reducing the negative log likelihood the residuals of the model are also reduced and the accuracy of the model is improved.</p>
</sec>
<sec id="s2-3-3">
<title>2.3.3 Optimal parameter values for the XGBOOST algorithm</title>
<p>Optimal values for the XGBOOST algorithm&#x2019;s parameters can be obtained through hyper parameter tuning and cross validation. Such parameters include: the maximum number of splits (tree depth) and the number of features that can be used in a single iteration (known as colsample by tree in R). In hyper parameter tuning, optimal values can be found through different search methods. In this analysis, the method for finding optimal values was based on a random search, thus random combination values were used and those that yielded the best sensitivity and specificity rates were used for the training of the model. In cross validation, the maximum number of iterations was also established. This was based on splitting the data into 5 subsets and testing within the training data, as to whether the model would be able to predict accurate responses or not. The iteration which had the lowest negative log likelihood value based upon the analysis of the test data in the cross validation, was selected as the optimal iteration number. With all the parameter values selected, the model could efficiently be built and was applied to the testing data.</p>
</sec>
<sec id="s2-3-4">
<title>2.3.4 Preparation of the testing data</title>
<p>The testing data does not undergo any imbalance transformation. However, the testing data do undergo missing value imputation and normalisation using the same processes as its training data counterpart. The model trained on the training data set was then used on the test data to predict the target variables for the patients.</p>
</sec>
</sec>
<sec id="s2-4">
<title>2.4 Feature importance using shapley additive explanation values</title>
<p>Although XGBOOST trees can be displayed to highlight which features influenced the model&#x2019;s decision to predict a particular output, the reality is that for complex models, there may be many trees which have multiple branches with different threshold values. This can make the overall output diagram challenging to interpret. An alternative approach is to compute the Shapley additive explanatory (SHAP) values, in order to assess which features contributed the most to the model predictions (<xref ref-type="bibr" rid="B24">Hart, 1989</xref>; <xref ref-type="bibr" rid="B33">Li et al., 2020</xref>). SHAP values are based on Game Theory, where each feature value has a contribution score to the overall model&#x2019;s response. This contribution score is based on the impact a specific feature value has on the model predictions and, the impact the feature value has in combination with other feature values, on the model predictions. The contribution score as well as the initial model&#x2019;s bias (0.5) are summed to yield final predicted score for each patient. Using the Blood and Lymphatic Disorders group SAE grade 0 vs. grade 3 as an example, the SHAP values that are lower than 0 would denote a decreased risk of developing SAEs. SHAP values that are greater than 0 denote an increased risk of developing SAE grade 3.</p>
<p>All of the analysis was performed using the software tool R using the following packages for the model implementation: xgboost (XGBOOST algorithm, training the data, cross validation), caret (splitting the data, missing data implementation and normalisation), mlr (hyper-parameter tuning), RANN (necessary for knn implementation), Dmwr (SMOTE implementation), and Proc (ROC curve analysis) (<xref ref-type="bibr" rid="B46">R Core Team, 2023</xref>; <xref ref-type="bibr" rid="B11">Chen, 2023</xref>; <xref ref-type="bibr" rid="B32">Kuhn, 2023</xref>; <xref ref-type="bibr" rid="B4">Bischl, 2016</xref>; <xref ref-type="bibr" rid="B2">Arya et al., 2019</xref>; <xref ref-type="bibr" rid="B51">Torgo, 2010</xref>; <xref ref-type="bibr" rid="B47">Robin, 2011</xref>). The relevant codes used for this analysis can be found at: <ext-link ext-link-type="uri" xlink:href="https://github.com/LindaWanika/SCLC-common-SAE-groups">https://github.com/LindaWanika/SCLC-common-SAE-groups</ext-link>.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<sec id="s3-1">
<title>3.1 Optimal iteration number for each of the models</title>
<p>
<xref ref-type="fig" rid="F1">Figure 1</xref> visualises the cross-validation process for each of the models.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Performance of each of the models during the cross-validation process. <bold>(A)</bold> Blood Grade 0 vs Grade 3 model. <bold>(B)</bold> Blood Grade 0 vs Grade 4 model. <bold>(C)</bold> Blood Grade 3 vs Grade 4 model. <bold>(D)</bold> Infec Grade 0 vs Grade 3 model. <bold>(E)</bold> Infec Grade 0 vs Grade 4 model. <bold>(F)</bold> Infec Grade 0 vs Grade 5 model. <bold>(G)</bold> Infec Grade 3 vs Grade 4 model. <bold>(H)</bold> Metab Grade 0 vs Grade 3 model. <bold>(I)</bold> Metab Grade 0 vs Grade 4 model. <bold>(J)</bold> Blood vs Infec model. <bold>(K)</bold> Blood vs Metab model. <bold>(L)</bold> Infec vs Metab model. A black line refers to the training log loss and a red line refers to the test log loss evaluation. Note that the term &#x201c;test&#x201d; does not refer to the testing data set but rather the cross-validation test data. Blood: Blood and Lymphatic Disorder group, Infec: Infections and Infestations SAE group. Metab: Metabolism and Nutrition Disorder group. Log loss refers to the negative log likelihood.</p>
</caption>
<graphic xlink:href="fdsfr-03-1267623-g001.tif"/>
</fig>
<p>The Blood and Lymphatic Disorders group, SAE grade 0 vs. grade 3 model (Blood grade 0 vs. grade 3), appears to be the only model in the cross-validation process where the log loss value is able to reach to 0 for both the training and testing evaluation (see <xref ref-type="fig" rid="F1">Figure 1A</xref>). The Infections and Infestation group, SAE grade 0 vs. grade 5 (Infec grade 0 vs. grade 5) has the highest negative loglikelihood (log loss) value of 0.7 even after the ideal iteration number has been given (<xref ref-type="fig" rid="F1">Figure 1F</xref>). In most of the model evaluations, it is apparent that the training evaluation performs better than the testing evaluation, moreover, most of the training evaluations are able to achieve a log loss of approximately 0. A summary of all the parameter values that were chosen for each of the models based on the random search during the hyper tuning process can be found in <xref ref-type="sec" rid="s11">Supplementary Table S2</xref> in the <xref ref-type="sec" rid="s11">Supplementary Materials</xref>. <xref ref-type="table" rid="T3">Table 3</xref> summarises the optimal iteration number for each of the models.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Summary of the best iterations for each of the models based on the cross-validation test evaluation.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Models number&#x2a;</th>
<th align="left">1</th>
<th align="left">2</th>
<th align="left">3</th>
<th align="left">4</th>
<th align="left">5</th>
<th align="left">6</th>
<th align="left">7</th>
<th align="left">8</th>
<th align="left">9</th>
<th align="left">10</th>
<th align="left">11</th>
<th align="left">12</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Best iteration</td>
<td align="left">21</td>
<td align="left">49</td>
<td align="left">38</td>
<td align="left">34</td>
<td align="left">24</td>
<td align="left">6</td>
<td align="left">61</td>
<td align="left">27</td>
<td align="left">41</td>
<td align="left">42</td>
<td align="left">26</td>
<td align="left">29</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The model number refers to the order that they appear in <xref ref-type="fig" rid="F1">Figure 1</xref>. For example, model 1 is Blood grade 0 vs. 3, model 2 is Blood grade 0 vs. grade 4, etc.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The Infec grade 0 vs. grade 5 model has the least number of iterations needed whereas the Infections and Infestations group SAE grade 3 vs. grade 4 (Infec grade 3 vs. grade 4), has the highest iteration number (<xref ref-type="table" rid="T3">Table 3</xref>).</p>
</sec>
<sec id="s3-2">
<title>3.2 Comparisons of the average test runs</title>
<p>
<xref ref-type="fig" rid="F2">Figure 2</xref> displays the average receiver operating haracteristic (ROC) curves for each of the models.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>ROC curves for each of the models based on the average model run. <bold>(A)</bold> Blood SAE group models. <bold>(B)</bold> Infec SAE group models. <bold>(C)</bold> Metab SAE group models. <bold>(D)</bold> Grade 3 SAE group models. Blood: Blood and Lymphatic Disorder group, Infec: Infections and Infestations SAE group. Metab: Metabolism and Nutrition Disorder group.</p>
</caption>
<graphic xlink:href="fdsfr-03-1267623-g002.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="F2">Figure 2</xref> shows the average AUC, sensitivity and specificity scores for the testing data based on 100 model runs. <xref ref-type="table" rid="T4">Table 4</xref> summarises all the scores.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Average results from all 100 testing runs for each model.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Models</th>
<th align="left">AUC</th>
<th align="left">Sensitivity</th>
<th align="left">Specificity</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Blood Group SAE 0 vs. 3</td>
<td align="left">1.000</td>
<td align="left">0.774</td>
<td align="left">0.896</td>
</tr>
<tr>
<td align="left">Blood Group SAE 0 vs. 4</td>
<td align="left">0.700</td>
<td align="left">0.594</td>
<td align="left">0.406</td>
</tr>
<tr>
<td align="left">Blood Group SAE 3 vs. 4</td>
<td align="left">0.651</td>
<td align="left">0.575</td>
<td align="left">0.575</td>
</tr>
<tr>
<td align="left">Infec Group SAE 0 vs. 3</td>
<td align="left">0.701</td>
<td align="left">0.660</td>
<td align="left">0.538</td>
</tr>
<tr>
<td align="left">Infec Group SAE 0 vs. 4</td>
<td align="left">0.700</td>
<td align="left">0.671</td>
<td align="left">0.527</td>
</tr>
<tr>
<td align="left">Infec Group SAE 0 vs. 5</td>
<td align="left">0.707</td>
<td align="left">0.696</td>
<td align="left">0.489</td>
</tr>
<tr>
<td align="left">Infec Group SAE 3 vs. 4</td>
<td align="left">0.716</td>
<td align="left">0.648</td>
<td align="left">0.559</td>
</tr>
<tr>
<td align="left">Metab Group SAE 0 vs. 3</td>
<td align="left">0.700</td>
<td align="left">0.682</td>
<td align="left">0.514</td>
</tr>
<tr>
<td align="left">Metab Group SAE 0 vs. 4</td>
<td align="left">0.707</td>
<td align="left">0.683</td>
<td align="left">0.521</td>
</tr>
<tr>
<td align="left">Blood vs. Infec Group (SAE 3)</td>
<td align="left">0.700</td>
<td align="left">0.684</td>
<td align="left">0.513</td>
</tr>
<tr>
<td align="left">Blood vs. Metab Group (SAE 3)</td>
<td align="left">0.709</td>
<td align="left">0.699</td>
<td align="left">0.507</td>
</tr>
<tr>
<td align="left">Infec vs. Metab Group (SAE 3)</td>
<td align="left">0.707</td>
<td align="left">0.635</td>
<td align="left">0.563</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>AUC: area under the curve. Blood: Blood and Lymphatic Disorder group, infec: Infections and Infestations SAE, group. Metab: Metabolism and Nutrition Disorder group.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>In both <xref ref-type="fig" rid="F2">Figure 2A</xref> and in <xref ref-type="table" rid="T4">Table 4</xref>, the Blood grade 0 vs. grade 3 model, has the highest AUC and on average the highest sensitivity and specificity rates. The Blood group grade 3 vs. grade 4 has, on average have the lowest AUC, and sensitivity rate (<xref ref-type="fig" rid="F2">Figure 2A</xref>). However, the Blood grade 0 vs. grade 4, on average has the lowest specificity rate at 0.406 (<xref ref-type="table" rid="T3">Table 3</xref>). The other models in comparison, appear to have similar AUC scores on average, at approximately 0.7 (<xref ref-type="fig" rid="F2">Figures 2B&#x2013;D</xref>).</p>
</sec>
<sec id="s3-3">
<title>3.3 Comparisons of the best test runs</title>
<p>
<xref ref-type="fig" rid="F3">Figure 3</xref> displays the best ROC curves for each of the models.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>ROC curves for each of the models based on the best test run. <bold>(A)</bold> Blood SAE group models. <bold>(B)</bold> Infec SAE group models. <bold>(C)</bold> Metab SAE group models. <bold>(D)</bold> Grade 3 SAE group models. Blood: Blood and Lymphatic Disorder group, Infec: Infections and Infestations SAE group. Metab: Metabolism and Nutrition Disorder group.</p>
</caption>
<graphic xlink:href="fdsfr-03-1267623-g003.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="F3">Figure 3</xref> shows that the sensitivity and specificity rates often fluctuate. <xref ref-type="table" rid="T5">Table 5</xref> summarises all the scores for the best test runs for each of the models.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Best test runs for each model.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Models</th>
<th align="left">AUC</th>
<th align="left">Sensitivity</th>
<th align="left">Specificity</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Blood Group SAE 0 vs. 3</td>
<td align="left">1.000</td>
<td align="left">0.840</td>
<td align="left">0.964</td>
</tr>
<tr>
<td align="left">Blood Group SAE 0 vs. 4</td>
<td align="left">0.759</td>
<td align="left">0.621</td>
<td align="left">0.635</td>
</tr>
<tr>
<td align="left">Blood Group SAE 3 vs. 4</td>
<td align="left">0.714</td>
<td align="left">0.606</td>
<td align="left">0.606</td>
</tr>
<tr>
<td align="left">Infec Group SAE 0 vs. 3</td>
<td align="left">0.851</td>
<td align="left">0.780</td>
<td align="left">0.566</td>
</tr>
<tr>
<td align="left">Infec Group SAE 0 vs. 4</td>
<td align="left">0.816</td>
<td align="left">0.771</td>
<td align="left">0.582</td>
</tr>
<tr>
<td align="left">Infec Group SAE 0 vs. 5</td>
<td align="left">0.799</td>
<td align="left">0.828</td>
<td align="left">0.621</td>
</tr>
<tr>
<td align="left">Infec Group SAE 3 vs. 4</td>
<td align="left">0.933</td>
<td align="left">0.795</td>
<td align="left">0.618</td>
</tr>
<tr>
<td align="left">Metab Group SAE 0 vs. 3</td>
<td align="left">0.794</td>
<td align="left">0.768</td>
<td align="left">0.521</td>
</tr>
<tr>
<td align="left">Metab Group SAE 0 vs. 4</td>
<td align="left">0.845</td>
<td align="left">0.806</td>
<td align="left">0.533</td>
</tr>
<tr>
<td align="left">Blood vs. Infec Group (SAE 3)</td>
<td align="left">0.794</td>
<td align="left">0.770</td>
<td align="left">0.519</td>
</tr>
<tr>
<td align="left">Blood vs. Metab Group (SAE 3)</td>
<td align="left">0.931</td>
<td align="left">0.910</td>
<td align="left">0.514</td>
</tr>
<tr>
<td align="left">Infec vs. Metab Group (SAE 3)</td>
<td align="left">0.819</td>
<td align="left">0.708</td>
<td align="left">0.597</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>AUC: area under the curve. Blood: Blood and Lymphatic Disorder group, infec: Infections and Infestations SAE, group. Metab: Metabolism and Nutrition Disorder group.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Similar to the values generated for the average run, the Blood group grade 0 vs. grade 3 also had the best AUC sensitivity and specificity rates (<xref ref-type="fig" rid="F3">Figure 3A</xref>; <xref ref-type="table" rid="T5">Table 5</xref>). In the Infections and Infestations group analysis (<xref ref-type="fig" rid="F3">Figure 3B</xref>), the average AUC score for all the models is 0.8 with the grade 3 vs. grade 4 group having the highest AUC at 0.933. For the Metabolism and Nutrition Disorders group analysis (<xref ref-type="fig" rid="F3">Figure 3C</xref>), the average AUC is also 0.8 and both of the models yield a higher sensitivity rate than specificity (<xref ref-type="table" rid="T5">Table 5</xref>). In the combinations group analysis (<xref ref-type="fig" rid="F3">Figure 3D</xref>), the Blood vs. Metabolism analysis yields the highest AUC and sensitivity rates out of all the models in general with a score of 0.910.</p>
<p>The confusion matrices for both the average test runs and the best testing runs can be found in <xref ref-type="sec" rid="s11">Supplementary Tables S3, S4</xref> in the <xref ref-type="sec" rid="s11">Supplementary Materials</xref>.</p>
</sec>
<sec id="s3-4">
<title>3.4 Feature analysis for all of the models</title>
<p>
<xref ref-type="fig" rid="F4">Figure 4</xref> provides the SHAP plots for the Blood and Lymphatic Disorders group SAE models.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>SHAP plots displaying the top five features for the Blood group SAE models. <bold>(A)</bold> Grade 0 vs Grade 3 model. <bold>(B)</bold> Grade 0 vs Grade 4 model. <bold>(C)</bold> Grade 3 vs Grade 4 model. Grey dots refer to feature values that are below the lower quartile range (LQR), blue dots refer to the feature values that fall between LQR and the mean. Orange dots refer to the feature values that fall between the mean and upper quartile range (UQR). Red dots refer to values that are above the UQR.</p>
</caption>
<graphic xlink:href="fdsfr-03-1267623-g004.tif"/>
</fig>
<p>For the grade 0 vs. grade 3 model, patients who were over 80 years old, had high total bilirubin, lactate dehydrogenase and alkaline phosphatase levels, obtained SHAP values that were below 0 (<xref ref-type="fig" rid="F4">Figure 4A</xref>). In <xref ref-type="fig" rid="F4">Figure 4B</xref>, female patients, low platelet, and haemoglobin levels as well as high creatinine levels yielded SHAP values that were above 0 for grade 0 vs. grade 4. In <xref ref-type="fig" rid="F4">Figure 4C</xref>, patients who were female and had a higher total bilirubin level obtained higher SHAP values, whereas patients who had higher creatinine levels obtained SHAP values less than 0.</p>
<p>
<xref ref-type="fig" rid="F5">Figure 5</xref> provides the SHAP plots for the Infections and Infestations group SAE models.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>SHAP plots displaying the top five features for the Infec group SAE models. <bold>(A)</bold> Grade 0 vs Grade 3 model. <bold>(B)</bold> Grade 0 vs Grade 4 model. <bold>(C)</bold>: Grade 0 vs Grade 5 model. <bold>(D)</bold> Grade 3 vs Grade 4 model. Grey dots refer to feature values that are below the lower quartile range (LQR), blue dots refer to the feature values that fall between LQR and the mean. Orange dots refer to the feature values that fall between mean the and upper quartile range (UQR). Red dots refer to values that are above the UQR. For Grade 0 vs Grade 5, only three features were used to perform the predictions.</p>
</caption>
<graphic xlink:href="fdsfr-03-1267623-g005.tif"/>
</fig>
<p>For the grade 0 vs. grade 3 model, low haemoglobin levels, patients who were under 45 years old and low sodium levels were associated with SHAP values that are above 0 (<xref ref-type="fig" rid="F5">Figure 5A</xref>). Low haemoglobin levels are also associated with SHAP values that are less than 0 for the grade 0 vs. grade 4, grade 0 vs. grade 5 and grade 3 vs. 4 models (<xref ref-type="fig" rid="F5">Figure 5</xref>). High alkaline phosphate levels are associated with grade 5 and only three features were used in total for the prediction (<xref ref-type="fig" rid="F5">Figure 5C</xref>). Respiratory medications and high total bilirubin and leukocytes levels are associated with higher SHAP values (<xref ref-type="fig" rid="F5">Figure 5B</xref>).</p>
<p>
<xref ref-type="fig" rid="F6">Figure 6</xref> provides the SHAP plots for the Metabolism and Nutrition Disorders group SAE models.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>SHAP plots displaying the top five features for the Metab group SAE models. <bold>(A)</bold> Grade 0 vs Grade 3 model. <bold>(B)</bold> Grade 0 vs Grade 4 model. Grey dots refer to feature values that are below the lower quartile range (LQR), blue dots refer to the feature values that fall between LQR and the mean. Orange dots refer to the feature values that fall between the mean and upper quartile range (UQR). Red dots refer to values that are above the UQR.</p>
</caption>
<graphic xlink:href="fdsfr-03-1267623-g006.tif"/>
</fig>
<p>Low albumin levels are associated with higher SHAP values (<xref ref-type="fig" rid="F6">Figure 6</xref>). High leukocyte levels in grade 0 vs. grade 3 are associated with low SHAP values (<xref ref-type="fig" rid="F6">Figure 6A</xref>). For both models, lower bilirubin levels are associated with SHAP values below 0.</p>
<p>
<xref ref-type="fig" rid="F7">Figure 7</xref> provides the SHAP plots for the comparison SAE groups.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>SHAP plots displaying the top five features for the grade 3 SAE group models. <bold>(A)</bold> Blood vs Infec model. <bold>(B)</bold> Blood vs Metab model. <bold>(C)</bold> Infec vs Metab model. Grey dots refer to feature values that are below the lower quartile range (LQR), blue dots refer to the feature values that fall between LQR and the mean. Orange dots refer to the feature values that fall between the mean and upper quartile range (UQR). Red dots refer to values that are above the UQR.</p>
</caption>
<graphic xlink:href="fdsfr-03-1267623-g007.tif"/>
</fig>
<p>For the Blood and Lymphatic Disorders group vs. Infections and Infestations group, high haemoglobin and sodium levels are associated with SHAP values below 0, whereas high neutrophils are associated with higher SHAP values (<xref ref-type="fig" rid="F7">Figure 7A</xref>). For the Blood and Lymphatic Disorders group vs. Metabolism and Nutrition Disorders group, patients who had respiratory medications, high platelet levels and patients who were aged between 65 and 69&#xa0;years were associated with high SHAP values, low neutrophils were associated lower SHAP values. In <xref ref-type="fig" rid="F7">Figure 7C</xref>, high alkaline phosphatase and haemoglobin were associated with high SHAP values whereas high leukocytes and total bilirubin and patients who are female were associated with low SHAP values.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<p>Based on the average model analysis, most of the models were able to correctly identify a higher number of patients who fell into the more severe group of the 1 vs. 1 analysis group compared to the number of patients who fell into the less severe group. This is evidenced in <xref ref-type="table" rid="T4">Table 4</xref>, with most of the average models having a higher sensitivity rate than specificity rate. Sensitivity rates measures the &#x201c;true positive rates&#x201d;, i.e., classes which fall to the right side of the vs. group, whereas specificity rates measure the &#x201c;true negative rate&#x201d;, which are classes which fall on the left side of the vs. group. In a real-world setting, this outcome is more beneficial as misclassifying a patient as at risk of developing SAE grade 4 when in reality they are not at risk of developing any SAE, is a better outcome than misclassifying a patient as not at risk of developing any SAE, when in fact the patient is at risk of developing SAE grade 4. However, for the comparison models between different groups, both sides in the analysis group have the same level of severity. A possible reason as to why more patients were predicted as at risk in the &#x201c;positive&#x201d; group could be that the model was trained on a training set where 43% of cases were positive, however the actual test set had less than 50% of &#x201c;positive&#x201d; cases for the Infections and Infestations group, the Metabolism and Nutrition Disorder group and the comparison models between different groups (<xref ref-type="table" rid="T2">Table 2</xref>). It is possible that the model predicted more positive cases simply because it assumes that more positive cases should exist.</p>
<p>In <xref ref-type="table" rid="T5">Table 5</xref>, the best test runs for each of the models were able to correctly classify at least 60% of patients who fell into the positive group and at least 50% of patients who fell into the negative group. By far the best model was the Blood and Lymphatic Disorder group SAE grade 0 vs. grade 3 which achieved, on average, a sensitivity rate of 0.774 and a specificity rate of 0.896 (<xref ref-type="table" rid="T4">Table 4</xref>). The best run for this model, in particular, achieved a sensitivity score of 0.840 and a specificity rate of 0.964 (<xref ref-type="table" rid="T5">Table 5</xref>). It is important to note that the sensitivity and specificity scores presented are the mean and not the maximum rates. Other models that achieved high predictive scores (based on the best test runs) were the Infections and Infestations group SAE grade 0 vs. grade 5 and the Blood and Lymphatic Disorder group vs. the Metabolism and Nutrition Disorder group (<xref ref-type="fig" rid="F2">Figures 2</xref>, <xref ref-type="fig" rid="F3">3</xref>; <xref ref-type="table" rid="T5">Table 5</xref>). While the Infections and Infestations group SAE grade 3 vs. grade 4 model achieved a higher AUC score (<xref ref-type="table" rid="T4">Tables 4</xref>, <xref ref-type="table" rid="T5">5</xref>), the grade 0 vs. 5 achieved higher sensitivity and specificity rates. Moreover, this model, in particular, had fewer iterations and thus was also simpler than the grade 3 vs. grade 4 (<xref ref-type="table" rid="T3">Table 3</xref>). For the Infections and Infestation group SAE grade 0 vs. grade 5, the model was able to correctly identify 82% of patients who were at risk of developing SAE grade 5 and identified 62% of patients who are not at risk of developing SAEs (<xref ref-type="table" rid="T5">Table 5</xref>). The Blood and Lymphatic Disorder group vs. the Metabolism and Nutrition Disorder group model was able to identify 91% of patients who were at risk of developing a Metabolism and Nutrition Disorder group SAE grade 3 and identified 51% of patients who were at risk of developing a Blood and Lymphatic Disorder group SAE grade 3.</p>
<p>The SHAP plots for each of the models provides a simple and clear overview of the top features that contributed the most for each model prediction. The SHAP plots that are presented here are based on the best test runs for each model. Values that are above 0 indicate that patients are more at risk of being in the positive class, whereas values that are less than 0 indicate that patients are either not at risk of developing SAE (for models that compare grade 0 to another grade), or the negative class. Note that the classification of the feature values (lower quartile, mean, upper quartile) does not necessarily denote that the values are abnormal readings (see <xref ref-type="sec" rid="s11">Supplementary Table S17&#x2013;S28</xref> for the summary statistics of each of the features for the respective models in the <xref ref-type="sec" rid="s11">Supplementary Materials</xref>).</p>
<p>For the Blood and Lymphatic Disorders group analysis, patients with low haemoglobin levels were associated with being at risk of developing Blood group SAE grade 4, whereas low total bilirubin levels were associated with patients not being at risk of developing SAE (<xref ref-type="fig" rid="F4">Figure 4</xref>). Low haemoglobin can also be associated with anaemia and other Blood Lymphatic Disorders (<xref ref-type="bibr" rid="B39">Mercadante et al., 2000</xref>; <xref ref-type="bibr" rid="B49">Rusciano et al., 2008</xref>). High bilirubin levels can also be associated with the breakdown of haemoglobin which may result in decreased levels of haemoglobin (<xref ref-type="fig" rid="F4">Figure 4C</xref>) (<xref ref-type="bibr" rid="B27">Kao et al., 2012</xref>). For the first model (grade 0 vs. grade 3), patients who were aged 80 or above were deemed as less likely to develop Blood SAE. In <xref ref-type="table" rid="T1">Table 1</xref>, the majority of patients are not in this age group range and have a higher incidence of SAEs, in general. This could potentially explain as to why the model highlighted this as an important feature. For the second model (grade 0 vs. grade 4) low platelet levels were also associated with a higher risk of grade 4 which is more in keeping with what is known, as thrombocytopenia is a common SAE and is often associated with chemotherapy treatment (<xref ref-type="bibr" rid="B54">Weycker et al., 2019</xref>).</p>
<p>Similar to the Blood and Lymphatic Disorders group, in the Infections and Infestations group (<xref ref-type="fig" rid="F5">Figure 5</xref>) low haemoglobin was associated with grade 3 severity, and in some instances grade 4 (<xref ref-type="fig" rid="F5">Figure 5B</xref>), and low bilirubin is associated with patients less at risk of developing an SAE. Lymphatic disorders can make patients more suspectable to infections as the levels of lymphocytes decrease (<xref ref-type="bibr" rid="B22">Francis et al., 2013</xref>). For grade 0 vs. grade 4, patients who were taking respiratory medications and had high leukocyte levels were also more at risk of developing grade 4. A possible reason for the respiratory link to infections could be that, prior to the treatment, these patients may have been prescribed cough supplements or other respiratory medications for the treatment of respiratory conditions caused by infections (<xref ref-type="bibr" rid="B48">Rosen, 2006</xref>). High leukocytes also tend to be present during inflammation which may have been caused by an infection (<xref ref-type="bibr" rid="B12">Chmielewski and Strzelec, 2018</xref>). Higher alkaline phosphatase levels are associated with patients who are at risk of developing grade 5 (<xref ref-type="fig" rid="F5">Figure 5C</xref>). High alkaline phosphatase levels can be associated with liver disorders which can also include infections (<xref ref-type="bibr" rid="B6">Blayney et al., 2008</xref>). For the Blood and Lymphatic Disorders group, patients who were female seem to have a higher susceptibility based on the SHAP values, to developing SAEs, even though the majority of patients who developed SAEs, in general were male (<xref ref-type="table" rid="T1">Table1</xref>). Some studies have found that females are more suspectable to infections and anaemia, as well as other blood conditions, which may be a possible reason for this difference (<xref ref-type="bibr" rid="B41">Nazir et al., 2011</xref>). Although <xref ref-type="fig" rid="F5">Figure 5D</xref> shows females as also being a contributing factor for the Infections and Infestations group, it should be noted that this feature does not appear in any of the other <xref ref-type="fig" rid="F5">Figure 5</xref> plots.</p>
<p>The Metabolism and Nutrition Disorder SAE group has a smaller test set compared to the previous analysis groups (<xref ref-type="table" rid="T2">Table 2</xref>). Low albumin levels were associated with patients being at risk of grades 3 and 4 (<xref ref-type="fig" rid="F6">Figure 6</xref>). Low albumin levels have been linked with hepatic disorders, which can potentially impact the metabolism process (<xref ref-type="bibr" rid="B37">Matthewson et al., 1986</xref>; <xref ref-type="bibr" rid="B9">Carvalho and Machado, 2018</xref>). This may also explain the relationship between bilirubin and the occurrence of Metabolism and Nutrition Disorder SAEs (<xref ref-type="bibr" rid="B37">Matthewson et al., 1986</xref>; <xref ref-type="bibr" rid="B23">Hamoud et al., 2018</xref>). Decreased or increased levels of minerals in the body are often associated with Metabolism and Nutrition Disorders (<xref ref-type="bibr" rid="B8">Cancer Therapy Evaluation Program, 2023</xref>).</p>
<p>In <xref ref-type="fig" rid="F7">Figure 7</xref>, low haemoglobins are more associated with the Infections and Infestations group SAE grade 3 when compared to the other SAE groups. Low platelet and neutrophils levels are more indicative of Blood and Lymphatic Disorders SAE grade 3 when compared to the other groups. High alkaline phosphatase seem are associated with patients at risk of developing Metabolism and Nutrition Disorders group SAE grade 3 when compared to the Infections and Infestations group SAE grade 3 groups, as well as higher platelet levels when compared to the Blood and Lymphatic Disorders group.</p>
<p>The application of machine learning to this dataset has enabled the identification of trends between common SAE groups and features which may have been overlooked through the application of statistical methods alone. During the training process, XGBOOST is able to analyse multiple features and split these features accordingly in order to determine adequate feature thresholds which would impact on the predictability of common SAE group&#x2019;s onset, within a short time frame (minutes). A significant amount of time would be required in order to achieve the same outcome using traditional statistical methods. Moreover, many of the traditional statistical methods rely on significant correlations between features and the predictive target. In <xref ref-type="sec" rid="s11">Supplementary Table S1</xref> (<xref ref-type="sec" rid="s11">Supplementary Materials</xref>), only six features have a <italic>p</italic>-value of less than 0.05 when associated with the onset of SAE. Total bilirubin and alkaline phosphatase are two features which were identified as common risk factors for the onset of common SAE groups however, both of them have correlation values of less than 0.1 and <italic>p</italic>-values greater than 0.5.</p>
<p>While machine learning does have advantages in supporting model predictions, it is important to note that in order to achieve optimal results, good quality data are needed, i.e., large in quantity and a balanced dataset with minimal missing values. For adverse event onset the data are usually imbalanced given that these occurrences are generally minimal and sometimes rare. In clinical trials, and to a greater extent with real world data, missing entries are common. As mentioned in the methods section, features that had up to 80% missing entries were included in the analysis. The features that were excluded may have been significant for the onset of common SAE groups, however, it is most likely that XGBOOST would have dismissed these features. In addition to this, while KNN was used to impute the missing values, the values selected may not have been adequate. In other words, it is possible that a clinician may see the value of one feature and be able to deduce that, for another feature, values should fall within a specific range. A possible solution, when applying this technique to real world data, would be to initially assess the quality of the data and consult with clinicians to determine which features should be included and if it is possible to infer missing values from other features. From such collaboration, the techniques explained in this paper could equally be applied to study the onset of other adverse events, including rare adverse events and also the onset of other diseases.</p>
<p>Many of the features that are presented in the SHAP plots seem to display varied results suggesting that there is not enough evidence to suggest whether extremities of the features could be used to identify whether patients are more at risk or less at risk of developing an SAE which falls into one of these groups. An important limitation of this analysis is that the SHAP plots are only based on the best models which are based on the data provided, the data split used, and the algorithm applied. Despite using the same data split and the same parameter values there was variability within the 100 testing runs (see <xref ref-type="sec" rid="s11">Supplementary Tables S16&#x2013;27</xref> in the <xref ref-type="sec" rid="s11">Supplementary Materials</xref> for all 100 runs for each model). It is possible that with more runs the AUC may change and that other sets of test runs may have yielded different top five features to be explored in the SHAP plots. SHAP values are also based on an unrealistic assumption that the features are independent from each other. This assumption can lead to features being identified as providing a significantly high contribution score to the prediction when in reality it could be that certain features are always dependent on other features and this is contributing to the final contribution score (<xref ref-type="bibr" rid="B1">Aas et al., 2021</xref>). It is important to take into account that the results presented here are based on many factors and that the training data which the models are based on also include synthetic data (for the missing data imputation). It is therefore crucial to investigate any possible correlations between the features and predictions and perform further evaluation using statistical methods under correct assumptions in order to determine whether these features indeed have possible causative relationships with the onset of common SAE groups.</p>
<p>To conclude, from this study the best models for each analysis group were able to achieve sensitivity rates of at least 0.6 and AUC scores of at least 0.7. The Blood and Lymphatic Disorder group SAE grade 0 vs. grade 3 model achieved the highest AUC of 1. Other high performing models include the Infections and Infestations group SAE grade 0 vs. grade 5 and the Blood and Lymphatic Disorders group SAE grade 3 vs. the Metabolism and Nutrition Disorders group SAE grade 3. For the Blood and Lymphatic Disorder group SAE grade 0 vs. grade 3 model, patients younger than 80 years old are associated with the occurrence of grade 3. Further work should be undertaken to further investigate whether these features can be robustly used to predict the onset of these SAEs as well as identifying risk factors for other SAE groups.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement </title>
<p>Pre-existing clinical data underpinning this publication are available from Project Data Sphere at <ext-link ext-link-type="uri" xlink:href="https://data.projectdatasphere.org/projectdatasphere/html/access">https://data.projectdatasphere.org/projectdatasphere/html/access</ext-link>.</p>
</sec>
<sec sec-type="ethics-statement" id="s6">
<title>Ethics statement </title>
<p>The studies involving humans were approved by the University of Warwick Biomedical and Scientific Research Ethics Committee. The studies were conducted in accordance with the local legislation and institutional requirements. The human samples used in this study were acquired from Project Data Sphere&#x2012;a clinical data sharing platform. Written informed consent for participation was not required from the participants or the participants&#x2019; legal guardians/next of kin in accordance with the national legislation and institutional requirements.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>LW: Conceptualization, Formal Analysis, Methodology, Writing&#x2013;original draft, Writing&#x2013;review and editing. NE: Supervision, Writing&#x2013;original draft, Writing&#x2013;review and editing, Conceptualization, Methodology. MC: Supervision, Writing&#x2013;original draft, Writing&#x2013;review and editing, Conceptualization, Methodology.</p>
</sec>
<sec id="s8">
<title>Funding</title>
<p>The authors declare financial support was received for the research, authorship, and/or publication of this article. The research was funded by the Warwick National AI Turing Strategy Award.</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest </title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s11">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fdsfr.2023.1267623/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fdsfr.2023.1267623/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table1.DOCX" id="SM1" mimetype="application/DOCX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Aas</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Jullum</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>L&#xf8;land</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Explaining individual predictions when features are dependent: more accurate approximations to Shapley values</article-title>. <source>Artif. Intell.</source> <volume>298</volume>, <fpage>103502</fpage>. <pub-id pub-id-type="doi">10.1016/j.artint.2021.103502</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Arya</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Mount</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Kemp</surname>
<given-names>S. E.</given-names>
</name>
<name>
<surname>Jefferis</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>RANN: fast nearest neighbour search (wraps ANN library) using L2 metric</article-title>. <comment>R package version 2.6.1. Available at: <ext-link ext-link-type="uri" xlink:href="https://CRAN.R-project.org/package=RANN">https://CRAN.R-project.org/package&#x3d;RANN</ext-link> (Accessed May 30, 2023)</comment>.</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Azab</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Alassaf</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Abu-Humdan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dardas</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Almousa</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Alsalem</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Genotoxicity of cisplatin and carboplatin in cultured human lymphocytes: a comparative study</article-title>. <source>Interdiscip. Toxicol.</source> <volume>12</volume> (<issue>2</issue>), <fpage>93</fpage>&#x2013;<lpage>97</lpage>. <pub-id pub-id-type="doi">10.2478/intox-2019-0011</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bischl</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>MLR: machine learning in R</article-title>. <source>J. Mach. Learn. Res.</source> <volume>17</volume> (<issue>170</issue>), <fpage>1</fpage>&#x2013;<lpage>5</lpage>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Blagus</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Lusa</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>SMOTE for high-dimensional class-imbalanced data</article-title>. <source>BMC Bioinforma.</source> <volume>14</volume> (<issue>1</issue>), <fpage>106</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-14-106</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Blayney</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Pisoni</surname>
<given-names>R. L.</given-names>
</name>
<name>
<surname>Bragg-Gresham</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Bommer</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Piera</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Saito</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2008</year>). <article-title>High alkaline phosphatase levels in hemodialysis patients are associated with higher risk of hospitalization and death</article-title>. <source>Kidney Int.</source> <volume>74</volume> (<issue>5</issue>), <fpage>655</fpage>&#x2013;<lpage>663</lpage>. <pub-id pub-id-type="doi">10.1038/ki.2008.248</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="web">
<collab>Cancer Research UK</collab> (<year>2023</year>). <article-title>Lung cancer risk</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://www.cancerresearchuk.org/health-professional/cancer-statistics/statistics-by-cancer-type/lung-cancer/risk-factors#heading-Zero">https://www.cancerresearchuk.org/health-professional/cancer-statistics/statistics-by-cancer-type/lung-cancer/risk-factors&#x23;heading-Zero</ext-link> (Accessed May 30, 2023)</comment>.</citation>
</ref>
<ref id="B8">
<citation citation-type="web">
<collab>Cancer Therapy Evaluation Program</collab> (<year>2023</year>). <article-title>Protocol development</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://ctep.cancer.gov/protocoldevelopment/electronic_applications/ctc.htm">https://ctep.cancer.gov/protocoldevelopment/electronic_applications/ctc.htm</ext-link> (Accessed May 30, 2023)</comment>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Carvalho</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>Machado</surname>
<given-names>M. V.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>New insights about albumin and liver disease</article-title>. <source>Ann. Hepatology</source> <volume>17</volume> (<issue>4</issue>), <fpage>547</fpage>&#x2013;<lpage>560</lpage>. <pub-id pub-id-type="doi">10.5604/01.3001.0012.0916</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Guestrin</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Xgboost: A scalable tree boosting system</article-title>,&#x201d; in <conf-name>KDD&#x2019;16:Proceedings of the 22Nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</conf-name>, <conf-loc>California, San Francisco, USA</conf-loc>, <conf-date>August 13 - 17, 2016</conf-date>, <fpage>785</fpage>&#x2013;<lpage>794</lpage>. <pub-id pub-id-type="doi">10.1145/2939672.2939785</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>xgboost: extreme Gradient Boosting</article-title>. <comment>R package version 1.5.0.2. Available at: <ext-link ext-link-type="uri" xlink:href="https://CRAN.R-project.org/package=xgboost">https://CRAN.R-project.org/package&#x3d;xgboost</ext-link> (Accessed May 30, 2023)</comment>.</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chmielewski</surname>
<given-names>P. P.</given-names>
</name>
<name>
<surname>Strzelec</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Elevated leukocyte count as a harbinger of systemic inflammation, disease progression, and poor prognosis: a review</article-title>. <source>Folia Morphol.</source> <volume>77</volume> (<issue>2</issue>), <fpage>171</fpage>&#x2013;<lpage>178</lpage>. <pub-id pub-id-type="doi">10.5603/FM.a2017.0101</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cho</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Bilegsaikhan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Suh</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Machine learning improves the prediction of febrile neutropenia in Korean inpatients undergoing chemotherapy for breast cancer</article-title>. <source>Sci. Rep.</source> <volume>10</volume> (<issue>1</issue>), <fpage>14803</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-020-71927-6</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="web">
<collab>ClinicalTrials.gov:NCT00119613</collab> (<year>2008</year>). <article-title>A study of subjects with previously untreated extensive-stage small-cell lung cancer (SCLC) treated with platinum plus etoposide chemotherapy with or without darbepoetin alfa</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://clinicaltrials.gov/ct2/show/NCT00119613">https://clinicaltrials.gov/ct2/show/NCT00119613</ext-link> (Accessed May 30, 2023)</comment>.</citation>
</ref>
<ref id="B15">
<citation citation-type="web">
<collab>ClinicalTrials.gov:NCT00143455</collab> (<year>2010</year>). <article-title>Study of irinotecan hydrochloride (campto(R)) and cisplatin versus etoposide and cisplatin in small cell lung cancer</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://clinicaltrials.gov/ct2/show/NCT00143455">https://clinicaltrials.gov/ct2/show/NCT00143455</ext-link> (Accessed May 30, 2023)</comment>.</citation>
</ref>
<ref id="B16">
<citation citation-type="web">
<collab>ClinicalTrials.gov:NCT00363415</collab> (<year>2009</year>). <article-title>Study of pemetrexed and carboplatin compared with etoposide carboplatin to treat extensive-stage small cell lung cancer</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://clinicaltrials.gov/ct2/show/NCT00363415">https://clinicaltrials.gov/ct2/show/NCT00363415</ext-link> (Accessed May 30, 2023)</comment>.</citation>
</ref>
<ref id="B17">
<citation citation-type="web">
<collab>ClinicalTrials.gov:NCT01439568</collab> (<year>2019</year>). <article-title>A study of LY2510924 in participants with extensive-stage small cell lung carcinoma</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://clinicaltrials.gov/ct2/show/NCT01439568">https://clinicaltrials.gov/ct2/show/NCT01439568</ext-link> (Accessed May 30, 2023)</comment>.</citation>
</ref>
<ref id="B18">
<citation citation-type="web">
<collab>ClinicalTrials.gov:NCT0249970</collab> (<year>2020</year>). <article-title>Trilaciclib (G1T28), a CDK 4/6 inhibitor, in combination with etoposide and carboplatin in extensive stage small cell lung cancer (SCLC)</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://clinicaltrials.gov/ct2/show/NCT02499770">https://clinicaltrials.gov/ct2/show/NCT02499770</ext-link> (Accessed May 30, 2023)</comment>.</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Deneka</surname>
<given-names>A. Y.</given-names>
</name>
<name>
<surname>Boumber</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Beck</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Golemis</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Tumor-targeted Drug conjugates as an emerging novel therapeutic approach in small cell lung cancer (SCLC)</article-title>. <source>Cancers</source> <volume>11</volume> (<issue>9</issue>), <fpage>1297</fpage>. <pub-id pub-id-type="doi">10.3390/cancers11091297</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dubey</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Thompson</surname>
<given-names>P. M.</given-names>
</name>
<name>
<surname>Ye</surname>
<given-names>J.</given-names>
</name>
</person-group>
<collab>Alzheimer&#x27;s Disease Neuroimaging Initiative</collab> (<year>2014</year>). <article-title>Analysis of sampling techniques for imbalanced data: an n&#x3d;648 ADNI study</article-title>. <source>NeuroImage</source> <volume>87</volume>, <fpage>220</fpage>&#x2013;<lpage>241</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroimage.2013.10.005</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Duncan</surname>
<given-names>K. E.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>L. Y.</given-names>
</name>
<name>
<surname>Patronas</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>MEK inhibitors: a new class of chemotherapeutic agents with ocular toxicity</article-title>. <source>Eye</source> <volume>29</volume> (<issue>8</issue>), <fpage>1003</fpage>&#x2013;<lpage>1012</lpage>. <pub-id pub-id-type="doi">10.1038/eye.2015.82</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Francis</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Kappos</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>O&#x2019;Connor</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Collins</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Mercier</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Temporal profile of lymphocyte counts and relationship with infections with fingolimod therapy</article-title>. <source>Multiple Scler. J.</source> <volume>20</volume> (<issue>4</issue>), <fpage>471</fpage>&#x2013;<lpage>480</lpage>. <pub-id pub-id-type="doi">10.1177/1352458513500551</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hamoud</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Weaver</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Stec</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Hinds</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Bilirubin in the liver&#x2013;gut signaling Axis</article-title>. <source>Trends Endocrinol. Metabolism</source> <volume>29</volume> (<issue>3</issue>), <fpage>140</fpage>&#x2013;<lpage>150</lpage>. <pub-id pub-id-type="doi">10.1016/j.tem.2018.01.002</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hart</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>1989</year>). <source>&#x201c;Shapley value&#x201d; in game theory</source>. <publisher-loc>London</publisher-loc>: <publisher-name>Palgrave Macmillan</publisher-name>, <fpage>210</fpage>&#x2013;<lpage>216</lpage>.</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hashimoto</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kobayashi</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Clinical pharmacokinetics and pharmacodynamics of glyceryl trinitrate and its metabolites</article-title>. <source>Clin. Pharmacokinet.</source> <volume>42</volume> (<issue>3</issue>), <fpage>205</fpage>&#x2013;<lpage>221</lpage>. <pub-id pub-id-type="doi">10.2165/00003088-200342030-00001</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kahnert</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kauffmann-Guerrero</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Huber</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>SCLC&#x2013;State of the art and what does the future have in store?</article-title>. <source>Clin. Lung Cancer</source> <volume>17</volume> (<issue>5</issue>), <fpage>325</fpage>&#x2013;<lpage>333</lpage>. <pub-id pub-id-type="doi">10.1016/j.cllc.2016.05.014</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kao</surname>
<given-names>T. W.</given-names>
</name>
<name>
<surname>Chou</surname>
<given-names>C. H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C. C.</given-names>
</name>
<name>
<surname>Chou</surname>
<given-names>C. C.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>W. L.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Associations between serum total bilirubin levels and functional dependence in the elderly</article-title>. <source>Intern. Med. J.</source> <volume>42</volume> (<issue>11</issue>), <fpage>1199</fpage>&#x2013;<lpage>1207</lpage>. <pub-id pub-id-type="doi">10.1111/j.1445-5994.2011.02620.x</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kishida</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kawahara</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Teramukai</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kubota</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Komuta</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Minato</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>Chemotherapy-induced neutropenia as a prognostic factor in advanced non-small-cell lung cancer: results from Japan multinational trial organization LC00-03</article-title>. <source>Br. J. Cancer</source> <volume>101</volume> (<issue>9</issue>), <fpage>1537</fpage>&#x2013;<lpage>1542</lpage>. <pub-id pub-id-type="doi">10.1038/sj.bjc.6605348</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kitao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Iimori</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kataoka</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wakasa</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Tokunaga</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Saeki</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>DNA replication stress and cancer chemotherapy</article-title>. <source>Cancer Sci.</source> <volume>109</volume> (<issue>2</issue>), <fpage>264</fpage>&#x2013;<lpage>271</lpage>. <pub-id pub-id-type="doi">10.1111/cas.13455</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kochanek</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Schalk</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>von Bergwelt-Baildon</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Beutel</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Buchheidt</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Hentrich</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Management of sepsis in neutropenic cancer patients: 2018 guidelines from the infectious diseases working party (AGIHO) and intensive care working party (iCHOP) of the German society of hematology and medical oncology (DGHO)</article-title>. <source>Ann. Hematol.</source> <volume>98</volume> (<issue>5</issue>), <fpage>1051</fpage>&#x2013;<lpage>1069</lpage>. <pub-id pub-id-type="doi">10.1007/s00277-019-03622-0</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Krohn</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ahrens</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yalcin</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Pl&#xf6;nes</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wehrle</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Taromi</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Tumor cell heterogeneity in small cell lung cancer (SCLC): phenotypical and functional differences associated with epithelial-mesenchymal transition (EMT) and DNA methylation changes</article-title>. <source>PLoS ONE</source> <volume>9</volume> (<issue>6</issue>), <fpage>e100249</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0100249</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Kuhn</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>CARET: classification and regression training</article-title>. <comment>R package version 6.0-86. Available at: <ext-link ext-link-type="uri" xlink:href="https://CRAN.R-project.org/package=caret">https://CRAN.R-project.org/package&#x3d;caret</ext-link> (Accessed May 30, 2023)</comment>.</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Shinde</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Glaser</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lyou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yuh</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Machine learning&#x2013;based interpretation and visualization of nonlinear interactions in prostate cancer survival</article-title>. <source>JCO Clin. Cancer Inf.</source> <volume>4</volume>, <fpage>637</fpage>&#x2013;<lpage>646</lpage>. <pub-id pub-id-type="doi">10.1200/CCI.20.00002</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ludwig</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Aapro</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bokemeyer</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Glaspy</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hedenus</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Littlewood</surname>
<given-names>T. J.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>A European patient record study on diagnosis and treatment of chemotherapy-induced anaemia</article-title>. <source>Support. Care Cancer</source> <volume>22</volume> (<issue>8</issue>), <fpage>2197</fpage>&#x2013;<lpage>2206</lpage>. <pub-id pub-id-type="doi">10.1007/s00520-014-2189-0</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lyman</surname>
<given-names>G. H.</given-names>
</name>
<name>
<surname>Abella</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Pettengell</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Risk factors for febrile neutropenia among patients with cancer receiving chemotherapy: A systematic review</article-title>. <source>Crit. Rev. Oncology/Hematology</source> <volume>90</volume> (<issue>3</issue>), <fpage>190</fpage>&#x2013;<lpage>199</lpage>. <pub-id pub-id-type="doi">10.1016/j.critrevonc.2013.12.006</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Malarvizhi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Thanamani</surname>
<given-names>A. S.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>K-nearest neighbor in missing data imputation</article-title>. <source>Int. J. Eng. Res. Dev.</source> <volume>5</volume> (<issue>1</issue>), <fpage>5</fpage>&#x2013;<lpage>7</lpage>.</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Matthewson</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Al Mardini</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Bartlett</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Record</surname>
<given-names>C. O.</given-names>
</name>
</person-group> (<year>1986</year>). <article-title>Impaired acetaldehyde metabolism in patients with non-alcoholic liver disorders</article-title>. <source>Gut</source> <volume>27</volume>, <fpage>756</fpage>&#x2013;<lpage>764</lpage>. <pub-id pub-id-type="doi">10.1136/gut.27.7.756</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McQuade</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Al Thaalibi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Nurgali</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Impact of chemotherapy-induced enteric nervous system toxicity on gastrointestinal mucositis</article-title>. <source>Curr. Opin. Support. Palliat. Care</source> <volume>14</volume> (<issue>3</issue>), <fpage>293</fpage>&#x2013;<lpage>300</lpage>. <pub-id pub-id-type="doi">10.1097/SPC.0000000000000515</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mercadante</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gebbia</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Marrazzo</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Filosto</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>Anaemia in cancer: pathophysiology and treatment</article-title>. <source>Cancer Treat. Rev.</source> <volume>26</volume> (<issue>4</issue>), <fpage>303</fpage>&#x2013;<lpage>311</lpage>. <pub-id pub-id-type="doi">10.1053/ctrv.2000.0181</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nasteski</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>An overview of the supervised machine learning methods</article-title>. <source>Horizons</source> <volume>4</volume>, <fpage>51</fpage>&#x2013;<lpage>62</lpage>. <pub-id pub-id-type="doi">10.20544/HORIZONS.B.04.1.17.P05</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nazir</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Naz</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ali</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Aziz</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Malik</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Qari</surname>
<given-names>I. H.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Anaemia: the neglected female health problem in developing countries</article-title>. <source>J. Ayub Med Coll Abbottabad</source> <volume>23</volume> (<issue>2</issue>), <fpage>8</fpage>&#x2013;<lpage>11</lpage>.</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nesher</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Rolston</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>The current spectrum of infection in cancer patients with chemotherapy related neutropenia</article-title>. <source>Infection</source> <volume>42</volume> (<issue>1</issue>), <fpage>5</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1007/s15010-013-0525-9</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Oun</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Moussa</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wheate</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>The side effects of platinum-based chemotherapy drugs: a review for chemists</article-title>. <source>Dalton Trans.</source> <volume>47</volume> (<issue>19</issue>), <fpage>6645</fpage>&#x2013;<lpage>6653</lpage>. <pub-id pub-id-type="doi">10.1039/c8dt00838h</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Prasad</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Suppression of hypercholesterolemic atherosclerosis by pentoxifylline and its mechanism</article-title>. <source>Atherosclerosis</source> <volume>192</volume> (<issue>2</issue>), <fpage>313</fpage>&#x2013;<lpage>322</lpage>. <pub-id pub-id-type="doi">10.1016/j.atherosclerosis.2006.07.034</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="web">
<collab>Project Data Sphere</collab> (<year>2023</year>). <article-title>Access data</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://data.projectdatasphere.org/projectdatasphere/html/access">https://data.projectdatasphere.org/projectdatasphere/html/access</ext-link> (Accessed May 30, 2023)</comment>.</citation>
</ref>
<ref id="B46">
<citation citation-type="book">
<collab>R Core Team</collab> (<year>2023</year>). <source>R: A language and environment for statistical computing</source>. <publisher-loc>Austria</publisher-loc>: <publisher-name>R Foundation for Statistical Computing Vienna</publisher-name>
<comment>. <ext-link ext-link-type="uri" xlink:href="https://www.R-project.org/">https://www.R-project.org/</ext-link>(Accessed May 30, 2023)</comment>.</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Robin</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Turck</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Hainard</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tiberti</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Lisacek</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Sanchez</surname>
<given-names>J. C.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>pROC: an open-source package for R and S&#x2b; to analyze and compare ROC curves</article-title>. <source>BMC Bioinforma.</source> <volume>12</volume>, <fpage>77</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-12-77</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rosen</surname>
<given-names>M. J.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Chronic cough due to tuberculosis and other infections: ACCP evidence-based clinical practice guidelines</article-title>. <source>Chest</source> <volume>129</volume> (<issue>1</issue>), <fpage>197S</fpage>&#x2013;<lpage>201S</lpage>. <pub-id pub-id-type="doi">10.1378/chest.129.1_suppl.197S</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rusciano</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>De Luca</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Pesce</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Sasso</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Raman tweezers as a diagnostic tool of hemoglobin-related blood disorders</article-title>. <source>Sensors</source> <volume>8</volume> (<issue>12</issue>), <fpage>7818</fpage>&#x2013;<lpage>7832</lpage>. <pub-id pub-id-type="doi">10.3390/s8127818</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Self-adaptive cost weights-based support vector machine cost-sensitive ensemble for imbalanced data classification</article-title>. <source>Inf. Sci.</source> <volume>487</volume>, <fpage>31</fpage>&#x2013;<lpage>56</lpage>. <pub-id pub-id-type="doi">10.1016/j.ins.2019.02.062</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Torgo</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Data Mining with R, learning with case studies Chapman and Hall/CRC</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="http://www.dcc.fc.up.pt/%7Eltorgo/DataMiningWithR">http://www.dcc.fc.up.pt/&#x223c;ltorgo/DataMiningWithR</ext-link> (Accessed May 30, 2023)</comment>.</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Trotti</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Colevas</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Setser</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Rusch</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Jaques</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Budach</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2003</year>). <article-title>CTCAE v3.0: development of a comprehensive grading system for the adverse effects of cancer treatment</article-title>. <source>Seminars Radiat. Oncol.</source> <volume>13</volume> (<issue>3</issue>), <fpage>176</fpage>&#x2013;<lpage>181</lpage>. <pub-id pub-id-type="doi">10.1016/S1053-4296(03)00031-6</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ven&#xe4;l&#xe4;inen</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Heerv&#xe4;</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Hirvonen</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Saraei</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Suomi</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Mikkola</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Improved risk prediction of chemotherapy&#x2010;induced neutropenia&#x2014;Model development and validation with real&#x2010;world data</article-title>. <source>Cancer Med.</source> <volume>11</volume> (<issue>3</issue>), <fpage>654</fpage>&#x2013;<lpage>663</lpage>. <pub-id pub-id-type="doi">10.1002/cam4.4465</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Weycker</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Hatfield</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Grossman</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hanau</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lonshteyn</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sharma</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Risk and consequences of chemotherapy-induced thrombocytopenia in US clinical practice</article-title>. <source>BMC Cancer</source> <volume>19</volume> (<issue>1</issue>), <fpage>151</fpage>. <pub-id pub-id-type="doi">10.1186/s12885-019-5354-5</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wiberg</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Montanaro</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Mather</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Birz</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Schneider</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Prediction of neutropenic events in chemotherapy patients: A machine learning approach</article-title>. <source>JCO Clin. Cancer Inf.</source> <volume>5</volume>, <fpage>904</fpage>&#x2013;<lpage>911</lpage>. <pub-id pub-id-type="doi">10.1200/CCI.21.00046</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="web">
<collab>World Cancer Research Fund International</collab> (<year>2023</year>). <article-title>Lung cancer statistics</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://www.wcrf.org/cancer-trends/lung-cancer-statistics/">https://www.wcrf.org/cancer-trends/lung-cancer-statistics/</ext-link>
</comment>(<comment>Accessed May 30, 2023)</comment>.</citation>
</ref>
</ref-list>
</back>
</article>