<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Oncol.</journal-id>
<journal-title>Frontiers in Oncology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Oncol.</abbrev-journal-title>
<issn pub-type="epub">2234-943X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fonc.2024.1403392</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Oncology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Construction of a risk prediction model for lung infection after chemotherapy in lung cancer patients based on the machine learning algorithm</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Sun</surname>
<given-names>Tao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2374989"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Jun</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2337714"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yuan</surname>
<given-names>Houqin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Xin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yan</surname>
<given-names>Hui</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Hematology and Oncology Laboratory, The Central Hospital of Shaoyang, Shaoyang</institution>, <addr-line>Hunan</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Scientific Research, The First Affiliated Hospital of Shaoyang University, Shaoyang</institution>, <addr-line>Hunan</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Xuanye Cao, University of Texas MD Anderson Cancer Center, United States</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Minghui Liu, University of Electronic Science and Technology of China, China</p>
<p>Alexandre Malek, Ochsner LSU Health, United States</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Tao Sun, <email xlink:href="mailto:taosun2023@126.com">taosun2023@126.com</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>09</day>
<month>08</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>14</volume>
<elocation-id>1403392</elocation-id>
<history>
<date date-type="received">
<day>19</day>
<month>03</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>23</day>
<month>07</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Sun, Liu, Yuan, Li and Yan</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Sun, Liu, Yuan, Li and Yan</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Purpose</title>
<p>The objective of this study was to create and validate a machine learning (ML)-based model for predicting the likelihood of lung infections following chemotherapy in patients with lung cancer.</p>
</sec>
<sec>
<title>Methods</title>
<p>A retrospective study was conducted on a cohort of 502 lung cancer patients undergoing chemotherapy. Data on age, Body Mass Index (BMI), underlying disease, chemotherapy cycle, number of hospitalizations, and various blood test results were collected from medical records. We used the Synthetic Minority Oversampling Technique (SMOTE) to handle unbalanced data. Feature screening was performed using the Boruta algorithm and The Least Absolute Shrinkage and Selection Operator (LASSO). Subsequently, six ML algorithms, namely Logistic Regression (LR), Random Forest (RF), Gaussian Naive Bayes (GNB), Multi-layer Perceptron (MLP), Support Vector Machine (SVM), and K-Nearest Neighbors (KNN) were employed to train and develop an ML model using a 10-fold cross-validation methodology. The model&#x2019;s performance was evaluated through various metrics, including the area under the receiver operating characteristic curve (ROC), accuracy, sensitivity, specificity, F1 score, calibration curve, decision curves, clinical impact curve, and confusion matrix. In addition, model interpretation was performed by the Shapley Additive Explanations (SHAP) analysis to clarify the importance of each feature of the model and its decision basis. Finally, we constructed nomograms to make the predictive model results more readable.</p>
</sec>
<sec>
<title>Results</title>
<p>The integration of Boruta and LASSO methodologies identified Gender, Smoke, Drink, Chemotherapy cycles, pleural effusion (PE), Neutrophil-lymphocyte count ratio (NLR), Neutrophil-monocyte count ratio (NMR), Lymphocytes (LYM) and Neutrophil (NEUT) as significant predictors. The LR model demonstrated superior performance compared to alternative ML algorithms, achieving an accuracy of 81.80%, a sensitivity of 81.1%, a specificity of 82.5%, an F1 score of 81.6%, and an AUC of 0.888(95%CI(0.863-0.911)). Furthermore, the SHAP method identified Chemotherapy cycles and Smoke as the primary decision factors influencing the ML model&#x2019;s predictions. Finally, this study successfully constructed interactive nomograms and dynamic nomograms.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>The ML algorithm, combining demographic and clinical factors, accurately predicted post-chemotherapy lung infections in cancer patients. The LR model performed well, potentially improving early detection and treatment in clinical practice.</p>
</sec>
</abstract>
<kwd-group>
<kwd>lung infection</kwd>
<kwd>chemotherapy</kwd>
<kwd>machine learning</kwd>
<kwd>logistic regression</kwd>
<kwd>predictive model</kwd>
<kwd>nomogram</kwd>
</kwd-group>
<counts>
<fig-count count="6"/>
<table-count count="4"/>
<equation-count count="0"/>
<ref-count count="36"/>
<page-count count="16"/>
<word-count count="7853"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Thoracic Oncology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Lung cancer, being one of the most prevalent malignant neoplasms globally, presents a substantial risk to both the survival and well-being of affected individuals (<xref ref-type="bibr" rid="B1">1</xref>). The World Health Organization&#x2019;s data indicates that lung cancer exhibits the highest incidence and mortality rates among all cancer types (<xref ref-type="bibr" rid="B2">2</xref>). Despite notable advancements in lung cancer therapy, the effective management of post-chemotherapy complications remains a significant hurdle (<xref ref-type="bibr" rid="B3">3</xref>&#x2013;<xref ref-type="bibr" rid="B5">5</xref>). Of particular concern is the high prevalence of lung infections following chemotherapy in lung cancer patients, which seriously affects the therapeutic effect and survival quality of patients (<xref ref-type="bibr" rid="B6">6</xref>). The presence of lung infections in lung cancer patients not only exacerbates their health status but also has the potential to impede or halt chemotherapy, thereby impacting the overall efficacy of treatment. Furthermore, lung infections contribute to escalated medical expenses, extended hospital stays, and heightened mortality rates (<xref ref-type="bibr" rid="B7">7</xref>). Consequently, the timely and precise identification of the likelihood of lung infections following chemotherapy is crucial for informing clinical interventions and enhancing patient outcomes.</p>
<p>The utilization of ML technology in the healthcare sector has experienced significant growth in recent years, showcasing robust data processing and pattern recognition capabilities. ML algorithms have exhibited promise and efficacy in lung cancer diagnosis, treatment selection, and prognosis assessment (<xref ref-type="bibr" rid="B8">8</xref>, <xref ref-type="bibr" rid="B9">9</xref>). Notably, the analysis of extensive clinical data through ML algorithms can aid healthcare professionals in identifying potential disease development patterns, facilitating personalized treatment strategies, and enhancing treatment outcomes (<xref ref-type="bibr" rid="B10">10</xref>&#x2013;<xref ref-type="bibr" rid="B12">12</xref>). Conventional approaches to evaluating the risk of lung infection rely heavily on the subjective judgment and clinical expertise of healthcare professionals, necessitating a greater degree of objectivity and precision. In light of this prevailing situation, the utilization of ML technology presents novel opportunities for addressing this issue by leveraging ML algorithms to analyze extensive patient data, potential correlations and patterns can be identified, enabling healthcare providers to make more precise predictions regarding the likelihood of lung infection following chemotherapy in individuals with lung cancer.</p>
<p>In recent studies, researchers have utilized various ML algorithms to create predictive models aimed at aiding physicians in evaluating the likelihood of complications in lung cancer patients following chemotherapy or surgical procedures. While previous research has explored the application of ML in forecasting complications in lung cancer patients, there is a notable scarcity of studies focusing on predicting the likelihood of lung infection following chemotherapy. Consequently, the current study seeks to address this gap by introducing and refining a prediction model utilizing ML algorithms to identify lung cancer patients at risk of post-chemotherapy lung infection. This study posits that an interpretable ML-based algorithm will achieve the most accurate predictions if significant predictors are identified through an effective feature selection method. Therefore, the objective of this study was to create and evaluate a proficient and interpretable ML system for forecasting the likelihood of lung infection following chemotherapy in Chinese lung cancer patients. Our research findings offer a novel approach for early identification of infection risk in lung cancer patients while also contributing to the advancement of ML in oncology clinical investigations. Moving forward, we intend to enhance the precision and reliability of the model, facilitate its integration into clinical settings, and offer enhanced scientific and precise assistance for the care and oversight of lung cancer patients.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>Study design</title>
<p>This study was conducted to develop a machine learning-based model for predicting the risk of lung infections following chemotherapy in lung cancer patients. The retrospective study included a cohort of 502 lung cancer patients who had undergone chemotherapy, aged 18 years and above, and had completed at least one cycle of treatment. Data encompassing demographic details, medical history, chemotherapy specifics, and blood test results were extracted from the hospital&#x2019;s electronic medical record system. The SMOTE algorithm is used to solve the category imbalance problem. The Boruta algorithm and LASSO regression performed feature screening to identify the features most associated with the risk of lung infection. Subsequently, a range of ML models, including LR, RF, GNB, MLP, SVM, and KNN, were developed and refined by applying a 10-fold cross-validation methodology. The performance of these models was assessed using various metrics, including accuracy, sensitivity, specificity, positive predictive value, negative predictive value, F1 score, Kappa score, AUC, calibration curve, calibration curves, Clinical Impact Curve and confusion matrix. To enhance the transparency and interpretability of the model, the SHAP method was employed to interpret the predicted results and elucidate the impact of each feature on the predictions, thereby offering a practical reference for clinicians. <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref> explains the overall workflow of the proposed system more clearly.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Research flowchart.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-14-1403392-g001.tif"/>
</fig>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Study data</title>
<p>This retrospective study examined data from lung cancer patients at The Central Hospital of Shaoyang between January 2020 and December 2023. The study included adult patients aged 18 years and older who had not experienced lung infections within a week before receiving chemotherapy. Patient records with missing or abnormal data were excluded to maintain data quality. The study&#x2019;s rigorous inclusion and exclusion criteria aimed to ensure the completeness and reliability of the information on included cases, thus providing a high-quality database for evaluating the risk of lung infections in lung cancer patients after chemotherapy. Inclusion criteria: (i) adult patients aged &#x2265;18 years, (ii) patients diagnosed with lung cancer and treated with chemotherapy, (iii) patients who did not have any lung infection before chemotherapy, and (iv) patients with complete clinical information; Exclusion criteria: (i) patients with mental illness or intellectual disability, (ii) patients with missing or abnormal data, and (iii) exclusion of patients with a combination of other tumors.</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Research variables</title>
<p>The study encompassed 36 predictors related to demographic factors (gender, age), lifestyle habits (history of alcohol consumption, history of smoking), medical history (history of diabetes, history of hypertension, history of coronary heart disease), physical characteristics (BMI), disease severity (stage at diagnosis, histologic features, presence or absence of pleural effusion), treatment information (cycles of chemotherapy, number of hospitalizations), and laboratory values (leukocytes, erythrocytes, hemoglobin, platelets, percentage of neutrophils, percentage of lymphocytes, percentage of monocytes, NLR, NMR, neutrophil-platelet count ratio (NPR), indirect bilirubin, alanine aminotransferase, glutamine aminotransferase, total bilirubin, direct bilirubin, total protein, albumin, globulin, white globule ratio, urea, creatinine, uric acid, and CEA). Of these, gender, age, history of alcohol consumption, history of smoking, history of diabetes mellitus, history of hypertension, history of coronary artery disease, BMI, tumor typing, cycles of chemotherapy, number of hospitalizations, and the presence or absence of pleural effusions were the data before the last chemotherapy session. The other laboratory data were obtained after the last chemotherapy. A brief description of the study variables is given in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Description of the study variables.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">SN</th>
<th valign="top" align="left">Predictors</th>
<th valign="top" align="left">Description</th>
<th valign="top" align="left">Types</th>
<th valign="top" align="left">Values</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1</td>
<td valign="top" align="left">Gender</td>
<td valign="top" align="left">Sex of the patient</td>
<td valign="top" align="left">Categorical</td>
<td valign="top" align="left">1 male<break/>2 female</td>
</tr>
<tr>
<td valign="top" align="left">2</td>
<td valign="top" align="left">Age</td>
<td valign="top" align="left">Age of the patient (years)</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">35-83</td>
</tr>
<tr>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Drink</td>
<td valign="top" align="left">History of alcohol consumption</td>
<td valign="top" align="left">Categorical</td>
<td valign="top" align="left">0 No history of alcohol consumption<break/>1 History of alcohol consumption</td>
</tr>
<tr>
<td valign="top" align="left">4</td>
<td valign="top" align="left">Smoke</td>
<td valign="top" align="left">History of smoking</td>
<td valign="top" align="left">Categorical</td>
<td valign="top" align="left">0 No history of smoking<break/>1 History of smoking</td>
</tr>
<tr>
<td valign="top" align="left">5</td>
<td valign="top" align="left">Diabetes</td>
<td valign="top" align="left">History of diabetes</td>
<td valign="top" align="left">Categorical</td>
<td valign="top" align="left">0 No history of diabetes<break/>1 History of diabetes</td>
</tr>
<tr>
<td valign="top" align="left">6</td>
<td valign="top" align="left">Hypertension</td>
<td valign="top" align="left">History of Hypertension</td>
<td valign="top" align="left">Categorical</td>
<td valign="top" align="left">0 No history of hypertension<break/>1 History of hypertension</td>
</tr>
<tr>
<td valign="top" align="left">7</td>
<td valign="top" align="left">CHD</td>
<td valign="top" align="left">History of coronary heart disease</td>
<td valign="top" align="left">Categorical</td>
<td valign="top" align="left">0 No history of coronary heart disease<break/>1 History of coronary heart disease</td>
</tr>
<tr>
<td valign="top" align="left">8</td>
<td valign="top" align="left">BMI</td>
<td valign="top" align="left">Body mass index (kg/m2)</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">11.43-31.83</td>
</tr>
<tr>
<td valign="top" align="left">9</td>
<td valign="top" align="left">Stage</td>
<td valign="top" align="left">Stage at diagnosis, Count (%)</td>
<td valign="top" align="left">Categorical</td>
<td valign="top" align="left">Stage 1 24(4.78%)<break/>Stage 2 59(11.75%)<break/>Stage 3 204(40.64%)<break/>Stage 4 215(42.83%)</td>
</tr>
<tr>
<td valign="top" align="left">10</td>
<td valign="top" align="left">Histology</td>
<td valign="top" align="left">Histologic features, Count (%). 1, Adenocarcinoma; 2, Squamous; 3, SCLC; 4, Other lung cancers</td>
<td valign="top" align="left">Categorical</td>
<td valign="top" align="left">Grade1 222(44.22%)<break/>Grade2 189(37.65%)<break/>Grade3 80(15.94%)<break/>Grade4 11(2.19%)</td>
</tr>
<tr>
<td valign="top" align="left">11</td>
<td valign="top" align="left">Chemotherapy cycles</td>
<td valign="top" align="left">The Number of chemotherapy cycles</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">1-32</td>
</tr>
<tr>
<td valign="top" align="left">12</td>
<td valign="top" align="left">Hospitalizations</td>
<td valign="top" align="left">Total number of hospitalizations</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">1-45</td>
</tr>
<tr>
<td valign="top" align="left">13</td>
<td valign="top" align="left">PE</td>
<td valign="top" align="left">The presence of pleural effusion</td>
<td valign="top" align="left">Categorical</td>
<td valign="top" align="left">0 No pleural effusion<break/>1 With pleural effusion</td>
</tr>
<tr>
<td valign="top" align="left">14</td>
<td valign="top" align="left">WBC</td>
<td valign="top" align="left">White blood cell</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">1.31-60.80</td>
</tr>
<tr>
<td valign="top" align="left">15</td>
<td valign="top" align="left">RBC</td>
<td valign="top" align="left">Red blood cell</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">1.44-6.20</td>
</tr>
<tr>
<td valign="top" align="left">16</td>
<td valign="top" align="left">HGB</td>
<td valign="top" align="left">Hemoglobin</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">53.00-9792.00</td>
</tr>
<tr>
<td valign="top" align="left">17</td>
<td valign="top" align="left">PLT</td>
<td valign="top" align="left">Platelet</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">22.00-631.00</td>
</tr>
<tr>
<td valign="top" align="left">18</td>
<td valign="top" align="left">NEUT</td>
<td valign="top" align="left">Percentage of Neutrophil</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">0.44-98.21</td>
</tr>
<tr>
<td valign="top" align="left">19</td>
<td valign="top" align="left">LYM</td>
<td valign="top" align="left">Percentage of Lymphocytes</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">1.42-65.50</td>
</tr>
<tr>
<td valign="top" align="left">20</td>
<td valign="top" align="left">NLR</td>
<td valign="top" align="left">Neutrophil-Lymphocyte count ratio</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">0.02-69.16</td>
</tr>
<tr>
<td valign="top" align="left">21</td>
<td valign="top" align="left">NMR</td>
<td valign="top" align="left">Neutrophil-Monocyte count ratio</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">0.01-311.37</td>
</tr>
<tr>
<td valign="top" align="left">22</td>
<td valign="top" align="left">NPR</td>
<td valign="top" align="left">Neutrophil-Platelet count ratio</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">0.01-3.67</td>
</tr>
<tr>
<td valign="top" align="left">23</td>
<td valign="top" align="left">MONO</td>
<td valign="top" align="left">Percentage of Monocytes</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">0.30-63.20</td>
</tr>
<tr>
<td valign="top" align="left">24</td>
<td valign="top" align="left">IBIL</td>
<td valign="top" align="left">Indirect bilirubin</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">2.20-90.40</td>
</tr>
<tr>
<td valign="top" align="left">25</td>
<td valign="top" align="left">ALT</td>
<td valign="top" align="left">Glutamic pyruvic transaminase</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">2.70-888.60</td>
</tr>
<tr>
<td valign="top" align="left">26</td>
<td valign="top" align="left">AST</td>
<td valign="top" align="left">Aspartate aminotransferase</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">3.80-591.20</td>
</tr>
<tr>
<td valign="top" align="left">27</td>
<td valign="top" align="left">TBIL</td>
<td valign="top" align="left">Total bilirubin</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">1.90-297.60</td>
</tr>
<tr>
<td valign="top" align="left">28</td>
<td valign="top" align="left">DBIL</td>
<td valign="top" align="left">Direct bilirubin</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">0.13-207.20</td>
</tr>
<tr>
<td valign="top" align="left">29</td>
<td valign="top" align="left">TP</td>
<td valign="top" align="left">Total protein</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">22.50-85.80</td>
</tr>
<tr>
<td valign="top" align="left">30</td>
<td valign="top" align="left">ALB</td>
<td valign="top" align="left">Albumin</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">10.70-51.04</td>
</tr>
<tr>
<td valign="top" align="left">31</td>
<td valign="top" align="left">GLB</td>
<td valign="top" align="left">Globulin</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">11.96-51.90</td>
</tr>
<tr>
<td valign="top" align="left">32</td>
<td valign="top" align="left">A/G</td>
<td valign="top" align="left">White ball ratio</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">0.48-3.99</td>
</tr>
<tr>
<td valign="top" align="left">33</td>
<td valign="top" align="left">Urea</td>
<td valign="top" align="left">Urea</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">1.25-32.97</td>
</tr>
<tr>
<td valign="top" align="left">34</td>
<td valign="top" align="left">CREA</td>
<td valign="top" align="left">Creatinine</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">34.70-367.90</td>
</tr>
<tr>
<td valign="top" align="left">35</td>
<td valign="top" align="left">UA</td>
<td valign="top" align="left">Uric acid</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">78.30-1201.40</td>
</tr>
<tr>
<td valign="top" align="left">36</td>
<td valign="top" align="left">CEA</td>
<td valign="top" align="left">CEA</td>
<td valign="top" align="left">Continuous</td>
<td valign="top" align="left">0.20-1500.00</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Diagnostic criteria of pulmonary infection after chemotherapy</title>
<p>The diagnostic criteria for pulmonary infection in patients with lung cancer following chemotherapy encompass a body temperature exceeding 38&#xb0;C, the presence of clinical symptoms indicative of pulmonary infection (e.g., cough and expectoration), the identification of moist rales in the lungs, and the visualization of a distinct infectious focus on CT imaging. Should a lung cancer patient meet at least three of these criteria within 14 days post-operation, a diagnosis of post-chemotherapy lung infection is warranted.</p>
</sec>
<sec id="s2_5">
<label>2.5</label>
<title>Feature screening</title>
<sec id="s2_5_1">
<label>2.5.1</label>
<title>Least absolute shrinkage and selection operator</title>
<p>The LASSO regression enhances model refinement by implementing a penalty function that compresses certain regression coefficients, thereby enforcing a constraint on the sum of their absolute values to be below a predetermined threshold (<xref ref-type="bibr" rid="B13">13</xref>, <xref ref-type="bibr" rid="B14">14</xref>). We utilize the glmnet package in R for LASSO regression, setting family=&#x201c;binomial&#x201d; to apply to our binary outcome data. The key parameter alpha is set to 1, and the LASSO method is used entirely. Through cross-validation with the cv.glmnet function, we chose two lambda values: lambda.min and lambda.1se. The former minimizes the cross-validation error, while the latter provides a cleaner model, which together help us to balance the complexity of the model with the prediction accuracy. Ultimately, we filter out variables that are significant to the predictions based on non-zero coefficients, simplifying the model and improving its interpretability.</p>
</sec>
<sec id="s2_5_2">
<label>2.5.2</label>
<title>Boruta</title>
<p>The Boruta algorithm is a Random Forest-based feature selection and packaging algorithm that evaluates the importance of features by generating &#x201c;shadow variables&#x201d; corresponding to each original variable in the dataset (<xref ref-type="bibr" rid="B15">15</xref>). In particular, Boruta (Version: 8.0.0) is executed to perform feature selection, where the algorithm iteratively compares the importance of each original variable with its shadow variable, and determines the importance of each variable over 500 iterations or until all variables are stable. Importance results are extracted with the attStats function and formatted with a customized adjustdata function (<xref ref-type="bibr" rid="B16">16</xref>).</p>
</sec>
</sec>
<sec id="s2_6">
<label>2.6</label>
<title>Machine learning algorithms</title>
<sec id="s2_6_1">
<label>2.6.1</label>
<title>Logistic regression algorithm</title>
<p>In this study, we used a logistic regression (LR) model to predict the probability of infection in patients receiving chemotherapy, defined as a binary classification problem that predicts the risk of infection based only on clinical features (<xref ref-type="bibr" rid="B17">17</xref>). The logistic regression model used L2 regularization with the regularization factor (C) set to 1.0, a maximum number of iterations of 100, and a convergence tolerance (tol) of 0.0001.These parameters help prevent model overfitting while ensuring convergence and computational efficiency of the algorithm.</p>
</sec>
<sec id="s2_6_2">
<label>2.6.2</label>
<title>Random forest algorithm</title>
<p>The RF algorithm is an ML technique that enhances predictive accuracy by generating multiple decision trees. RFs excel in analyzing extensive datasets with high-dimensional features, effectively managing intricate relationships among data variables (<xref ref-type="bibr" rid="B18">18</xref>). In this research, RFs are employed to identify non-linear associations and enhance the model&#x2019;s ability to generalize. In the Random Forest model, the Gini Index is used as the splitting criterion, the number of trees is set to 20, the maximum depth of the tree is not restricted, and the minimum impurity reduction is set to 0.0. This parameter configuration is designed to allow the model to fully learn the complex structure in the data, and to improve the accuracy and generalization of the prediction.</p>
</sec>
<sec id="s2_6_3">
<label>2.6.3</label>
<title>Gaussian Naive Bayes algorithm</title>
<p>The GNB classifier is a straightforward probabilistic model grounded in Bayes&#x2019; theorem, predicated on the feature independence assumption. While this assumption may not hold true in all practical scenarios, GNB remains highly effective in numerous instances owing to its simplicity and computational efficiency (<xref ref-type="bibr" rid="B19">19</xref>). The Gaussian Naive Bayes model does not set a specific prior probability, and the variable smoothing parameter is set to 1e-09. this setting allows the model to be more accurate when performing probability calculations, especially when dealing with datasets with continuous characteristics.</p>
</sec>
<sec id="s2_6_4">
<label>2.6.4</label>
<title>Multi-layer perceptron algorithm</title>
<p>The MLP is a feed-forward artificial neural network model capable of processing data through multiple layers to learn non-linear features (<xref ref-type="bibr" rid="B20">20</xref>). It is well-suited for complex pattern recognition tasks. In this research, we employ MLP to develop a sophisticated predictive model for assessing the risk of lung infection following chemotherapy, the multilayer perceptron model uses ReLU as the activation function, and the structure of the hidden layer is set to two layers containing 20 and 10 neurons, respectively, with a maximum number of iterations of 20.</p>
</sec>
<sec id="s2_6_5">
<label>2.6.5</label>
<title>Support vector machine algorithm</title>
<p>Support Vector Machine (SVM) is robust classifiers utilized to discern between classes by identifying optimal decision boundaries within data points. SVMs are especially adept at processing high-dimensional data and excel in scenarios where data boundaries are ambiguous (<xref ref-type="bibr" rid="B21">21</xref>, <xref ref-type="bibr" rid="B22">22</xref>). In this study, the SVM model selects Radial Basis Function (RBF) as the kernel function, with the regularization parameter C set to 1.0 and the tolerance to 0.001. This setting helps the model to effectively identify complex decision boundaries while controlling overfitting when dealing with high-dimensional data.</p>
</sec>
<sec id="s2_6_6">
<label>2.6.6</label>
<title>K-Nearest neighbor algorithm</title>
<p>The KNN is utilized to predict the category of a given sample point by examining the categories of its K-nearest neighbors. This method, known for its simplicity and intuitive nature, does not necessitate explicit model training (<xref ref-type="bibr" rid="B23">23</xref>). In this study, The number of neighbors of the KNN model is set to 5 and a uniform weighting method is used. This setting simplifies the computational process of the model and allows the model to predict the classification of new samples based directly on the nearest few samples for effective classification.</p>
</sec>
</sec>
<sec id="s2_7">
<label>2.7</label>
<title>SHAP interpretability analysis</title>
<p>The SHAP is a technique utilized to interpret predictions generated by ML models, particularly those that are intricate and incorporate numerous features (<xref ref-type="bibr" rid="B24">24</xref>). The fundamental principle underlying this method involves the computation of the incremental impact of individual features on the model&#x2019;s output, enabling interpretation of the model&#x2019;s behavior at both a global and local scale. This is achieved through the development of an additive explanatory model that considers all features as contributors, thereby facilitating the calculation of the average incremental impact of each feature across all feasible feature combinations to derive a SHAP value for each feature, which provides both global and local interpretations, helping to understand which features are the main influences on model predictions, as well as the predictions of individual samples&#x2014;factors, as well as the prediction results for a single sample (<xref ref-type="bibr" rid="B25">25</xref>).</p>
</sec>
<sec id="s2_8">
<label>2.8</label>
<title>Statistical analysis</title>
<p>All data analyses in this study were performed using SPSS (17.0), R language (version 4.3.2), Matlab (version R2021a), and Python (version 3.7). The initial analysis of the data set involved the application of descriptive statistics. Data points adhering to a normal distribution were represented as mean &#xb1; standard deviation, while those deviating from normal distribution were represented as median (quartiles). Subsequently, the independent samples t-test was employed to compare two groups with normally distributed data. In contrast, the Mann-Whitney U test was utilized to compare two groups with non-normally distributed data. For count data, frequencies and percentages were used to characterize group variances, while the chi-square test or Fisher&#x2019;s exact probability method was employed to assess inter-group discrepancies. We solved the problem of sample imbalance by oversampling a small number of classes and thereby solving the sample imbalance problem through the SMOTE algorithm based on Matlab software. To construct the predictive model, the dataset was partitioned randomly into a training subset comprising 70% of the total data and a test subset comprising 30%. Subsequently, six ML algorithms were employed to train the model using the training subset data. During the model training process, a 10-fold cross-validation method is used to optimize the model parameters and prevent the occurrence of overfitting phenomenon. LASSO regression analysis was conducted utilizing the glmnet package [4.1.7] in R to analyze cleaned data and derive coefficient values of variables, logarithmic values of lambda, and regularized values of L1, followed by data visualization. The Boruta algorithm was implemented using Boruta 8.0.0 [4.1.7] in R. Interpretability analysis was carried out using the Python libraries shap=0.43.0. Statistical significance levels were established at <italic>P</italic>&lt;0.05.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results</title>
<sec id="s3_1">
<label>3.1</label>
<title>Patient characteristics</title>
<p>This study assembled a cohort of 502 lung cancer patients who did not have lung infections before undergoing chemotherapy. The median age of the patients was 65 years (range: 58-71 years), with 404 (80.48%) being male and 98 (19.52%) being female. We used the SMOTE algorithm for data imbalance. The original data of 502 cases contained 404 non-infected cases, 98 infected cases, and 19.52% of infected cases, and the processed data of 808 cases contained 404 non-infected cases, 404 infected cases, and 50.00% of infected cases. A comparison of baseline characteristics between the two groups revealed statistically significant differences in chemotherapy cycles, hospitalizations, WBC, pulmonary embolism, Gender, CREA, Histology, alcohol consumption, smoke, CHD, NEUT, LYM, NMR, NPR, IBIL, TBIL, and NLR (<italic>P</italic> &lt; 0.05), as shown in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Baseline characterization and comparison.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center" rowspan="2">Variables</th>
<th valign="top" align="center" rowspan="2">Total (n = 808)</th>
<th valign="top" colspan="2" align="center">Pulmonary infection after chemotherapy for lung cancer</th>
<th valign="top" align="center" rowspan="2">
<italic>P</italic>
</th>
</tr>
<tr>
<th valign="top" align="center">No (n = 404)</th>
<th valign="top" align="center">Yes (n = 404)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Age</td>
<td valign="top" align="left">65.00 [59.00, 70.00]</td>
<td valign="top" align="left">65.00 [58.00, 71.00]</td>
<td valign="top" align="left">65.00 [59.00, 69.00]</td>
<td valign="top" align="center">0.985</td>
</tr>
<tr>
<td valign="top" align="left">BMI</td>
<td valign="top" align="left">21.50 [19.70, 23.70]</td>
<td valign="top" align="left">21.80 [19.50, 24.10]</td>
<td valign="top" align="left">21.20 [19.80, 23.40]</td>
<td valign="top" align="center">0.129</td>
</tr>
<tr>
<td valign="top" align="left">Chemotherapy cycles</td>
<td valign="top" align="left">5.00 [2.00, 8.00]</td>
<td valign="top" align="left">3.00 [1.00, 5.00]</td>
<td valign="top" align="left">7.00 [5.00, 11.00]</td>
<td valign="top" align="center">
<bold>&lt;0.001</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">Hospitalizations</td>
<td valign="top" align="left">7.00 [4.00, 12.00]</td>
<td valign="top" align="left">4.50 [2.00, 7.00]</td>
<td valign="top" align="left">10.00 [6.00, 15.30]</td>
<td valign="top" align="center">
<bold>&lt;0.001</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">WBC</td>
<td valign="top" align="left">6.90 [5.49, 9.17]</td>
<td valign="top" align="left">6.56 [5.19, 8.78]</td>
<td valign="top" align="left">7.07 [5.93, 9.55]</td>
<td valign="top" align="center">
<bold>&lt;0.001</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">RBC</td>
<td valign="top" align="left">3.77 [3.30, 4.15]</td>
<td valign="top" align="left">3.76 [3.28, 4.17]</td>
<td valign="top" align="left">3.78 [3.33, 4.15]</td>
<td valign="top" align="center">0.943</td>
</tr>
<tr>
<td valign="top" align="left">HGB</td>
<td valign="top" align="left">112.00 [99.90, 125.00]</td>
<td valign="top" align="left">112.00 [99.00, 125.00]</td>
<td valign="top" align="left">112.00 [101.00, 124.00]</td>
<td valign="top" align="center">0.603</td>
</tr>
<tr>
<td valign="top" align="left">PLT</td>
<td valign="top" align="left">209.00 [160.00, 258.00]</td>
<td valign="top" align="left">208.00 [161.00,268.00]</td>
<td valign="top" align="left">212.00 [160.00,241.00]</td>
<td valign="top" align="center">0.357</td>
</tr>
<tr>
<td valign="top" align="left">NEUT</td>
<td valign="top" align="left">72.10 [64.30, 79.20]</td>
<td valign="top" align="left">70.60 [63.10, 78.30]</td>
<td valign="top" align="left">74.10 [66.00, 79.60]</td>
<td valign="top" align="center">
<bold>&lt;0.001</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">LYM</td>
<td valign="top" align="left">17.30 [12.00, 23.30]</td>
<td valign="top" align="left">18.80 [12.80, 25.10]</td>
<td valign="top" align="left">16.10 [11.50, 21.60]</td>
<td valign="top" align="center">
<bold>&lt;0.001</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">MONO</td>
<td valign="top" align="left">7.30 [5.40, 9.48]</td>
<td valign="top" align="left">7.40 [5.50, 9.73]</td>
<td valign="top" align="left">7.11 [5.20, 9.10]</td>
<td valign="top" align="center">0.147</td>
</tr>
<tr>
<td valign="top" align="left">NLR</td>
<td valign="top" align="left">4.38 [2.83, 7.09]</td>
<td valign="top" align="left">3.74 [2.53, 6.10]</td>
<td valign="top" align="left">4.90 [3.33, 7.67]</td>
<td valign="top" align="center">
<bold>&lt;0.001</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">NMR</td>
<td valign="top" align="left">10.10 [7.11, 14.40]</td>
<td valign="top" align="left">9.34 [6.74, 13.20]</td>
<td valign="top" align="left">10.90 [7.57, 15.40]</td>
<td valign="top" align="center">
<bold>&lt;0.001</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">NPR</td>
<td valign="top" align="left">0.35 [0.27, 0.43]</td>
<td valign="top" align="left">0.33 [0.26, 0.42]</td>
<td valign="top" align="left">0.36 [0.29, 0.44]</td>
<td valign="top" align="center">
<bold>0.001</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">IBIL</td>
<td valign="top" align="left">7.30 [5.70, 9.39]</td>
<td valign="top" align="left">7.00 [5.31, 9.42]</td>
<td valign="top" align="left">7.45 [6.10, 9.31]</td>
<td valign="top" align="center">
<bold>0.007</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">ALT</td>
<td valign="top" align="left">18.00 [12.70, 26.30]</td>
<td valign="top" align="left">17.90 [13.00, 28.30]</td>
<td valign="top" align="left">18.20 [12.30, 24.90]</td>
<td valign="top" align="center">0.218</td>
</tr>
<tr>
<td valign="top" align="left">AST</td>
<td valign="top" align="left">23.40 [19.40, 29.40]</td>
<td valign="top" align="left">23.80 [18.90, 29.80]</td>
<td valign="top" align="left">23.20 [19.80, 28.70]</td>
<td valign="top" align="center">0.858</td>
</tr>
<tr>
<td valign="top" align="left">TBIL</td>
<td valign="top" align="left">9.89 [7.63, 12.60]</td>
<td valign="top" align="left">9.46 [7.22, 12.70]</td>
<td valign="top" align="left">10.10 [8.08, 12.30]</td>
<td valign="top" align="center">
<bold>0.013</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">DBIL</td>
<td valign="top" align="left">2.40 [1.59, 3.40]</td>
<td valign="top" align="left">2.30 [1.50, 3.43]</td>
<td valign="top" align="left">2.50 [1.63, 3.38]</td>
<td valign="top" align="center">0.199</td>
</tr>
<tr>
<td valign="top" align="left">TP</td>
<td valign="top" align="left">66.80 [62.30, 69.90]</td>
<td valign="top" align="left">66.30 [61.70, 71.10]</td>
<td valign="top" align="left">66.90 [62.50, 69.20]</td>
<td valign="top" align="center">0.636</td>
</tr>
<tr>
<td valign="top" align="left">ALB</td>
<td valign="top" align="left">40.00 [36.50, 42.50]</td>
<td valign="top" align="left">39.90 [36.50, 42.70]</td>
<td valign="top" align="left">40.20 [36.80, 42.30]</td>
<td valign="top" align="center">0.764</td>
</tr>
<tr>
<td valign="top" align="left">GLB</td>
<td valign="top" align="left">26.30 [23.50, 29.40]</td>
<td valign="top" align="left">26.30 [22.30, 30.60]</td>
<td valign="top" align="left">26.30 [24.30, 28.60]</td>
<td valign="top" align="center">0.897</td>
</tr>
<tr>
<td valign="top" align="left">A/G</td>
<td valign="top" align="left">1.54 [1.31, 1.75]</td>
<td valign="top" align="left">1.52 [1.26, 1.81]</td>
<td valign="top" align="left">1.55 [1.34, 1.71]</td>
<td valign="top" align="center">0.943</td>
</tr>
<tr>
<td valign="top" align="left">Urea</td>
<td valign="top" align="left">5.89 [4.74, 7.56]</td>
<td valign="top" align="left">5.73 [4.58, 7.20]</td>
<td valign="top" align="left">6.05 [4.99, 7.64]</td>
<td valign="top" align="center">0.059</td>
</tr>
<tr>
<td valign="top" align="left">CREA</td>
<td valign="top" align="left">78.80 [66.00, 92.30]</td>
<td valign="top" align="left">76.60 [63.70, 91.90]</td>
<td valign="top" align="left">82.00 [68.60, 93.10]</td>
<td valign="top" align="center">
<bold>0.004</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">UA</td>
<td valign="top" align="left">330.00 [278.00,394.00]</td>
<td valign="top" align="left">326.00 [265.00,398.00]</td>
<td valign="top" align="left">332.00[288.00,393.00]</td>
<td valign="top" align="center">0.180</td>
</tr>
<tr>
<td valign="top" align="left">CEA</td>
<td valign="top" align="left">3.69 [2.11, 9.74]</td>
<td valign="top" align="left">3.70 [2.05, 9.43]</td>
<td valign="top" align="left">3.68 [2.25, 9.81]</td>
<td valign="top" align="center">0.556</td>
</tr>
<tr>
<th valign="top" align="left">PE</th>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="center">&lt;0.001</th>
</tr>
<tr>
<td valign="top" align="left">No</td>
<td valign="top" align="left">608 (75.20%)</td>
<td valign="top" align="left">353 (87.40%)</td>
<td valign="top" align="left">255 (63.10%)</td>
<td valign="top" align="center"/>
</tr>
<tr>
<td valign="top" align="left">Yes</td>
<td valign="top" align="left">200 (24.80%)</td>
<td valign="top" align="left">51 (12.60%)</td>
<td valign="top" align="left">149 (36.90%)</td>
<td valign="top" align="center"/>
</tr>
<tr>
<th valign="top" align="left">Gender</th>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="center">&lt;0.001</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Male</td>
<td valign="top" align="left">683 (84.50%)</td>
<td valign="top" align="left">315 (78.00%)</td>
<td valign="top" align="left">368 (91.10%)</td>
<td valign="top" align="center"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Female</td>
<td valign="top" align="left">125 (15.50%)</td>
<td valign="top" align="left">89 (22.00%)</td>
<td valign="top" align="left">36 (8.90%)</td>
<td valign="top" align="center"/>
</tr>
<tr>
<th valign="top" align="left">Drink</th>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="center">&lt;0.001</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;No</td>
<td valign="top" align="left">683 (84.50%)</td>
<td valign="top" align="left">377 (93.30%)</td>
<td valign="top" align="left">306 (75.70%)</td>
<td valign="top" align="center"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Yes</td>
<td valign="top" align="left">125 (15.50%)</td>
<td valign="top" align="left">27 (6.70%)</td>
<td valign="top" align="left">98 (24.30%)</td>
<td valign="top" align="center"/>
</tr>
<tr>
<th valign="top" align="left">Smoke</th>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="center">&lt;0.001</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;No</td>
<td valign="top" align="left">518 (64.10%)</td>
<td valign="top" align="left">322 (79.70%)</td>
<td valign="top" align="left">196 (48.50%)</td>
<td valign="top" align="center"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Yes</td>
<td valign="top" align="left">290 (35.90%)</td>
<td valign="top" align="left">82 (20.30%)</td>
<td valign="top" align="left">208 (51.5%)</td>
<td valign="top" align="center"/>
</tr>
<tr>
<th valign="top" align="left">Diabetes</th>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="center">0.999</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;No</td>
<td valign="top" align="left">731 (90.50%)</td>
<td valign="top" align="left">366 (90.60%)</td>
<td valign="top" align="left">365 (90.30%)</td>
<td valign="top" align="center"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Yes</td>
<td valign="top" align="left">77 (9.50%)</td>
<td valign="top" align="left">38 (9.40%)</td>
<td valign="top" align="left">39 (9.70%)</td>
<td valign="top" align="center"/>
</tr>
<tr>
<th valign="top" align="left">Hypertension</th>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="center">0.667</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;No</td>
<td valign="top" align="left">636 (78.70%)</td>
<td valign="top" align="left">321 (79.50%)</td>
<td valign="top" align="left">315 (78.00%)</td>
<td valign="top" align="center"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Yes</td>
<td valign="top" align="left">172 (21.30%)</td>
<td valign="top" align="left">83 (20.50%)</td>
<td valign="top" align="left">89 (22.00%)</td>
<td valign="top" align="center"/>
</tr>
<tr>
<th valign="top" align="left">CHD</th>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="center">0.008</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;No</td>
<td valign="top" align="left">759 (93.90%)</td>
<td valign="top" align="left">370 (91.60%)</td>
<td valign="top" align="left">389 (96.30%)</td>
<td valign="top" align="center"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Yes</td>
<td valign="top" align="left">49 (6.10%)</td>
<td valign="top" align="left">34 (8.40%)</td>
<td valign="top" align="left">15 (3.70%)</td>
<td valign="top" align="center"/>
</tr>
<tr>
<th valign="top" align="left">Stage</th>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="center">0.053</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Stage I</td>
<td valign="top" align="left">29 (3.60%)</td>
<td valign="top" align="left">21 (5.20%)</td>
<td valign="top" align="left">8 (2.00%)</td>
<td valign="top" align="center"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Stage II</td>
<td valign="top" align="left">96 (11.90%)</td>
<td valign="top" align="left">46 (11.40%)</td>
<td valign="top" align="left">50 (12.40%)</td>
<td valign="top" align="center"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Stage III</td>
<td valign="top" align="left">330 (40.80%)</td>
<td valign="top" align="left">171 (42.30%)</td>
<td valign="top" align="left">159 (39.40%)</td>
<td valign="top" align="center"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Stage IV</td>
<td valign="top" align="left">353 (43.70%)</td>
<td valign="top" align="left">166 (41.10%)</td>
<td valign="top" align="left">187 (46.30%)</td>
<td valign="top" align="center"/>
</tr>
<tr>
<th valign="top" align="left">Histology</th>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="center">0.005</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Adenocarcinoma</td>
<td valign="top" align="left">327 (40.50%)</td>
<td valign="top" align="left">183 (45.30%)</td>
<td valign="top" align="left">144 (35.60%)</td>
<td valign="top" align="center"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Squamous</td>
<td valign="top" align="left">333 (41.20%)</td>
<td valign="top" align="left">151 (37.40%)</td>
<td valign="top" align="left">182 (45.00%)</td>
<td valign="top" align="center"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;SCLC</td>
<td valign="top" align="left">135 (16.70%)</td>
<td valign="top" align="left">60 (14.90%)</td>
<td valign="top" align="left">75 (18.60%)</td>
<td valign="top" align="center"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Other lung cancers</td>
<td valign="top" align="left">13 (1.60%)</td>
<td valign="top" align="left">10 (2.50%)</td>
<td valign="top" align="left">1 (0.70%)</td>
<td valign="top" align="center"/>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Statistically significant differences are marked with bold font.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Predictor screening</title>
<p>A total of 808 patients undergoing chemotherapy for lung cancer after data imbalance were divided into a training group consisting of 565 patients and a test group consisting of 243 patients, following a ratio of 7:3. Statistical analysis revealed no significant differences between the two groups (<xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>). Utilizing the Boruta algorithm, an extension of the RF algorithm, enabled the identification of the actual feature set by accurately estimating the importance of each feature. The Boruta algorithm identified 35 key factors, including Drink, Smoke, Chemotherapy cycles, Hospitalizations, PE, NEUT, LYM, MONO, NLR, and NMR, etc (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2A</bold>
</xref>). In contrast, LASSO regression serves as a compression estimation method that accomplishes variable selection and complexity adjustment through the formulation of an optimization objective function incorporating penalty terms. In this study, LASSO regression was utilized to identify characteristic factors such as Gender, Drink, Smoke, Chemotherapy cycles, PE, NEUT, NLR, NMR, and AST (<xref ref-type="fig" rid="f2">
<bold>Figures&#xa0;2B, C</bold>
</xref>). Through a comparative analysis of the outcomes obtained from LASSO regression and Boruta algorithm screening, we identified a common subset of feature variables selected by both methods. These selected features were ultimately utilized in the construction of the model and consisted of Gender, Drink, Smoke, Chemotherapy cycles, PE, NEUT, AST, NLR, and NMR (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2D</bold>
</xref>).</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Training set and Test set variability analysis.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Variable</th>
<th valign="top" align="left">Total (N = 808)</th>
<th valign="top" align="left">Train set (N = 565)</th>
<th valign="top" align="left">Test set (N = 243)</th>
<th valign="top" align="left">
<italic>P</italic>
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Age</td>
<td valign="top" align="left">65.00 [59.00, 70.00]</td>
<td valign="top" align="left">65.00 [59.00, 70.00]</td>
<td valign="top" align="left">65.00 [58.00, 70.00]</td>
<td valign="top" align="left">0.727</td>
</tr>
<tr>
<td valign="top" align="left">BMI</td>
<td valign="top" align="left">21.50 [19.7, 23.70]</td>
<td valign="top" align="left">21.50 [19.60, 23.70]</td>
<td valign="top" align="left">21.60 [20.00, 23.60]</td>
<td valign="top" align="left">0.497</td>
</tr>
<tr>
<td valign="top" align="left">Chemotherapy cycles</td>
<td valign="top" align="left">5.00 [2.00, 8.00]</td>
<td valign="top" align="left">5.00 [2.00, 8.00]</td>
<td valign="top" align="left">5.00 [2.00, 9.00]</td>
<td valign="top" align="left">0.737</td>
</tr>
<tr>
<td valign="top" align="left">Hospitalizations</td>
<td valign="top" align="left">7.00[4.00, 12.00]</td>
<td valign="top" align="left">7.00 [4.00, 12.00]</td>
<td valign="top" align="left">7.00 [3.00, 12.00]</td>
<td valign="top" align="left">0.927</td>
</tr>
<tr>
<td valign="top" align="left">WBC</td>
<td valign="top" align="left">6.90 [5.49, 9.17]</td>
<td valign="top" align="left">6.82 [5.49, 8.94]</td>
<td valign="top" align="left">7.37 [5.47, 9.89]</td>
<td valign="top" align="left">0.075</td>
</tr>
<tr>
<td valign="top" align="left">RBC</td>
<td valign="top" align="left">3.77 [3.30, 4.15]</td>
<td valign="top" align="left">3.77 [3.33, 4.14]</td>
<td valign="top" align="left">3.76 [3.29, 4.19]</td>
<td valign="top" align="left">0.644</td>
</tr>
<tr>
<td valign="top" align="left">HGB</td>
<td valign="top" align="left">112.00 [99.90, 125.00]</td>
<td valign="top" align="left">112.00 [99.30, 124.00]</td>
<td valign="top" align="left">112.00 [100.00, 125.00]</td>
<td valign="top" align="left">0.810</td>
</tr>
<tr>
<td valign="top" align="left">PLT</td>
<td valign="top" align="left">209.00 [160.00, 258.00]</td>
<td valign="top" align="left">210.00 [159.00, 262.00]</td>
<td valign="top" align="left">209.00 [163.00, 243.00]</td>
<td valign="top" align="left">0.411</td>
</tr>
<tr>
<td valign="top" align="left">NEUT</td>
<td valign="top" align="left">72.10 [64.30, 79.20]</td>
<td valign="top" align="left">71.60 [63.60, 78.90]</td>
<td valign="top" align="left">73.30 [66.10, 80.10]</td>
<td valign="top" align="left">0.073</td>
</tr>
<tr>
<td valign="top" align="left">LYM</td>
<td valign="top" align="left">17.30 [12.00, 23.30]</td>
<td valign="top" align="left">17.50 [12.20, 23.70]</td>
<td valign="top" align="left">16.80 [11.40, 23.00]</td>
<td valign="top" align="left">0.249</td>
</tr>
<tr>
<td valign="top" align="left">MONO</td>
<td valign="top" align="left">7.30 [5.40, 9.48]</td>
<td valign="top" align="left">7.23 [5.50, 9.50]</td>
<td valign="top" align="left">7.40 [5.05, 9.30]</td>
<td valign="top" align="left">0.456</td>
</tr>
<tr>
<td valign="top" align="left">NLR</td>
<td valign="top" align="left">4.38 [2.83, 7.09]</td>
<td valign="top" align="left">4.27 [2.73, 6.89]</td>
<td valign="top" align="left">4.46 [2.97, 7.60]</td>
<td valign="top" align="left">0.137</td>
</tr>
<tr>
<td valign="top" align="left">NMR</td>
<td valign="top" align="left">10.10 [6.79, 14.03]</td>
<td valign="top" align="left">10.20 [7.00, 14.40]</td>
<td valign="top" align="left">10.00 [7.46, 14.50]</td>
<td valign="top" align="left">0.375</td>
</tr>
<tr>
<td valign="top" align="left">NPR</td>
<td valign="top" align="left">0.35 [0.27, 0.43]</td>
<td valign="top" align="left">0.34 [0.27, 0.43]</td>
<td valign="top" align="left">0.36 [0.28, 0.44]</td>
<td valign="top" align="left">0.108</td>
</tr>
<tr>
<td valign="top" align="left">IBIL</td>
<td valign="top" align="left">7.30 [5.70, 9.39]</td>
<td valign="top" align="left">7.24 [5.79, 9.30]</td>
<td valign="top" align="left">7.40 [5.60, 9.72]</td>
<td valign="top" align="left">0.700</td>
</tr>
<tr>
<td valign="top" align="left">ALT</td>
<td valign="top" align="left">18.00 [12.70, 26.30]</td>
<td valign="top" align="left">18.20 [12.40, 27.00]</td>
<td valign="top" align="left">17.40 [13.30, 24.30]</td>
<td valign="top" align="left">0.853</td>
</tr>
<tr>
<td valign="top" align="left">AST</td>
<td valign="top" align="left">23.40 [19.40, 29.40]</td>
<td valign="top" align="left">23.40 [19.30, 29.00]</td>
<td valign="top" align="left">23.50 [19.70, 29.70]</td>
<td valign="top" align="left">0.805</td>
</tr>
<tr>
<td valign="top" align="left">TBIL</td>
<td valign="top" align="left">9.89 [7.63, 12.60]</td>
<td valign="top" align="left">9.80 [7.70, 12.40]</td>
<td valign="top" align="left">9.90 [7.50, 12.80]</td>
<td valign="top" align="left">0.955</td>
</tr>
<tr>
<td valign="top" align="left">DBIL</td>
<td valign="top" align="left">2.40 [1.59, 3.40]</td>
<td valign="top" align="left">2.44 [1.60, 3.40]</td>
<td valign="top" align="left">2.30 [1.48, 3.39]</td>
<td valign="top" align="left">0.467</td>
</tr>
<tr>
<td valign="top" align="left">TP</td>
<td valign="top" align="left">66.80 [62.30, 69.90]</td>
<td valign="top" align="left">66.70 [62.20, 70.00]</td>
<td valign="top" align="left">67.20 [62.40, 69.80]</td>
<td valign="top" align="left">0.939</td>
</tr>
<tr>
<td valign="top" align="left">ALB</td>
<td valign="top" align="left">40.00 [36.50, 42.50]</td>
<td valign="top" align="left">39.90 [36.70, 42.40]</td>
<td valign="top" align="left">40.20 [36.30, 42.70]</td>
<td valign="top" align="left">0.931</td>
</tr>
<tr>
<td valign="top" align="left">GLB</td>
<td valign="top" align="left">26.30 [23.50, 29.40]</td>
<td valign="top" align="left">26.30 [23.70, 29.10]</td>
<td valign="top" align="left">26.40 [22.80, 30.00]</td>
<td valign="top" align="left">0.967</td>
</tr>
<tr>
<td valign="top" align="left">A/G</td>
<td valign="top" align="left">1.54 [1.31, 1.75]</td>
<td valign="top" align="left">1.54 [1.32, 1.75]</td>
<td valign="top" align="left">1.55 [1.26, 1.77]</td>
<td valign="top" align="left">0.975</td>
</tr>
<tr>
<td valign="top" align="left">Urea</td>
<td valign="top" align="left">5.89 [4.74, 7.56]</td>
<td valign="top" align="left">5.92 [4.69, 7.63]</td>
<td valign="top" align="left">5.84 [4.89, 7.32]</td>
<td valign="top" align="left">0.540</td>
</tr>
<tr>
<td valign="top" align="left">CREA</td>
<td valign="top" align="left">78.80 [66.00, 92.30]</td>
<td valign="top" align="left">78.50 [66.60, 92.20]</td>
<td valign="top" align="left">79.80 [64.00, 92.90]</td>
<td valign="top" align="left">0.889</td>
</tr>
<tr>
<td valign="top" align="left">UA</td>
<td valign="top" align="left">330.00 [278.00, 394.00]</td>
<td valign="top" align="left">330.00 [279.00, 394.00]</td>
<td valign="top" align="left">330.00 [277.00, 395.00]</td>
<td valign="top" align="left">0.951</td>
</tr>
<tr>
<td valign="top" align="left">CEA</td>
<td valign="top" align="left">3.69 [2.11, 9.74]</td>
<td valign="top" align="left">3.69 [2.14, 9.87]</td>
<td valign="top" align="left">3.66 [2.03, 8.29]</td>
<td valign="top" align="left">0.447</td>
</tr>
<tr>
<th valign="top" align="left">Gender, n (%)</th>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left">0.298</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Male</td>
<td valign="top" align="left">683 (84.50%)</td>
<td valign="top" align="left">483 (85.50%)</td>
<td valign="top" align="left">200 (82.30%)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Female</td>
<td valign="top" align="left">125 (15.50%)</td>
<td valign="top" align="left">82 (14.50%)</td>
<td valign="top" align="left">43 (17.70%)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<th valign="top" align="left">Drink, n (%)</th>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left">0.385</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;No</td>
<td valign="top" align="left">683 (84.50%)</td>
<td valign="top" align="left">473 (83.70%)</td>
<td valign="top" align="left">210 (86.40%)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Yes</td>
<td valign="top" align="left">125 (15.50%)</td>
<td valign="top" align="left">92 (16.30%)</td>
<td valign="top" align="left">33 (13.60%)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<th valign="top" align="left">Smoke, n (%)</th>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left">0.087</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;No</td>
<td valign="top" align="left">518 (64.10%)</td>
<td valign="top" align="left">351 (62.10%)</td>
<td valign="top" align="left">167 (68.70%)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Yes</td>
<td valign="top" align="left">290 (35.90%)</td>
<td valign="top" align="left">214 (37.90%)</td>
<td valign="top" align="left">76 (31.30%)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<th valign="top" align="left">Diabetes, n (%)</th>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left">0.864</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;No</td>
<td valign="top" align="left">731 (90.50%)</td>
<td valign="top" align="left">510 (90.30%)</td>
<td valign="top" align="left">221 (90.90%)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Yes</td>
<td valign="top" align="left">77 (9.50%)</td>
<td valign="top" align="left">55 (9.70%)</td>
<td valign="top" align="left">22 (9.10%)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<th valign="top" align="left">Hypertension, n (%)</th>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left">0.371</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;No</td>
<td valign="top" align="left">636 (78.70%)</td>
<td valign="top" align="left">450 (79.60%)</td>
<td valign="top" align="left">186 (76.50%)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Yes</td>
<td valign="top" align="left">172 (21.30%)</td>
<td valign="top" align="left">115 (20.40%)</td>
<td valign="top" align="left">57 (23.50%)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<th valign="top" align="left">CHD, n (%)</th>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left">0.226</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;No</td>
<td valign="top" align="left">759 (93.90%)</td>
<td valign="top" align="left">535 (94.70%)</td>
<td valign="top" align="left">224 (92.20%)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Yes</td>
<td valign="top" align="left">49 (6.10%)</td>
<td valign="top" align="left">30 (5.30%)</td>
<td valign="top" align="left">19 (7.80%)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<th valign="top" align="left">Stage, n (%)</th>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left">0.779</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Stage I</td>
<td valign="top" align="left">29 (3.60%)</td>
<td valign="top" align="left">20 (3.50%)</td>
<td valign="top" align="left">9 (3.70%)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Stage II</td>
<td valign="top" align="left">96 (11.90%)</td>
<td valign="top" align="left">71 (12.60%)</td>
<td valign="top" align="left">25 (10.30%)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Stage III</td>
<td valign="top" align="left">330 (40.80%)</td>
<td valign="top" align="left">232 (41.10%)</td>
<td valign="top" align="left">98 (40.30%)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Stage IV</td>
<td valign="top" align="left">353 (43.70%)</td>
<td valign="top" align="left">242 (42.80%)</td>
<td valign="top" align="left">111 (45.70%)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<th valign="top" align="left">Histology, n (%)</th>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left">0.537</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Adenocarcinoma</td>
<td valign="top" align="left">327 (40.50%)</td>
<td valign="top" align="left">221 (39.10%)</td>
<td valign="top" align="left">106 (43.60%)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Squamous</td>
<td valign="top" align="left">333 (41.20%)</td>
<td valign="top" align="left">241 (42.70%)</td>
<td valign="top" align="left">92 (37.90%)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;SCLC</td>
<td valign="top" align="left">135 (16.70%)</td>
<td valign="top" align="left">93 (16.50%)</td>
<td valign="top" align="left">42 (17.30%)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Other lung cancers</td>
<td valign="top" align="left">13 (1.60%)</td>
<td valign="top" align="left">10 (1.80%)</td>
<td valign="top" align="left">3 (1.20%)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<th valign="top" align="left">PE</th>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left">0.237</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;No</td>
<td valign="top" align="left">608 (75.20%)</td>
<td valign="top" align="left">418 (74.00%)</td>
<td valign="top" align="left">190 (78.20%)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Yes</td>
<td valign="top" align="left">200 (24.80%)</td>
<td valign="top" align="left">147 (26.00%)</td>
<td valign="top" align="left">53 (21.80%)</td>
<td valign="top" align="left"/>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Predictor screening results. <bold>(A)</bold> Boruta; <bold>(B)</bold> Factor screening based on the LASSO regression model, with the left dashed line indicating the best lambda value for the evaluation metrics (lambda. min) and the right dashed line indicating the lambda value for the model where the evaluation metrics are in the range of the best value by one standard error (lambda.1se); <bold>(C)</bold> LASSO regression model screening variable trajectories; <bold>(D)</bold> common predictors between Boruta and LASSO.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-14-1403392-g002.tif"/>
</fig>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Model performance</title>
<p>In the training dataset, the RF model exhibited superior predictive performance with an AUC of 1.00, indicating a high level of accuracy in prediction. In contrast, the AUC values for the remaining five models were as follows: 0.888, 95%CI(0.863-0.911) for LR, 0.822, 95%CI(0.791-0.852) for GNB, 0.792, 95%CI(0.760-0.825) for MLP, 0.719, 95%CI(0.681-0.758) for SVM, and 1.000, 95%CI(NaN- NaN) for KNN (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3A</bold>
</xref>). The F1 scores for these models were as follows: LR 0.816, RF 0.998, GNB 0.756, MLP 0.736, SVM 0.679, and KNN nan. In the test set, the AUC values for LR, RF, GNB, MLP, SVM, and KNN were 0.876(95%CI(0.806-0.953)), 0.923(95%CI(0.866-0.979)), 0.817(95%CI(0.726-0.909)), 0.777(95%CI(0.674-0.880)), 0.709(95%CI(0.590-0.828)), and 0.837(95%CI(0.750-0.923)), respectively (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3B</bold>
</xref>). The corresponding F1 scores were 0.791, 0.837, 0.747, 0.716, 0.658, and nan for LR, RF, GNB, MLP, SVM, and KNN, respectively. The forest plot comparing the AUC scores of the six ML models is presented in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3C</bold>
</xref>. In this study, the accuracy, sensitivity, specificity, positive predictive value, negative predictive value, and kappa value of each model were computed and compared (<xref ref-type="fig" rid="f3">
<bold>Figures&#xa0;3D, E</bold>
</xref>). While the RF model exhibited exceptional performance on the training set, the Logistic Regression model was ultimately selected as the optimal model due to concerns regarding potential overfitting.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>The performance and comparison of six different predictive models. <bold>(A)</bold> The training set ROC curve; <bold>(B)</bold> The test set ROC curve; <bold>(C)</bold> Forest plot of AUC values; <bold>(D)</bold> Evaluation metrics for the training set; <bold>(E)</bold> Evaluation metrics for the test set.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-14-1403392-g003.tif"/>
</fig>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>The logistic regression model</title>
<p>The results of the univariate logistic analysis are summarized in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table&#xa0;1</bold>
</xref>. 12 variables were statistically significant: Gender, Drink, Smoke, CHD, Chemotherapy cycles, Hospitalizations, PE, NEUT, LYM, NLR, NMR, and CEA. <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref> presents the coefficients and odds ratios (OR) for the nine predictor variables included in the model. The logistic equation was as follows:</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Risk factors and their parameters of the logistic model.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Variables</th>
<th valign="top" align="left">Coefficients</th>
<th valign="top" align="left">OR(95%CI)</th>
<th valign="top" align="left">p</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Intercept</td>
<td valign="top" align="left">-2.954</td>
<td valign="top" align="left">0.052(0.006-0.395)</td>
<td valign="top" align="left">
<bold>0.006</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">Gender</td>
<td valign="top" align="left">-0.424</td>
<td valign="top" align="left">0.655(0.321-1.297)</td>
<td valign="top" align="left">0.233</td>
</tr>
<tr>
<td valign="top" align="left">Drink</td>
<td valign="top" align="left">-0.049</td>
<td valign="top" align="left">0.952(0.464-1.963)</td>
<td valign="top" align="left">0.893</td>
</tr>
<tr>
<td valign="top" align="left">Smoke</td>
<td valign="top" align="left">1.754</td>
<td valign="top" align="left">5.776(3.292-10.375)</td>
<td valign="top" align="left">
<bold>&lt;0.001</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">Chemotherapy cycles</td>
<td valign="top" align="left">0.395</td>
<td valign="top" align="left">1.484(1.380-1.606)</td>
<td valign="top" align="left">
<bold>&lt;0.001</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">PE</td>
<td valign="top" align="left">1.417</td>
<td valign="top" align="left">4.123(2.389-7.274)</td>
<td valign="top" align="left">
<bold>&lt;0.001</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">NLR</td>
<td valign="top" align="left">0.083</td>
<td valign="top" align="left">1.087(1.007-1.174)</td>
<td valign="top" align="left">
<bold>0.034</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">NMR</td>
<td valign="top" align="left">0.017</td>
<td valign="top" align="left">1.018(0.998-1.038)</td>
<td valign="top" align="left">0.077</td>
</tr>
<tr>
<td valign="top" align="left">AST</td>
<td valign="top" align="left">0.009</td>
<td valign="top" align="left">1.009(1.003-1.019)</td>
<td valign="top" align="left">
<bold>0.020</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">NEUT</td>
<td valign="top" align="left">-0.008</td>
<td valign="top" align="left">0.992(0.962-1.026)</td>
<td valign="top" align="left">0.639</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>OR, odds ratio; CI, confidence interval.</p>
</fn>
<fn>
<p>Statistically significant differences are marked with bold font.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>y = - 2.954 - 0.424&#xd7;Gender - 0.049&#xd7;Drink + 1.754&#xd7;Smoke + 0.395&#xd7;Chemotherapy cycles + 1.417&#xd7;PE + 0.083&#xd7;NLR + 0.017&#xd7;NMR + 0.009&#xd7;AST - 0.008&#xd7;NEUT. In this study, we evaluated the prediction accuracy and calibration of the model by calibration curve analysis of the training and test sets. The calibration curve results showed that the model in the training set had high prediction accuracy with a Somers&#x2019; D coefficient of 0.777 and an area under the ROC curve of 0.888, indicating that the model had excellent discriminative ability (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4A</bold>
</xref>). In addition, the logistic regression calibration slope of the training set model was close to the ideal value of 1.000, with an intercept of 0.000, showing excellent calibration. The Brier score of 0.134 reflected the high reliability of the model predictions. In contrast, the model in the test set maintained a high discriminative power with an area under the ROC curve of 0.876, although there was a slight decrease in prediction accuracy (Somers&#x2019; D coefficient of 0.751) (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4B</bold>
</xref>). The decision curve for the training set (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4C</bold>
</xref>) shows that the model provides significantly higher net gains than the baseline strategy when the threshold probabilities are between 0.1 and 0.9. On the test set (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4D</bold>
</xref>), the model similarly demonstrates good net returns, especially in the range of threshold probabilities from 0.1 to 0.85, where it maintains a high level of net returns. The confusion matrix results show the difference in the model&#x2019;s performance on different datasets. In the training set (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4E</bold>
</xref>), the model correctly identified 320 true negatives and 283 true positives, and misidentified 42 false positives and 82 false negatives, with a true positive rate (sensitivity) of 77.5% and a true negative rate (specificity) of 88.4%. In the test set (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4F</bold>
</xref>), the model correctly identified 32 true negatives and 27 true positives and misidentified 10 false positives and 12 false negatives, for a true positive rate of 69.2% and a true negative rate of 76.2%. Finally, we plotted clinical impact curves (CICs) to assess the net gain in clinical utility and applicability of the model with the highest diagnostic value. The clinical impact curves (<xref ref-type="fig" rid="f4">
<bold>Figures&#xa0;4G, H</bold>
</xref>) provide information on the ability of the models to predict high-risk patients at different cost-benefit ratio thresholds. The curves for both the training and test sets show that when the threshold probability is greater than the 65% predictive score probability value, the predictive model&#x2019;s determination of those at high risk of developing an infection in the lungs after chemotherapy is highly matched to those who actually develop an infection, confirming that the predictive model is clinically highly effective.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Comprehensive evaluation of the logistic regression model. <bold>(A)</bold> Calibration curve for the training set; <bold>(B)</bold> Calibration curve for the test set; <bold>(C)</bold> Decision curve analysis for the training set; <bold>(D)</bold> Decision curve analysis for the test set; <bold>(E)</bold> Confounding matrix for the training set; <bold>(F)</bold> Confounding matrix for the test set; <bold>(G)</bold> Clinical impact curve for the training set; <bold>(H)</bold> Clinical impact curve for the test set.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-14-1403392-g004.tif"/>
</fig>
</sec>
<sec id="s3_5">
<label>3.5</label>
<title>SHAP-based model interpretability analysis</title>
<p>This study assessed the relative significance of various factors influencing the susceptibility to lung infections following chemotherapy in patients with lung cancer. <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5A</bold>
</xref> visually represents this ranking, with each point denoting a sample and the color gradient from blue to red indicating the magnitude of the sample eigenvalues. The vertical axis displays the importance ranking of features, along with the correlation and distribution of each eigenvalue with the SHAP value. The impact of the top nine features in the importance ranking on prediction outcomes is illustrated in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5B</bold>
</xref>. Specifically, Chemotherapy cycles, Smoke, and PE exhibit positive contributions to the predictive results, while NEUT demonstrate negative influences on the model&#x2019;s output. <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5B</bold>
</xref> illustrates the hierarchical significance of features in the logistic regression model. The vertical axis displays individual features in descending order of importance, while the horizontal axis represents average SHAP values. The analysis reveals that Chemotherapy cycles, Smoke, PE, NMR, and NLR are the top five features ranked by importance, indicating their critical influence on the presence of a lung infection. To enhance comprehension of the model&#x2019;s decision-making process at the individual level, we conducted a detailed interpretability analysis on two representative samples, as illustrated in <xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5C, D</bold>
</xref>. By visualizing the SHAP values of these samples, we could discern the impact of each feature on the model&#x2019;s predictions for these specific instances.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Interpretability analysis of logistic regression models. <bold>(A)</bold> SHAP dendrogram of features of the logistic regression model. <bold>(B)</bold> Importance ranking plot of features of the logistic regression model. <bold>(C, D)</bold> Interpretability analysis of 2 independent samples.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-14-1403392-g005.tif"/>
</fig>
</sec>
<sec id="s3_6">
<label>3.6</label>
<title>Construction of nomograms</title>
<p>In this study, two nomograms were constructed, integrating nine important predictor variables such as alcohol consumption, smoking, and chemotherapy cycle to visually assess the risk of lung infection after chemotherapy. <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6A</bold>
</xref> shows an interactive nomogram with a score of 3.51 for the example patient, corresponding to a 94.5% probability of infection, providing a quick and easy-to-interpret risk assessment. <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6B</bold>
</xref> illustrates a dynamic nomogram with different risk profiles derived from 10 combinations of variables.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Construct two different nomograms. <bold>(A)</bold> Interactive Nomogram. <bold>(B)</bold> Dynamic Nomogram showing risk profiles for ten scenarios.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-14-1403392-g006.tif"/>
</fig>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<label>4</label>
<title>Discussion</title>
<p>This research investigated the predictive factors associated with post-chemotherapy lung infection in patients with lung cancer and developed a logistic regression-based predictive model that effectively estimates the likelihood of lung infection following chemotherapy. By employing meticulous feature selection and conducting multi-model comparative validation, this study highlights the significance of various key predictors and offers a valuable tool to aid in clinical decision-making.</p>
<p>Zhou D et&#xa0;al. conducted a retrospective analysis of 244 non-small cell lung cancer (NSCLC) patients who underwent surgical interventions from June 2015 to January 2017. Through applying LASSO regression and logistic regression analyses, the researchers identified independent risk factors for postoperative pulmonary infection (PPI) in NSCLC patients and subsequently developed a predictive model based on these findings (<xref ref-type="bibr" rid="B26">26</xref>). Jong-Ho Kim and colleagues pioneered the application of ML techniques for the prognostication of postoperative pulmonary complications (PPCs), employing a suite of five algorithms, namely LR, random forests (RFs), light-gradient boosting machines (LightGBM), extreme-gradient boosting machines (XGBoost) and MLP for the construction and assessment of predictive models (<xref ref-type="bibr" rid="B27">27</xref>). Xue et&#xa0;al. established a predictive model utilizing preoperative and intraoperative data to detect the likelihood of postoperative pneumonia. Their research delved into the application of machine learning in predicting a range of postoperative complications, including pneumonia, within the context of PPCs. Nevertheless, the authors failed to emphasize unique characteristics and risk factors beyond pneumonia linked to PPCs, potentially diverting attention away from PPCs (<xref ref-type="bibr" rid="B28">28</xref>). While predictive models have been created for complications in lung cancer patients, there is a scarcity of predictive models utilizing ML algorithms for assessing the risk of lung infection following chemotherapy for lung cancer.</p>
<p>The dysregulation of the autoimmune system, exacerbated by chemotherapy-induced immune cell depletion, tumor cell infiltration, impaired antibody-complement generation, and dysregulation of the inflammatory system, disrupts immune homeostasis and heightens susceptibility to concurrent lung infections (<xref ref-type="bibr" rid="B29">29</xref>&#x2013;<xref ref-type="bibr" rid="B31">31</xref>). This risk is further compounded in individuals with comorbidities such as chronic bronchitis, chronic obstructive pulmonary disease, interstitial lung disease, pulmonary atelectasis, and other organic diseases (<xref ref-type="bibr" rid="B32">32</xref>, <xref ref-type="bibr" rid="B33">33</xref>). The occurrence of lung infection during chemotherapy is a prevalent and challenging complication that hinders the efficacy of treatment and exacerbates the health status of patients, ultimately impacting their prognosis and increasing the financial burden of medical care. As such, our research holds significant clinical importance in examining the determinants of lung infection during chemotherapy and implementing timely and efficient interventions for patients with lung cancer.</p>
<p>This study employed a dual methodology of Boruta&#x2019;s algorithm and LASSO regression to identify predictors for accurate feature selection and model stability. The selected features encompassed variables such as alcohol consumption status, smoking habits, chemotherapy cycles, hospitalization frequency, presence of lung pleural fluid, neutrophil count, AST, NLR, and NMR, all of which have demonstrated significant correlations with the prognosis of lung cancer patients in prior research. Wei Guo et&#xa0;al. colleagues created a predictive model utilizing artificial neural network (ANN) technology to forecast infection rates in lung cancer patients undergoing chemotherapy (<xref ref-type="bibr" rid="B34">34</xref>). The researchers employed a logistic regression (LR) model to analyze the data and identify statistically significant variables. Their results indicated a positive correlation between length of hospital stay and infection risk, which aligns with our research findings. However, the researchers discovered that a prior diagnosis of diabetes was linked to an increased likelihood of lung infection, a finding that did not align with our results. This discrepancy may be attributed to the limited sample size of the previous study, which only included 80 cases. Zhouzhou Ding et&#xa0;al. explored the risk factors for PPI in patients with non-small cell lung cancer (NSCLC), developed a risk model, and conducted predictive modeling for PPI. Their research revealed that the chemotherapy cycle, identified as an independent risk factor, had a notable impact on the occurrence of PPI (<xref ref-type="bibr" rid="B26">26</xref>). This is in general agreement with our findings. Our findings emphasize the importance of monitoring and managing these factors during chemotherapy management.</p>
<p>After comparing these models, it is observed that while the RF model exhibits superior performance in the training set, its propensity for overfitting necessitates the selection of the logistic regression model as the optimal choice due to its strong generalization capabilities in the external test set. Logistic regression models are favored for their predictive accuracy and interpretability, which are essential qualities for practical clinical implementation. The importance of constructing disease prediction models lies in identifying high-risk patients and mitigating the risk for individuals who may fall into the high-risk category, thereby benefiting patients overall. Consequently, the clinical interpretability of ML models holds significant value in medical practice. In this research, we utilized the SHAP method to provide both global and local interpretations of the ML model, enhancing its visual representation and transparency. Kaidi Gong et&#xa0;al. have observed that the SHAP method exhibits superior consistency and performance compared to conventional weight-based interpretation methods, and the SHAP algorithm demonstrates greater stability across various models. In contrast to the Local Interpretable Model-agnostic Explanations (LIME) method, SHAP demonstrates strong performance in both global and individual interpretation tasks, while LIME shows less consistency in individual analysis (<xref ref-type="bibr" rid="B35">35</xref>). Yasunobu Nohara and colleagues further substantiated that SHAP values exhibit superior interpretability compared to the coefficients of generalized linear regression models, as evidenced through a comparative analysis of interpretation outcomes with other established methodologies. Additionally, they found that SHAP summary plots offer more effective visualization of results than feature importance plots (<xref ref-type="bibr" rid="B36">36</xref>). The utilization of SHAP value analysis in this research offers a novel lens through which to comprehend the model&#x2019;s decision-making process. Through this method, we were able to elucidate the specific contributions of individual predictors to the model&#x2019;s decision-making, ultimately improving the transparency and interpretability of the model. Notably, factors such as chemotherapy cycle, smoking, PE, and NMR were underscored for their significance, consistent with prior research findings and reaffirmed their pivotal role in predicting post-chemotherapy lung infections.</p>
<p>Despite the results of this study, there are some limitations. Firstly, being a retrospective study, there is a potential for omitted data and selection bias to impact the results. Secondly, the small sample size of this study and the fact that the sample was collected from a single center may limit the generalizability of the findings. The potential incorporation of prospective design and multicenter data in future studies, coupled with integrating additional patient data and utilizing advanced machine learning techniques, is anticipated to enhance model performance. This improvement aims to validate the robustness and generalizability of the model, ultimately leading to the development of more personalized and precise treatment management strategies for patients with lung cancer.</p>
</sec>
<sec id="s5" sec-type="conclusions">
<label>5</label>
<title>Conclusion</title>
<p>This study has effectively developed a predictive tool utilizing logistic regression modeling to forecast lung infections following chemotherapy in lung cancer patients. The tool demonstrates high&#xa0;predictive accuracy and holds substantial clinical relevance. By identifying and assessing crucial predictors, this research establishes a valuable scientific foundation for the prevention and treatment of post-chemotherapy complications in lung cancer patients, ultimately enhancing patient survival quality and prognostic outcomes. Future work will focus on further validating the model&#x2019;s validity and exploring integrating these predictive tools into clinical practice to improve the prediction of treatment consequences in lung cancer patients.</p>
</sec>
</body>
<back>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material</bold>
</xref>. Further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s7" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>The studies involving humans were approved by The Medical Ethics Committee of the Central Hospital of Shaoyang (No: 20231207). The studies were conducted in accordance with the local legislation and institutional requirements. Written informed consent for participation was not required from the participants or the participants&#x2019; legal guardians/next of kin in accordance with the national legislation and institutional requirements.</p>
</sec>
<sec id="s8" sec-type="author-contributions">
<title>Author contributions</title>
<p>TS: Data curation, Investigation, Methodology, Project administration, Supervision, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. JL: Formal analysis, Software, Writing &#x2013; original draft. HQY: Methodology, Software, Validation, Writing &#x2013; review &amp; editing. XL: Conceptualization, Investigation, Writing &#x2013; review &amp; editing. HY: Resources, Software, Validation, Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s9" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research, authorship, and/or publication of this article.</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>We thank the Extreme Analytics platform for statistical support (<ext-link ext-link-type="uri" xlink:href="https://www.xsmartanalysis.com">https://www.xsmartanalysis.com</ext-link>); we would also like to thank Home for Researchers for proofreading the paper (<ext-link ext-link-type="uri" xlink:href="https://www.home-for-researchers.com">https://www.home-for-researchers.com</ext-link>).</p>
</ack>
<sec id="s10" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s11" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors&#xa0;and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fonc.2024.1403392/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fonc.2024.1403392/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Table_1.xlsx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Choi</surname> <given-names>E</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>SJ</given-names>
</name>
<name>
<surname>Aredo</surname> <given-names>JV</given-names>
</name>
<name>
<surname>Backhus</surname> <given-names>LM</given-names>
</name>
<name>
<surname>Wilkens</surname> <given-names>LR</given-names>
</name>
<name>
<surname>Su</surname> <given-names>CC</given-names>
</name>
<etal/>
</person-group>. <article-title>The survival impact of second primary lung cancer in patients with lung cancer</article-title>. <source>J Natl Cancer Institute</source>. (<year>2022</year>) <volume>114</volume>:<page-range>618&#x2013;25</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/jnci/djab224</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Aberle</surname> <given-names>DR</given-names>
</name>
<name>
<surname>Black</surname> <given-names>WC</given-names>
</name>
<name>
<surname>Chiles</surname> <given-names>C</given-names>
</name>
<name>
<surname>Church</surname> <given-names>TR</given-names>
</name>
<name>
<surname>Gareen</surname> <given-names>IF</given-names>
</name>
<name>
<surname>Gierada</surname> <given-names>DS</given-names>
</name>
<etal/>
</person-group>. <article-title>Lung cancer incidence and mortality with extended follow-up in the national lung screening trial</article-title>. <source>J Thorac Oncol</source>. (<year>2019</year>) <volume>14</volume>:<page-range>1732&#x2013;42</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jtho.2019.05.044</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vaid</surname> <given-names>AK</given-names>
</name>
<name>
<surname>Gupta</surname> <given-names>S</given-names>
</name>
<name>
<surname>Doval</surname> <given-names>DC</given-names>
</name>
<name>
<surname>Agarwal</surname> <given-names>S</given-names>
</name>
<name>
<surname>Nag</surname> <given-names>S</given-names>
</name>
<name>
<surname>Patil</surname> <given-names>P</given-names>
</name>
<etal/>
</person-group>. <article-title>Expert consensus on effective management of chemotherapy-induced nausea and vomiting: an Indian perspective</article-title>. <source>Front Oncol</source>. (<year>2020</year>) <volume>10</volume>:<elocation-id>400</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fonc.2020.00400</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lavdaniti</surname> <given-names>M</given-names>
</name>
<name>
<surname>Papastergiou</surname> <given-names>K</given-names>
</name>
</person-group>. <article-title>AB013. Nausea-vomiting in lung cancer patients undergoing chemotherapy</article-title>. <source>Ann Transl Med</source>. (<year>2016</year>) <volume>4</volume>:<elocation-id>AB013</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.21037/atm.2016.AB013</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Waddle</surname> <given-names>MR</given-names>
</name>
<name>
<surname>Ko</surname> <given-names>S</given-names>
</name>
<name>
<surname>Johnson</surname> <given-names>MM</given-names>
</name>
<name>
<surname>Lou</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Miller</surname> <given-names>RC</given-names>
</name>
<name>
<surname>Harrell</surname> <given-names>AC</given-names>
</name>
<etal/>
</person-group>. <article-title>Post-operative radiation therapy in locally advanced non-small cell lung cancer and the impact of sequential versus concurrent chemotherapy</article-title>. <source>Trans Lung Cancer Res</source>. (<year>2018</year>) <volume>7</volume>:<fpage>S171</fpage>&#x2013;<lpage>s175</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.21037/tlcr.2018.03.21</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Toi</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Sugawara</surname> <given-names>S</given-names>
</name>
<name>
<surname>Kobayashi</surname> <given-names>T</given-names>
</name>
<name>
<surname>Terayama</surname> <given-names>K</given-names>
</name>
<name>
<surname>Honda</surname> <given-names>Y</given-names>
</name>
</person-group>. <article-title>Observational study of chemotherapy-induced Clostridium difficile infection in patients with lung cancer</article-title>. <source>Int J Clin Oncol</source>. (<year>2018</year>) <volume>23</volume>:<page-range>1046&#x2013;51</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10147-018-1304-5</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fitzpatrick</surname> <given-names>ME</given-names>
</name>
<name>
<surname>Sethi</surname> <given-names>S</given-names>
</name>
<name>
<surname>Daley</surname> <given-names>CL</given-names>
</name>
<name>
<surname>Ray</surname> <given-names>P</given-names>
</name>
<name>
<surname>Beck</surname> <given-names>JM</given-names>
</name>
<name>
<surname>Gingo</surname> <given-names>MR</given-names>
</name>
</person-group>. <article-title>Infections in &#x201c;noninfectious&#x201d; lung diseases</article-title>. <source>Ann Am Thorac Society</source>. (<year>2014</year>) <volume>11 Suppl 4</volume>:<page-range>S221&#x2013;6</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1513/AnnalsATS.201401-041PL</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Forte</surname> <given-names>GC</given-names>
</name>
<name>
<surname>Altmayer</surname> <given-names>S</given-names>
</name>
<name>
<surname>Silva</surname> <given-names>RF</given-names>
</name>
<name>
<surname>Stefani</surname> <given-names>MT</given-names>
</name>
<name>
<surname>Libermann</surname> <given-names>LL</given-names>
</name>
<name>
<surname>Cavion</surname> <given-names>CC</given-names>
</name>
<etal/>
</person-group>. <article-title>Deep learning algorithms for diagnosis of lung cancer: A systematic review and meta-analysis</article-title>. <source>Cancers</source>. (<year>2022</year>) <volume>14</volume>(<issue>16</issue>):<fpage>3856</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/cancers14163856</pub-id>
</citation>
</ref>
<ref id="B9">
<label>9</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>J</given-names>
</name>
<name>
<surname>Han</surname> <given-names>C</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>X</given-names>
</name>
<name>
<surname>Chong</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Z</given-names>
</name>
<etal/>
</person-group>. <article-title>Preoperative prediction of lymph node metastasis in patients with early-T-stage non-small cell lung cancer by machine learning algorithms</article-title>. <source>Front Oncol</source>. (<year>2020</year>) <volume>10</volume>:<elocation-id>743</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fonc.2020.00743</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname> <given-names>CS</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>AY</given-names>
</name>
</person-group>. <article-title>Clinical applications of continual learning machine learning</article-title>. <source>Lancet Digital Health</source>. (<year>2020</year>) <volume>2</volume>:<page-range>e279&#x2013;81</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S2589-7500(20)30102-3</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shamout</surname> <given-names>F</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>T</given-names>
</name>
<name>
<surname>Clifton</surname> <given-names>DA</given-names>
</name>
</person-group>. <article-title>Machine learning for clinical outcome prediction</article-title>. <source>IEEE Rev Biomed Engineering</source>. (<year>2021</year>) <volume>14</volume>:<page-range>116&#x2013;26</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/RBME.4664312</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname> <given-names>F</given-names>
</name>
<name>
<surname>Azuma</surname> <given-names>K</given-names>
</name>
<name>
<surname>Nakahara</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Saito</surname> <given-names>H</given-names>
</name>
<name>
<surname>Matsuo</surname> <given-names>N</given-names>
</name>
<name>
<surname>Tagami</surname> <given-names>T</given-names>
</name>
<etal/>
</person-group>. <article-title>Machine learning for prediction of immunotherapeutic outcome in non-small-cell lung cancer based on circulating cytokine signatures</article-title>. <source>J Immunother Cancer</source>. (<year>2023</year>) <volume>11</volume>(<issue>7</issue>):<elocation-id>e006788</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1136/jitc-2023-006788</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Frost</surname> <given-names>HR</given-names>
</name>
<name>
<surname>Amos</surname> <given-names>CI</given-names>
</name>
</person-group>. <article-title>Gene set selection via LASSO penalized regression (SLPR)</article-title>. <source>Nucleic Acids Res</source>. (<year>2017</year>) <volume>45</volume>:<fpage>e114</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkx291</pub-id>
</citation>
</ref>
<ref id="B14">
<label>14</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname> <given-names>S</given-names>
</name>
<name>
<surname>Gornitz</surname> <given-names>N</given-names>
</name>
<name>
<surname>Xing</surname> <given-names>EP</given-names>
</name>
<name>
<surname>Heckerman</surname> <given-names>D</given-names>
</name>
<name>
<surname>Lippert</surname> <given-names>C</given-names>
</name>
</person-group>. <article-title>Ensembles of lasso screening rules</article-title>. <source>IEEE Trans Pattern Anal Mach Intelligence</source>. (<year>2018</year>) <volume>40</volume>:<page-range>2841&#x2013;52</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TPAMI.34</pub-id>
</citation>
</ref>
<ref id="B15">
<label>15</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>J</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>H</given-names>
</name>
<name>
<surname>Song</surname> <given-names>W</given-names>
</name>
<name>
<surname>Qiao</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>Y</given-names>
</name>
<etal/>
</person-group>. <article-title>Diabetes mellitus early warning and factor analysis using ensemble Bayesian networks with SMOTE-ENN and Boruta</article-title>. <source>Sci Rep</source>. (<year>2023</year>) <volume>13</volume>:<fpage>12718</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-023-40036-5</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Saleem</surname> <given-names>J</given-names>
</name>
<name>
<surname>Zakar</surname> <given-names>R</given-names>
</name>
<name>
<surname>Butt</surname> <given-names>MS</given-names>
</name>
<name>
<surname>Aadil</surname> <given-names>RM</given-names>
</name>
<name>
<surname>Ali</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Bukhari</surname> <given-names>GMJ</given-names>
</name>
<etal/>
</person-group>. <article-title>Application of the Boruta algorithm to assess the multidimensional determinants of malnutrition among children under five years living in southern Punjab, Pakistan</article-title>. <source>BMC Public Health</source>. (<year>2024</year>) <volume>24</volume>:<fpage>167</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12889-024-17701-z</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pan</surname> <given-names>X</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>Y</given-names>
</name>
</person-group>. <article-title>A safe feature elimination rule for L(1)-regularized logistic regression</article-title>. <source>IEEE Trans Pattern Anal Mach Intelligence</source>. (<year>2022</year>) <volume>44</volume>:<page-range>4544&#x2013;54</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/tpami.2021.3071138</pub-id>
</citation>
</ref>
<ref id="B18">
<label>18</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Motamedi</surname> <given-names>F</given-names>
</name>
<name>
<surname>P&#xe9;rez-S&#xe1;nchez</surname> <given-names>H</given-names>
</name>
<name>
<surname>Mehridehnavi</surname> <given-names>A</given-names>
</name>
<name>
<surname>Fassihi</surname> <given-names>A</given-names>
</name>
<name>
<surname>Ghasemi</surname> <given-names>F</given-names>
</name>
</person-group>. <article-title>Accelerating big data analysis through LASSO-random forest algorithm in QSAR studies</article-title>. <source>Bioinf (Oxford England)</source>. (<year>2022</year>) <volume>38</volume>:<page-range>469&#x2013;75</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btab659</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>D</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>C</given-names>
</name>
</person-group>. <article-title>NeuroCNN_GNB: an ensemble model to predict neuropeptides based on a convolution neural network and Gaussian naive Bayes</article-title>. <source>Front Genet</source>. (<year>2023</year>) <volume>14</volume>:<elocation-id>1226905</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fgene.2023.1226905</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hong</surname> <given-names>H</given-names>
</name>
<name>
<surname>Tsangaratos</surname> <given-names>P</given-names>
</name>
<name>
<surname>Ilia</surname> <given-names>I</given-names>
</name>
<name>
<surname>Loupasakis</surname> <given-names>C</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y</given-names>
</name>
</person-group>. <article-title>Introducing a novel multi-layer perceptron network based on stochastic gradient descent optimized by a meta-heuristic algorithm for landslide susceptibility mapping</article-title>. <source>Sci Total Environment</source>. (<year>2020</year>) <volume>742</volume>:<elocation-id>140549</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.scitotenv.2020.140549</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rezvani</surname> <given-names>S</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Handling multi-class problem by intuitionistic fuzzy twin support vector machines based on relative density information</article-title>. <source>IEEE Trans Pattern Anal Mach Intelligence</source>. (<year>2023</year>) <volume>45</volume>:<page-range>14653&#x2013;64</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TPAMI.2023.3310908</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Davenport</surname> <given-names>MA</given-names>
</name>
<name>
<surname>Baraniuk</surname> <given-names>RG</given-names>
</name>
<name>
<surname>Scott</surname> <given-names>CD</given-names>
</name>
</person-group>. <article-title>Tuning support vector machines for minimax and Neyman-Pearson classification</article-title>. <source>IEEE Trans Pattern Anal Mach Intelligence</source>. (<year>2010</year>) <volume>32</volume>:<page-range>1888&#x2013;98</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TPAMI.2010.29</pub-id>
</citation>
</ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goin</surname> <given-names>JE</given-names>
</name>
</person-group>. <article-title>Classification bias of the k-nearest neighbor algorithm</article-title>. <source>IEEE Trans Pattern Anal Mach Intelligence</source>. (<year>1984</year>) <volume>6</volume>:<page-range>379&#x2013;81</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TPAMI.1984.4767533</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname> <given-names>C</given-names>
</name>
<name>
<surname>Xiu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Qiao</surname> <given-names>K</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>X</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>S</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>Y</given-names>
</name>
</person-group>. <article-title>Prediction of lymph node metastasis in patients with breast invasive micropapillary carcinoma based on machine learning and SHapley Additive exPlanations framework</article-title>. <source>Front Oncol</source>. (<year>2022</year>) <volume>12</volume>:<elocation-id>981059</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fonc.2022.981059</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bifarin</surname> <given-names>OO</given-names>
</name>
</person-group>. <article-title>Interpretable machine learning with tree-based shapley additive explanations: Application to metabolomics datasets for binary classification</article-title>. <source>PloS One</source>. (<year>2023</year>) <volume>18</volume>:<fpage>e0284315</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0284315</pub-id>
</citation>
</ref>
<ref id="B26">
<label>26</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ding</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>S</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Risk factors for postoperative pulmonary infection in patients with non-small cell lung cancer: analysis based on regression models and construction of a nomogram prediction model</article-title>. <source>Am J Trans Res</source>. (<year>2023</year>) <volume>15</volume>:<page-range>3375&#x2013;84</page-range>.</citation>
</ref>
<ref id="B27">
<label>27</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname> <given-names>JH</given-names>
</name>
<name>
<surname>Cheon</surname> <given-names>BR</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>MG</given-names>
</name>
<name>
<surname>Hwang</surname> <given-names>SM</given-names>
</name>
<name>
<surname>Lim</surname> <given-names>SY</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>JJ</given-names>
</name>
<etal/>
</person-group>. <article-title>Harnessing machine learning for prediction of postoperative pulmonary complications: retrospective cohort design</article-title>. <source>J Clin Med</source>. (<year>2023</year>) <volume>12</volume>(<issue>17</issue>):<fpage>5681</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/jcm12175681</pub-id>
</citation>
</ref>
<ref id="B28">
<label>28</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xue</surname> <given-names>B</given-names>
</name>
<name>
<surname>Li</surname> <given-names>D</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>C</given-names>
</name>
<name>
<surname>King</surname> <given-names>CR</given-names>
</name>
<name>
<surname>Wildes</surname> <given-names>T</given-names>
</name>
<name>
<surname>Avidan</surname> <given-names>MS</given-names>
</name>
<etal/>
</person-group>. <article-title>Use of machine learning to develop and evaluate models using preoperative and intraoperative data to identify risks of postoperative complications</article-title>. <source>JAMA Network Open</source>. (<year>2021</year>) <volume>4</volume>:<fpage>e212240</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1001/jamanetworkopen.2021.2240</pub-id>
</citation>
</ref>
<ref id="B29">
<label>29</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Morelli</surname> <given-names>T</given-names>
</name>
<name>
<surname>Fujita</surname> <given-names>K</given-names>
</name>
<name>
<surname>Redelman-Sidi</surname> <given-names>G</given-names>
</name>
<name>
<surname>Elkington</surname> <given-names>PT</given-names>
</name>
</person-group>. <article-title>Infections due to dysregulated immunity: an emerging complication of cancer immunotherapy</article-title>. <source>Thorax</source>. (<year>2022</year>) <volume>77</volume>:<page-range>304&#x2013;11</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1136/thoraxjnl-2021-217260</pub-id>
</citation>
</ref>
<ref id="B30">
<label>30</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>T</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Si</surname> <given-names>X</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>Opportunistic infections complicating immunotherapy for non-small cell lung cancer</article-title>. <source>Thorac Cancer</source>. (<year>2020</year>) <volume>11</volume>:<page-range>1689&#x2013;94</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/1759-7714.13422</pub-id>
</citation>
</ref>
<ref id="B31">
<label>31</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vento</surname> <given-names>S</given-names>
</name>
<name>
<surname>Cainelli</surname> <given-names>F</given-names>
</name>
<name>
<surname>Temesgen</surname> <given-names>Z</given-names>
</name>
</person-group>. <article-title>Lung infections after cancer chemotherapy</article-title>. <source>Lancet Oncol</source>. (<year>2008</year>) <volume>9</volume>:<page-range>982&#x2013;92</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S1470-2045(08)70255-9</pub-id>
</citation>
</ref>
<ref id="B32">
<label>32</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karam</surname> <given-names>JD</given-names>
</name>
<name>
<surname>Noel</surname> <given-names>N</given-names>
</name>
<name>
<surname>Voisin</surname> <given-names>AL</given-names>
</name>
<name>
<surname>Lanoy</surname> <given-names>E</given-names>
</name>
<name>
<surname>Michot</surname> <given-names>JM</given-names>
</name>
<name>
<surname>Lambotte</surname> <given-names>O</given-names>
</name>
</person-group>. <article-title>Infectious complications in patients treated with immune checkpoint inhibitors</article-title>. <source>Eur J Cancer (Oxford England: 1990)</source>. (<year>2020</year>) <volume>141</volume>:<page-range>137&#x2013;42</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ejca.2020.09.025</pub-id>
</citation>
</ref>
<ref id="B33">
<label>33</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Luo</surname> <given-names>YH</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>CI</given-names>
</name>
<name>
<surname>Chiang</surname> <given-names>CL</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>HC</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>YM</given-names>
</name>
</person-group>. <article-title>Dynamic immune signatures of patients with advanced non-small-cell lung cancer for infection prediction after immunotherapy</article-title>. <source>Front Immunol</source>. (<year>2024</year>) <volume>15</volume>:<elocation-id>1269253</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fimmu.2024.1269253</pub-id>
</citation>
</ref>
<ref id="B34">
<label>34</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname> <given-names>W</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>G</given-names>
</name>
<name>
<surname>Dai</surname> <given-names>J</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>Q</given-names>
</name>
</person-group>. <article-title>Prediction of lung infection during palliative chemotherapy of lung cancer based on artificial neural network</article-title>. <source>Comput Math Methods Med</source>. (<year>2022</year>) <volume>2022</volume>:<elocation-id>4312117</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1155/2022/4312117</pub-id>
</citation>
</ref>
<ref id="B35">
<label>35</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gong</surname> <given-names>K</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>HK</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>K</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>X</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>A prediction and interpretation framework of acute kidney injury in critical care</article-title>. <source>J Biomed Informatics</source>. (<year>2021</year>) <volume>113</volume>:<elocation-id>103653</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jbi.2020.103653</pub-id>
</citation>
</ref>
<ref id="B36">
<label>36</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nohara</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Matsumoto</surname> <given-names>K</given-names>
</name>
<name>
<surname>Soejima</surname> <given-names>H</given-names>
</name>
<name>
<surname>Nakashima</surname> <given-names>N</given-names>
</name>
</person-group>. <article-title>Explanation of machine learning models using shapley additive explanation and application for real data in hospital</article-title>. <source>Comput Methods Programs Biomed</source>. (<year>2022</year>) <volume>214</volume>:<elocation-id>106584</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cmpb.2021.106584</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>