<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Endocrinol.</journal-id>
<journal-title>Frontiers in Endocrinology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Endocrinol.</abbrev-journal-title>
<issn pub-type="epub">1664-2392</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fendo.2024.1390352</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Endocrinology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A stacking ensemble model for predicting the occurrence of carotid atherosclerosis</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Xiaoshuai</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Tang</surname>
<given-names>Chuanping</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Shuohuan</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Wei</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yang</surname>
<given-names>Wangxuan</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2739939"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Di</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Qinghuan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Tang</surname>
<given-names>Fang</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">*</xref>
<uri xlink:href="https://loop.frontiersin.org/people/799851"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Data Science, School of Statistics and Mathematics, Shandong University of Finance and Economics</institution>, <addr-line>Jinan</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Information Technology Division, Shandong International Trust Co., Ltd.</institution>, <addr-line>Jinan</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Medical Ultrasound, The First Affiliated Hospital of Shandong First Medical University &amp; Shandong Provincial Qianfoshan Hospital, Shandong Engineering Research Center of Diagnosis and Treatment Technology for Bariatric and Metabolism-Associated Surgery</institution>, <addr-line>Jinan</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>School of Public Health, Harbin Medical University</institution>, <addr-line>Harbin</addr-line>, <country>China</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Center for Big Data Research in Health and Medicine, The First Affiliated Hospital of Shandong First Medical University &amp; Shandong Provincial Qianfoshan Hospital Shandong Data Open Innovative Application Laboratory</institution>, <addr-line>Jinan</addr-line>, <country>China</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Shandong Provincial Qianfoshan Hospital, Cheeloo College of Medicine, Shandong University</institution>, <addr-line>Jinan</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Eliza Russu, George Emil Palade University of Medicine, Pharmacy, Sciences and Technology of T&#xe2;rgu Mure&#x15f;, Romania</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Niranjana Sampathila, Manipal Academy of Higher Education, India</p>
<p>Stoian Adina, George Emil Palade University of Medicine, Pharmacy, Sciences and Technology of T&#xe2;rgu Mure&#x15f;, Romania</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Fang Tang, <email xlink:href="mailto:tangfangsdu@gmail.com">tangfangsdu@gmail.com</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>23</day>
<month>07</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1390352</elocation-id>
<history>
<date date-type="received">
<day>23</day>
<month>02</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>09</day>
<month>07</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Zhang, Tang, Wang, Liu, Yang, Wang, Wang and Tang</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Zhang, Tang, Wang, Liu, Yang, Wang, Wang and Tang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Background</title>
<p>Carotid atherosclerosis (CAS) is a significant risk factor for cardio-cerebrovascular events. The objective of this study is to employ stacking ensemble machine learning techniques to enhance the prediction of CAS occurrence, incorporating a wide range of predictors, including endocrine-related markers.</p>
</sec>
<sec>
<title>Methods</title>
<p>Based on data from a routine health check-up cohort, five individual prediction models for CAS were established based on logistic regression (LR), random forest (RF), support vector machine (SVM), extreme gradient boosting (XGBoost) and gradient boosting decision tree (GBDT) methods. Then, a stacking ensemble algorithm was used to integrate the base models to improve the prediction ability and address overfitting problems. Finally, the SHAP value method was applied for an in-depth analysis of variable importance at both the overall and individual levels, with a focus on elucidating the impact of endocrine-related variables.</p>
</sec>
<sec>
<title>Results</title>
<p>A total of 441 of the 1669 subjects in the cohort were finally diagnosed with CAS. Seventeen variables were selected as predictors. The ensemble model outperformed the individual models, with AUCs of 0.893 in the testing set and 0.861 in the validation set. The ensemble model has the optimal accuracy, precision, recall and F1 score in the validation set, with considerable performance in the testing set. Carotid stenosis and age emerged as the most significant predictors, alongside notable contributions from endocrine-related factors.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>The ensemble model shows enhanced accuracy and generalizability in predicting CAS risk, underscoring its utility in identifying individuals at high risk. This approach integrates a comprehensive analysis of predictors, including endocrine markers, affirming the critical role of endocrine dysfunctions in CAS development. It represents a promising tool in identifying high-risk individuals for the prevention of CAS and cardio-cerebrovascular diseases.</p>
</sec>
</abstract>
<kwd-group>
<kwd>carotid atherosclerosis</kwd>
<kwd>endocrine-related markers</kwd>
<kwd>prediction</kwd>
<kwd>stacking</kwd>
<kwd>machine learning</kwd>
</kwd-group>
<counts>
<fig-count count="3"/>
<table-count count="2"/>
<equation-count count="0"/>
<ref-count count="36"/>
<page-count count="8"/>
<word-count count="3710"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Cardiovascular Endocrinology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Carotid atherosclerosis (CAS) is a multifaceted disease characterized by the progressive accumulation of atherosclerotic plaques within the carotid arteries (<xref ref-type="bibr" rid="B1">1</xref>). As a manifestation of atherosclerosis in local blood vessels, the continuous development of CAS is a major and potentially preventable cause of ischaemic stroke (<xref ref-type="bibr" rid="B2">2</xref>). Early manifestations of CAS such as intermittent dizziness or mild headaches are subtle, often leading to missed diagnoses. As CAS progresses, it severely impacts the physical and psychological well-being of individuals, imposing substantial financial strains on their families. Therefore, early prediction and prevention of CAS are crucial to mitigate the risk of subsequent cardio-cerebrovascular events.</p>
<p>Current research on CAS has mainly focused on the analysis of risk factors, the most common of which include age, smoking status, physical inactivity, abnormal blood glucose levels, hypertension and others (<xref ref-type="bibr" rid="B3">3</xref>&#x2013;<xref ref-type="bibr" rid="B7">7</xref>). These factors, particularly hyperglycemia and hypertension, indicative of underlying metabolic and hormonal imbalances, contribute to the endothelial dysfunction, inflammation, and subsequent plaque formation characteristic of atherosclerosis. Despite numerous studies on CAS risk factors, there is a scarcity of research dedicated to developing predictive models for CAS, with existing models primarily using cross-sectional data for disease diagnosis rather than predicting its onset.</p>
<p>Machine learning methods offer the potential to achieve precise predictive ability to assess diagnostic and prognostic outcomes (<xref ref-type="bibr" rid="B8">8</xref>&#x2013;<xref ref-type="bibr" rid="B11">11</xref>). Among various machine learning approaches, ensemble learning, which includes techniques like bagging (e.g., random forests), boosting (e.g., XGBoost, GBDT), and stacking, stands out by integrating multiple weak classifiers to form a robust classifier, thereby improving prediction accuracy and model generalizability (<xref ref-type="bibr" rid="B12">12</xref>&#x2013;<xref ref-type="bibr" rid="B15">15</xref>). Stacking ensemble models, which train different weak learners in parallel, have shown superior performance across various domains, from healthcare to financial forecasting (<xref ref-type="bibr" rid="B16">16</xref>&#x2013;<xref ref-type="bibr" rid="B20">20</xref>). However, they also present significant challenges such as computational complexity and a lack of interpretability, often referred to as the &#x201c;black box&#x201d; phenomenon, which can obscure understanding of decision-making processes (<xref ref-type="bibr" rid="B21">21</xref>).</p>
<p>In this study, we employ a stacking ensemble learning algorithm to construct a risk prediction model for the occurrence of CAS based on a routine health checkup cohort. The predictive performance of the ensemble model was compared with that of the individual models. We utilize the SHapley Additive exPlanation (SHAP) method (<xref ref-type="bibr" rid="B22">22</xref>) to elucidate the predictive relationships between CAS and various risk factors, with a particular focus on endocrine-related markers.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>Study design and data collection</title>
<p>The study cohort was derived from the routine health check-up system of the First Affiliated Hospital of Shandong First Medical University in Jinan, China. All the participants were free of CAS at the first check-up and underwent three health checks during the follow-up. Individuals who had been diagnosed with coronary heart disease, previous coronary heart disease, cerebral ischaemia, cerebral infarction, cerebral artery stenosis, cerebral artery spasm, coronary artery stenosis, coronary atherosclerotic heart disease, and those with missing information were excluded. CAS was diagnosed by carotid B-mode ultrasonography as a carotid intima-media thickness of 1.0&#xa0;mm or greater or plaque formation. The study was approved by the Ethics Committee of the First Affiliated Hospital of Shandong First Medical University, and informed consent was obtained from all eligible participants.</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Study variables</title>
<p>The study variables consisted of three sets of data: demographic data, laboratory indicators, and clinical history. All the individuals in this study cohort underwent anthropometric and laboratory tests. The height and weight of the participants were measured while they were wearing light clothing and no shoes. Peripheral blood samples were collected from the subjects after an overnight fast, and the variables included blood urea nitrogen (BUN), lymphocyte percentage (LYM), aspartate aminotransferase (AST), red cell volume distribution width standard deviation (RDW-SD), red blood cell count (RBC), mean corpuscular haemoglobin concentration (MCHC), mean platelets (MPV), fasting blood glucose (GLU), platelet count (PLT), eosinophil percent (PEOS), white blood cell count (WBC), and carcinoembryonic antigen (CEA). Disease history was also collected, such as carotid stenosis (CS), diabetes mellitus (DM), and hypertension. All the measurements were collected following the same standard procedures.</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>SMOTE sampling</title>
<p>To address the issue of data imbalance, we applied the synthetic minority oversampling technique (SMOTE) in our study. Since the number of individuals without CAS was larger than those with CAS, SMOTE was employed to generate synthetic samples of the minority class (<xref ref-type="bibr" rid="B22">22</xref>). The new synthetic records were generated using the existing samples of the minority class via linear interpolation. After we obtain new minority sample data, a balanced dataset can be obtained by merging with majority samples.</p>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Variable selection</title>
<p>Variable selection is an important step in the application of machine learning to ensure that the most relevant predictors are used. Both filtering and embedded feature selection methods were used to select the predictors. The variables were first selected using univariate logical regression with a threshold <italic>P</italic> value of 0.1. Second, we applied three tree-based machine learning methods-random forest (RF), eXtreme Gradient Boost (XGBoost), and gradient boosting decision tree (GBDT))-to assess the importance of each variable. These methods are well-suited for identifying important variables due to their ability to capture complex interactions and non-linear relationships. Variables were ranked based on their importance scores from these models. Finally, the correlation coefficients of the continuous variables were calculated to address multicollinearity, which can distort the model&#x2019;s estimates and reduce interpretability. Features with low importance among the highly related variables were eliminated for determining the predictors.</p>
</sec>
<sec id="s2_5">
<label>2.5</label>
<title>Statistical analysis</title>
<p>The baseline characteristics were assessed for CAS and non-CAS patients during the follow-up. Continuous variables were described by the mean and standard deviation (SD), and categorical features were described as proportions; we compared the baseline features using the <italic>t</italic> test and the chi-square test. To predict the probability of CAS, we employed five machine learning models: logistic regression (LR), support vector machine (SVM), RF, XGBoost, and GBDT (<xref ref-type="bibr" rid="B11">11</xref>, <xref ref-type="bibr" rid="B23">23</xref>&#x2013;<xref ref-type="bibr" rid="B26">26</xref>). We then used a stacking ensemble model, specifically the super learner, which combines these individual models by assigning them different weights to optimize predictive performance (<xref ref-type="bibr" rid="B27">27</xref>). The final predicted value is a weighted sum of the individual model predictions, where the weights are determined to minimize the cross-validation risk (<xref ref-type="bibr" rid="B28">28</xref>). <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref> shows the framework of the super learner ensemble model. We compared the predictive performance of the super learner against the individual models (LR, RF, SVM, XGBoost, and GBDT).</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Framework of the stacking ensemble model. (LR, logistic regression; RF, random forest; SVM, support vector machine; XGBoost, extreme gradient boosting; GBDT, gradient boosting decision tree.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fendo-15-1390352-g001.tif"/>
</fig>
</sec>
<sec id="s2_6">
<label>2.6</label>
<title>Hyperparameter tuning</title>
<p>We optimized the hyperparameters using a random search method, which is a traditional and efficient technique for tuning in classification methods (<xref ref-type="bibr" rid="B29">29</xref>).This process was conducted within a 5-fold cross-validation framework. Specifically, each model configuration was trained on four folds and validated on the remaining fold. This cycle was repeated five times, with each fold serving as the validation set once, to ensure a comprehensive evaluation across the entire dataset. We performed the random search 1,000 times, selecting the hyperparameter combination with the highest average areas under the receiver operating characteristic curve (AUC).</p>
</sec>
<sec id="s2_7">
<label>2.7</label>
<title>Predictive performance assessment</title>
<p>The performance of the prediction model was validated using both the testing and hold-out validation set. Several metrics were used to evaluate the performance of the prediction models: accuracy, precision, recall, F1 score, and AUC. Compared to commonly used performance metrics, the AUC better reflects model performance in unbalanced datasets. Hence, the AUC was the main metric, while the others were considered of secondary priority.</p>
</sec>
<sec id="s2_8">
<label>2.8</label>
<title>Model interpretation</title>
<p>To solve the &#x201c;black box&#x201d; problem in machine learning, we report the feature importance ranking of each predictor based on SHAP values (<xref ref-type="bibr" rid="B22">22</xref>). SHAP values are useful for explaining the prediction of a machine learning model by computing the contribution of each feature to the prediction. Kernel-based SHAP values were used to rank the variables in terms of their ability to predict the CAS, which is an additive feature attribution method using kernel functions enabling consistent explanation of feature importance (<xref ref-type="bibr" rid="B30">30</xref>).</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results</title>
<sec id="s3_1">
<label>3.1</label>
<title>Data description</title>
<p>A total of 1669 participants were included in this study, including 1426 (85.4%) males and 243 (14.6%) females. A total of 441 participants were diagnosed with CAS during the follow-up, including 395 men and 46 women. A total of 1228 participants were not diagnosed with CAS. The SMOTE method was used to address the sample imbalance problem. <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref> shows the roadmap of the data processing. All the samples were first divided into a training set and a hold-out validation set. Specifically, 5% of the subjects were randomly selected in advance as the validation set, and the remaining 95% were used for model construction; 419 patients with CAS and 1167 non-CAS patients were included. Of the remaining data, 70% of the data were used as the training set where SMOTE resampling was applied to address class imbalance. The remaining 30% was used as the testing set. In the original dataset, the ratio of non-CAS to CAS was approximately 2.78 to 1, which was adjusted in the training set to approximately 1 to 1.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Dataset partitioning in the modelling process.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fendo-15-1390352-g002.tif"/>
</fig>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Variable selection results</title>
<p>A total of 28 variables with a <italic>P</italic> value &lt; 0.1 were retained in the univariate logistic regression and subsequently included in the three machine learning models. <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S1</bold>
</xref> shows the variable importance rankings for RF, GBDT, and XGBoost. A total of 22 of these variables were common among all three machine learning algorithms (please see the <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary File</bold>
</xref>). The correlations between the selected continuous variables are shown in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S2</bold>
</xref> in the <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary File</bold>
</xref>. Variables with high correlations and the lowest importance were removed; thus, five variables were excluded. Finally, 17 variables were selected, including age, sex, BUN, LYM, AST, RDW-SD, RBC, MCHC, MPV, GLU, PLT, PEOS, WBC, CEA, CS, hypertension and DM.</p>
<p>
<xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> summarizes the baseline characteristics of the incident CAS status. Overall, individuals who developed CAS were more likely to be male CS, DM, hypertension, older age, MCHC, BUN, RDW-SD, MPV, GLU, PEOS, WBC, and CEA and lower PLT, LYM, AST, and RBC at baseline; these variables were significantly different at a level of 0.1.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Baseline characteristics by incident CAS status.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Variables</th>
<th valign="middle" align="center">N=1669</th>
<th valign="middle" align="center">Non-CAS<break/>N=1228 (73.6%)</th>
<th valign="middle" align="center">CAS<break/>N=441 (26.4%)</th>
<th valign="middle" align="center"><italic>t/X</italic>
<sup>2</sup>
</th>
<th valign="middle" align="center">
<italic>P</italic>
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center" style="">Age</td>
<td valign="middle" align="center" style="">55.1 &#xb1; 7.6</td>
<td valign="middle" align="center" style="">53.7 &#xb1; 6.2</td>
<td valign="middle" align="center" style="">59.1 &#xb1; 9.4</td>
<td valign="middle" align="center" style="">-11.28</td>
<td valign="middle" align="center" style="">&lt;0.001</td>
</tr>
<tr>
<td valign="middle" align="center" style="">MCHC</td>
<td valign="middle" align="center" style="">339.4 &#xb1; 10.1</td>
<td valign="middle" align="center" style="">339.0 &#xb1; 10.4</td>
<td valign="middle" align="center" style="">340.3 &#xb1; 9.3</td>
<td valign="middle" align="center" style="">-2.29</td>
<td valign="middle" align="center" style="">0.022</td>
</tr>
<tr>
<td valign="middle" align="center" style="">PLT</td>
<td valign="middle" align="center" style="">221.1 &#xb1; 50.9</td>
<td valign="middle" align="center" style="">222.6 &#xb1; 50.2</td>
<td valign="middle" align="center" style="">216.9 &#xb1; 52.7</td>
<td valign="middle" align="center" style="">1.98</td>
<td valign="middle" align="center" style="">0.048</td>
</tr>
<tr>
<td valign="middle" align="center" style="">BUN</td>
<td valign="middle" align="center" style="">5.2 &#xb1; 1.3</td>
<td valign="middle" align="center" style="">5.2 &#xb1; 1.3</td>
<td valign="middle" align="center" style="">5.4 &#xb1; 1.3</td>
<td valign="middle" align="center" style="">-2.39</td>
<td valign="middle" align="center" style="">0.017</td>
</tr>
<tr>
<td valign="middle" align="center" style="">LYM</td>
<td valign="middle" align="center" style="">0.4 &#xb1; 0.1</td>
<td valign="middle" align="center" style="">0.4 &#xb1; 0.1</td>
<td valign="middle" align="center" style="">0.3 &#xb1; 0.1</td>
<td valign="middle" align="center" style="">3.29</td>
<td valign="middle" align="center" style="">0.001</td>
</tr>
<tr>
<td valign="middle" align="center" style="">AST</td>
<td valign="middle" align="center" style="">20.0 &#xb1; 8.0</td>
<td valign="middle" align="center" style="">20.4 &#xb1; 8.5</td>
<td valign="middle" align="center" style="">19.0 &#xb1; 6.0</td>
<td valign="middle" align="center" style="">3.72</td>
<td valign="middle" align="center" style="">&lt;0.001</td>
</tr>
<tr>
<td valign="middle" align="center" style="">RDW-SD</td>
<td valign="middle" align="center" style="">42.4 &#xb1; 2.7</td>
<td valign="middle" align="center" style="">42.2 &#xb1; 2.7</td>
<td valign="middle" align="center" style="">42.8 &#xb1; 2.6</td>
<td valign="middle" align="center" style="">-3.63</td>
<td valign="middle" align="center" style="">&lt;0.001</td>
</tr>
<tr>
<td valign="middle" align="center" style="">RBC</td>
<td valign="middle" align="center" style="">4.9 &#xb1; 0.4</td>
<td valign="middle" align="center" style="">4.9 &#xb1; 0.4</td>
<td valign="middle" align="center" style="">4.8 &#xb1; 0.4</td>
<td valign="middle" align="center" style="">2.46</td>
<td valign="middle" align="center" style="">0.014</td>
</tr>
<tr>
<td valign="middle" align="center" style="">MPV</td>
<td valign="middle" align="center" style="">10.3 &#xb1; 0.9</td>
<td valign="middle" align="center" style="">10.3 &#xb1; 0.9</td>
<td valign="middle" align="center" style="">10.4 &#xb1; 0.8</td>
<td valign="middle" align="center" style="">-2.52</td>
<td valign="middle" align="center" style="">0.012</td>
</tr>
<tr>
<td valign="middle" align="center" style="">GLU</td>
<td valign="middle" align="center" style="">5.7 &#xb1; 1.3</td>
<td valign="middle" align="center" style="">5.6 &#xb1; 1.1</td>
<td valign="middle" align="center" style="">5.9 &#xb1; 1.7</td>
<td valign="middle" align="center" style="">-4.43</td>
<td valign="middle" align="center" style="">&lt;0.001</td>
</tr>
<tr>
<td valign="middle" align="center" style="">EOS</td>
<td valign="middle" align="center" style="">0.026 &#xb1; 0.022</td>
<td valign="middle" align="center" style="">0.025 &#xb1; 0.020</td>
<td valign="middle" align="center" style="">0.028 &#xb1; 0.026</td>
<td valign="middle" align="center" style="">-1.87</td>
<td valign="middle" align="center" style="">0.061</td>
</tr>
<tr>
<td valign="middle" align="center" style="">WBC</td>
<td valign="middle" align="center" style="">6.3 &#xb1; 1.6</td>
<td valign="middle" align="center" style="">6.2 &#xb1; 1.5</td>
<td valign="middle" align="center" style="">6.4 &#xb1; 1.7</td>
<td valign="middle" align="center" style="">-2.29</td>
<td valign="middle" align="center" style="">0.023</td>
</tr>
<tr>
<td valign="middle" align="center" style="">CEA</td>
<td valign="middle" align="center" style="">2.0 &#xb1; 1.4</td>
<td valign="middle" align="center" style="">1.9 &#xb1; 1.3</td>
<td valign="middle" align="center" style="">2.1 &#xb1; 1.6</td>
<td valign="middle" align="center" style="">-2.93</td>
<td valign="middle" align="center" style="">0.004</td>
</tr>
<tr>
<td valign="middle" align="center" style="">Sex: n (%)<break/>&#x2003;0<break/>&#x2003;1</td>
<td valign="bottom" align="center" style="">1426(85.4%)<break/>243(14.6%)</td>
<td valign="bottom" align="center" style="">1031(84.0%)<break/>197(16.0%)</td>
<td valign="bottom" align="center" style="">395(89.6%)<break/>46 (10.4%)</td>
<td valign="middle" align="center" style="">8.21</td>
<td valign="middle" align="center" style="">0.004</td>
</tr>
<tr>
<td valign="middle" align="center" style="">CS: n (%)<break/>&#x2003;0<break/>&#x2003;1</td>
<td valign="bottom" align="center" style="">1088(65.2%)<break/>581 (34.8%)</td>
<td valign="bottom" align="center" style="">895(72.9%)<break/>333(27.1%)</td>
<td valign="bottom" align="center" style="">193(43.8%)<break/>248(56.2%)</td>
<td valign="middle" align="center" style="">121.24</td>
<td valign="middle" align="center" style="">&lt;0.001</td>
</tr>
<tr>
<td valign="middle" align="center" style="">DM: n (%)<break/>&#x2003;0<break/>&#x2003;1</td>
<td valign="bottom" align="center" style="">1475(88.4%)<break/>194(11.6%)</td>
<td valign="bottom" align="center" style="">1115(90.8%)<break/>113(9.2%)</td>
<td valign="bottom" align="center" style="">360(81.6%)<break/>81(18.4%)</td>
<td valign="middle" align="center" style="">26.53</td>
<td valign="middle" align="center" style="">&lt;0.001</td>
</tr>
<tr>
<td valign="middle" align="center" style="">Hypertension: n (%)<break/>&#x2003;0<break/>&#x2003;1</td>
<td valign="bottom" align="center" style="">1150(68.9%)<break/>519(31.1%)</td>
<td valign="bottom" align="center" style="">895(72.9%)<break/>333(27.1%)</td>
<td valign="bottom" align="center" style="">255(57.8%)<break/>186(42.2%)</td>
<td valign="middle" align="center" style="">34.34</td>
<td valign="middle" align="center" style="">&lt;0.001</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Model comparison</title>
<p>The super learner algorithm creates an optimal weighted average of the five models (LR, RF, SVM, XGBoost, and GBDT). <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S1</bold>
</xref> in the <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary File</bold>
</xref> depicts the weight coefficients of the super learner model. <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S2</bold>
</xref> shows the hyperparameters used in the models. The weight coefficients of LR and SVM are 0, indicating that they were not used in the super learner model, while the coefficient of RF is 0.820, which is much greater than that of the other four models, indicating that RF contributes most in the prediction model.</p>
<p>
<xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> shows the predictive performance of the six models on the testing set and the hold-out validation set. It can be seen that the predictive performance varies across the five models. SVM has the highest precision, while its performance in the validation set is inferior. Logistic regression had the lowest performance in the testing set. These results also indicate the reason that the two methods are not selected in the super learner model. Combined with the advantages of RF, XGBoost and GBDT, the performance of the super learner was improved. The super learner model had the optimal performance measures in the validation set, with considerable performance in the testing set. The ROC curves of the different machine learning models are shown in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S3</bold>
</xref>. The super learner has the largest ROC curve area in the validation set, and its AUC is 0.861. Overall, the predictive performance of the super learner model is superior to that of the other five models, especially regarding the overfitting problem.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Predictive performance of the six machine learning models.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="left">Models</th>
<th valign="middle" colspan="5" align="center">Performance metrics<xref ref-type="table-fn" rid="fnT2_1">
<sup>a</sup>
</xref>
</th>
</tr>
<tr>
<th valign="middle" align="left">accuracy</th>
<th valign="middle" align="left">precision</th>
<th valign="middle" align="left">recall</th>
<th valign="middle" align="left">F1</th>
<th valign="middle" align="left">AUC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">LR</td>
<td valign="middle" align="left">0.734(0.783)</td>
<td valign="middle" align="left">0.732(0.563)</td>
<td valign="middle" align="left">0.701(0.818)</td>
<td valign="middle" align="left">0.716(0.667)</td>
<td valign="middle" align="left">0.797(0.857)</td>
</tr>
<tr>
<td valign="middle" align="left">SVM</td>
<td valign="middle" align="left">0.825(0.602)</td>
<td valign="middle" align="left">0.793(0.359)</td>
<td valign="middle" align="left">0.859(0.636)</td>
<td valign="middle" align="left">0.825(0.459)</td>
<td valign="middle" align="left">0.909(0.639)</td>
</tr>
<tr>
<td valign="middle" align="left">RF</td>
<td valign="middle" align="left">0.805(0.759)</td>
<td valign="middle" align="left">0.783(0.528)</td>
<td valign="middle" align="left">0.822(0.864)</td>
<td valign="middle" align="left">0.802(0.655)</td>
<td valign="middle" align="left">0.891(0.848)</td>
</tr>
<tr>
<td valign="middle" align="left">XGBoost</td>
<td valign="middle" align="left">0.777(0.723)</td>
<td valign="middle" align="left">0.761(0.487)</td>
<td valign="middle" align="left">0.780(0.864)</td>
<td valign="middle" align="left">0.770(0.623)</td>
<td valign="middle" align="left">0.872(0.817)</td>
</tr>
<tr>
<td valign="middle" align="left">GBDT</td>
<td valign="middle" align="left">0.789(0.759)</td>
<td valign="middle" align="left">0.771(0.529)</td>
<td valign="middle" align="left">0.797(0.818)</td>
<td valign="middle" align="left">0.784(0.642)</td>
<td valign="middle" align="left">0.866(0.828)</td>
</tr>
<tr>
<td valign="middle" align="left">Super learner</td>
<td valign="middle" align="left">0.795(0.795)</td>
<td valign="middle" align="left">0.785(0.571)</td>
<td valign="middle" align="left">0.788(0.909)</td>
<td valign="middle" align="left">0.786(0.701)</td>
<td valign="middle" align="left">0.893(0.861)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="fnT2_1">
<label>a</label>
<p>The predictive performance values for the hold-out validation set are shown in parentheses.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Model interpretation</title>
<p>In this paper, the SHAP value was used to quantify the impact of each variable on the prediction of CAS, and the results are shown in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>. <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3A</bold>
</xref> shows the contribution of all the features to the prediction, which was sorted according to the average SHAP values. CS and age are the two most important predictors with the largest SHAP values, followed by RDW-SD, GLU and hypertension.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Results of the SHAP analysis. <bold>(A)</bold> Mean (|SHAP value|) of each variable; <bold>(B)</bold> the contribution of each variable in non-CAS individual I; <bold>(C)</bold> the contribution of each variable in CAS individual II.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fendo-15-1390352-g003.tif"/>
</fig>
<p>To further explain how each variable affects the occurrence of CAS, we illustrate two sample cases. <xref ref-type="fig" rid="f3">
<bold>Figures&#xa0;3B, C</bold>
</xref> depict the SHAP value of each variable for individuals I and II. The blue bars on the left (SHAP value less than 0) indicate variables that reduce the probability of the individual being predicted as CAS; the orange bars on the right (SHAP value greater than 0) indicate variables that increase the probability of the individual being predicted as CAS. Larger areas indicate greater impacts of that factor. For individual I, diabetes and an increase in glucose are the main reasons for the increased risk of CAS. Due to the absence of CS, relatively young age and other variables with negative impacts, the predicted probability of CAS for individuals is low. In contrast, for individual II, CS is the main reason for the increased risk of CAS, and most of the variables have a positive impact in predicting CAS. The probability of CAS for individual II is only slightly reduced by the absence of hypertension, diabetes and young age; thus, this individual is more likely to develop CAS in the future. Therefore, through the SHAP framework, we can directly determine the main causes for the increased individual probability of CAS; thus, corresponding interventions could be taken to reduce the risk.</p>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<label>4</label>
<title>Discussion</title>
<p>In this study, based on a routine health check-up cohort, we constructed a stacking ensemble prediction model for quantifying the risks of incident CAS. Demographic information, such as age and sex, and clinical factors, including BUN, LYM, AST, RDW-SD, RBC, MCHC, MPV, GLU, PLT, PEOS, WBC, CEA, CS, hypertension and DM, were important predictors of CAS.</p>
<p>We established five machine learning models to predict CAS and found that the performance of the individual models varied in the testing and validation sets. Most of the models performed better on the testing set and inferiorly on the hold-out validation set, indicating the overfitting problem. Therefore, we used the super learner algorithm to integrate the models, which significantly improved their performance. The super learner model not only demonstrated superior discrimination but also effectively managed overfitting, with AUC scores of 0.893 and 0.861 in the testing and validation sets, respectively. Our findings align with recent research that demonstrates the superior performance of stacking models in various biomedical applications (<xref ref-type="bibr" rid="B31">31</xref>&#x2013;<xref ref-type="bibr" rid="B33">33</xref>). Studies like those conducted by Zhou have shown that stacking models provide enhanced accuracy in predicting diabetes which is consistent with our results (<xref ref-type="bibr" rid="B34">34</xref>).</p>
<p>In accordance with several studies (<xref ref-type="bibr" rid="B4">4</xref>, <xref ref-type="bibr" rid="B5">5</xref>, <xref ref-type="bibr" rid="B35">35</xref>), age was identified as a risk factor for CAS. According to the results of the feature importance analysis for the three machine learning models and the SHAP explanatory framework, we found that age and CS were the two most important factors affecting the occurrence of CAS. The demographic shift towards an aging population warrants increased societal attention, given the anticipated rise in CAS incidence (<xref ref-type="bibr" rid="B36">36</xref>). Additionally, since CS is a symptom of CAS, its presence is an important signal of CAS, and these two groups of people in particular need to take corresponding measures to prevent CAS. Moreover, our analysis extends beyond conventional risk factors by incorporating endocrine-related markers within the predictive framework. The integration of these markers, including but not limited to abnormal blood glucose levels and hypertension, underscores the intricate relationship between endocrine dysfunctions and atherosclerosis. Through the SHAP framework, the contribution of each feature to the CAS risk is quantified, offering a personalized risk assessment. It underscores the necessity of a multifaceted risk assessment strategy that not only considers traditional factors like age and CS but also gives weight to the underlying endocrine dysfunctions contributing to the disease&#x2019;s pathogenesis.</p>
<p>The SHAP score becomes an invaluable tool for clinicians, enhancing the interpretability of machine learning predictions and enabling personalized preventive measures. For instance, older individuals with CS may benefit from increased screening and early intervention strategies, facilitating early detection and management of CAS. Similarly, for individuals with abnormal blood glucose or hypertension, personalized medical interventions including adjustments in medication and lifestyle changes such as diet and exercise could be advised based on their specific risk profiles. Regular monitoring of blood pressure and glucose levels can further aid early intervention and management, demonstrating the dynamic utility of predictive models in clinical settings.</p>
<p>One of the limitations of our study was that information about important risk factors for CAS, such as lifestyle, was not available. However, the model in our study still achieved acceptable performance without these predictors. Moreover, study subjects in the routine check-up cohort were limited to a single source area, and the prediction model was only internally validated. External validation with an independent population is needed to evaluate the generalizability of the model. Future studies will aim to collaborate with various institutions across different geographic regions to ensure that our models are robust and applicable to a broader population. This approach will not only help in validating our current model externally but also in assessing its effectiveness across different demographic settings.</p>
</sec>
<sec id="s5" sec-type="conclusion">
<label>5</label>
<title>Conclusion</title>
<p>In this study, we developed a stacking model to predict the risk of incident CAS, enhancing the application of machine learning in the disease prediction. This approach not only provides a new method for the risk calculation of CAS but also highlight the critical role of endocrine dysfunctions in CAS development. By integrating a comprehensive analysis of predictors, and utilizing SHAP for model interpretation, our model effectively identifies high-risk individuals. This allows for targeted interventions that could substantially reduce the health and economic burdens associated with CAS. The study demonstrates the potential of advanced machine learning techniques to enhance preventive healthcare strategies.</p>
</sec>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors. Further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s7" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>The studies involving humans were approved by the Ethics Committee of the First Affiliated Hospital of Shandong First Medical University, and informed consent was obtained from all the participants The studies were conducted in accordance with the local legislation and institutional requirements.</p>
</sec>
<sec id="s8" sec-type="author-contributions">
<title>Author contributions</title>
<p>XZ: Conceptualization, Formal analysis, Methodology, Writing &#x2013; review &amp; editing. CT: Formal analysis, Writing &#x2013; original draft. SW: Writing &#x2013; review &amp; editing. WL: Writing &#x2013; review &amp; editing. WY: Writing &#x2013; original draft. DW: Writing &#x2013; original draft. QW: Writing &#x2013; original draft. FT: Conceptualization, Data curation, Writing &#x2013; review &amp; editing.</p>
</sec>
</body>
<back>
<sec id="s9" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. The work was supported by National Natural Science Foundation of China (81903410&amp;71804093).</p>
</sec>
<sec id="s10" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>Author SW were employed by Shandong International Trust Co., Ltd.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s11" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fendo.2024.1390352/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fendo.2024.1390352/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet_1.pdf" id="SM1" mimetype="application/pdf"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sirimarco</surname> <given-names>G</given-names>
</name>
<name>
<surname>Amarenco</surname> <given-names>P</given-names>
</name>
<name>
<surname>Labreuche</surname> <given-names>J</given-names>
</name>
<name>
<surname>Touboul</surname> <given-names>P-J</given-names>
</name>
<name>
<surname>Alberts</surname> <given-names>M</given-names>
</name>
<name>
<surname>Goto</surname> <given-names>S</given-names>
</name>
<etal/>
</person-group>. <article-title>Carotid atherosclerosis and risk of subsequent coronary event in outpatients with atherothrombosis</article-title>. <source>Stroke</source>. (<year>2013</year>) <volume>44</volume>:<page-range>373&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1161/STROKEAHA.112.673129</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Martinez</surname> <given-names>E</given-names>
</name>
<name>
<surname>Martorell</surname> <given-names>J</given-names>
</name>
<name>
<surname>Riambau</surname> <given-names>V</given-names>
</name>
</person-group>. <article-title>Review of serum biomarkers in carotid atherosclerosis</article-title>. <source>J Vasc Surg</source>. (<year>2020</year>) <volume>71</volume>:<page-range>329&#x2013;41</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jvs.2019.04.488</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hollander</surname> <given-names>M</given-names>
</name>
<name>
<surname>Bots</surname> <given-names>ML</given-names>
</name>
<name>
<surname>Del Sol</surname> <given-names>AI</given-names>
</name>
<name>
<surname>Koudstaal</surname> <given-names>PJ</given-names>
</name>
<name>
<surname>Witteman</surname> <given-names>JMC</given-names>
</name>
<name>
<surname>Grobbee</surname> <given-names>DE</given-names>
</name>
<etal/>
</person-group>. <article-title>Carotid plaques increase the risk of stroke and subtypes of cerebral infarction in asymptomatic elderly: the Rotterdam study</article-title>. <source>Circulation</source>. (<year>2002</year>) <volume>105</volume>:<page-range>2872&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1161/01.CIR.0000018650.58984.75</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Taylor</surname> <given-names>BA</given-names>
</name>
<name>
<surname>Zaleski</surname> <given-names>AL</given-names>
</name>
<name>
<surname>Capizzi</surname> <given-names>JA</given-names>
</name>
<name>
<surname>Ballard</surname> <given-names>KD</given-names>
</name>
<name>
<surname>Troyanos</surname> <given-names>C</given-names>
</name>
<name>
<surname>Baggish</surname> <given-names>AL</given-names>
</name>
<etal/>
</person-group>. <article-title>Influence of chronic exercise on carotid atherosclerosis in marathon runners</article-title>. <source>BMJ Open</source>. (<year>2014</year>) <volume>4</volume>:<elocation-id>e004498</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1136/bmjopen-2013-004498</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>van den Munckhof</surname> <given-names>ICL</given-names>
</name>
<name>
<surname>Jones</surname> <given-names>H</given-names>
</name>
<name>
<surname>Hopman</surname> <given-names>MTE</given-names>
</name>
<name>
<surname>de Graaf</surname> <given-names>J</given-names>
</name>
<name>
<surname>Nyakayiru</surname> <given-names>J</given-names>
</name>
<name>
<surname>van Dijk</surname> <given-names>B</given-names>
</name>
<etal/>
</person-group>. <article-title>Relation between age and carotid artery intima-medial thickness: a systematic review</article-title>. <source>Clin Cardiol</source>. (<year>2018</year>) <volume>41</volume>:<fpage>698</fpage>&#x2013;<lpage>704</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/clc.22934</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname> <given-names>D</given-names>
</name>
<name>
<surname>Li</surname> <given-names>C</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Xiong</surname> <given-names>H</given-names>
</name>
<name>
<surname>Tian</surname> <given-names>X</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>W</given-names>
</name>
<etal/>
</person-group>. <article-title>Influence of blood pressure variability on early carotid atherosclerosis in hypertension with and without diabetes</article-title>. <source>Med (Baltimore)</source>. (<year>2016</year>) <volume>95</volume>:<elocation-id>e3864</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1097/MD.0000000000003864</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yuan</surname> <given-names>T</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>T</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>H</given-names>
</name>
<name>
<surname>Fu</surname> <given-names>D</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>New insights into oxidative stress and inflammation during diabetes mellitus-accelerated atherosclerosis</article-title>. <source>Redox Biol</source>. (<year>2019</year>) <volume>20</volume>:<page-range>247&#x2013;60</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.redox.2018.09.025</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Mao</surname> <given-names>H</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>H</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>P</given-names>
</name>
<name>
<surname>Garry</surname> <given-names>W</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>H</given-names>
</name>
<etal/>
</person-group>. <article-title>Machine learning-based models to support decision-making in emergency department triage for patients with suspected cardiovascular disease</article-title>. <source>Int J Med Inform</source>. (<year>2021</year>) <volume>145</volume>:<elocation-id>104326</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ijmedinf.2020.104326</pub-id>
</citation>
</ref>
<ref id="B9">
<label>9</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Byra</surname> <given-names>M</given-names>
</name>
<name>
<surname>Galperin</surname> <given-names>M</given-names>
</name>
<name>
<surname>Ojeda-Fournier</surname> <given-names>H</given-names>
</name>
<name>
<surname>Olson</surname> <given-names>L</given-names>
</name>
<name>
<surname>O&#x2019;Boyle</surname> <given-names>M</given-names>
</name>
<name>
<surname>Comstock</surname> <given-names>C</given-names>
</name>
<etal/>
</person-group>. <article-title>Breast mass classification in sonography with transfer learning using a deep convolutional neural network and color conversion</article-title>. <source>Med Phys</source>. (<year>2019</year>) <volume>46</volume>:<page-range>746&#x2013;55</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/mp.13361</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Danielsen</surname> <given-names>AA</given-names>
</name>
<name>
<surname>Fenger</surname> <given-names>MHJ</given-names>
</name>
<name>
<surname>&#xd8;stergaard</surname> <given-names>SD</given-names>
</name>
<name>
<surname>Nielbo</surname> <given-names>KL</given-names>
</name>
<name>
<surname>Mors</surname> <given-names>O</given-names>
</name>
</person-group>. <article-title>Predicting mechanical restraint of psychiatric inpatients by applying machine learning on electronic health data</article-title>. <source>Acta Psychiatr Scand</source>. (<year>2019</year>) <volume>140</volume>:<page-range>147&#x2013;57</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/acps.13061</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname> <given-names>D</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Su</surname> <given-names>C</given-names>
</name>
<name>
<surname>Han</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Duan</surname> <given-names>X</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>R</given-names>
</name>
<etal/>
</person-group>. <article-title>Copy number variation in plasma as a tool for lung cancer prediction using Extreme Gradient Boosting (XGBoost) classifier</article-title>. <source>Thorac Cancer</source>. (<year>2020</year>) <volume>11</volume>:<fpage>95</fpage>&#x2013;<lpage>102</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/1759-7714.13204</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schultebraucks</surname> <given-names>K</given-names>
</name>
<name>
<surname>Shalev</surname> <given-names>AY</given-names>
</name>
<name>
<surname>Michopoulos</surname> <given-names>V</given-names>
</name>
<name>
<surname>Grudzen</surname> <given-names>CR</given-names>
</name>
<name>
<surname>Shin</surname> <given-names>S-M</given-names>
</name>
<name>
<surname>Stevens</surname> <given-names>JS</given-names>
</name>
<etal/>
</person-group>. <article-title>A validated predictive algorithm of post-traumatic stress course following emergency department admission after a traumatic stressor</article-title>. <source>Nat Med</source>. (<year>2020</year>) <volume>26</volume>:<page-range>1084&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41591-020-0951-z</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shim</surname> <given-names>J-G</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>DW</given-names>
</name>
<name>
<surname>Ryu</surname> <given-names>K-H</given-names>
</name>
<name>
<surname>Cho</surname> <given-names>E-A</given-names>
</name>
<name>
<surname>Ahn</surname> <given-names>J-H</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>J-I</given-names>
</name>
<etal/>
</person-group>. <article-title>Application of machine learning approaches for osteoporosis risk prediction in postmenopausal women</article-title>. <source>Arch Osteoporos</source>. (<year>2020</year>) <volume>15</volume>:<fpage>169</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11657-020-00802-8</pub-id>
</citation>
</ref>
<ref id="B14">
<label>14</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>van Os</surname> <given-names>HJA</given-names>
</name>
<name>
<surname>Ramos</surname> <given-names>LA</given-names>
</name>
<name>
<surname>Hilbert</surname> <given-names>A</given-names>
</name>
<name>
<surname>van Leeuwen</surname> <given-names>M</given-names>
</name>
<name>
<surname>van Walderveen</surname> <given-names>MAA</given-names>
</name>
<name>
<surname>Kruyt</surname> <given-names>ND</given-names>
</name>
<etal/>
</person-group>. <article-title>Predicting outcome of endovascular treatment for acute ischemic stroke: potential value of machine learning algorithms</article-title>. <source>Front Neurol</source>. (<year>2018</year>) <volume>9</volume>:<elocation-id>784</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fneur.2018.00784</pub-id>
</citation>
</ref>
<ref id="B15">
<label>15</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname> <given-names>F</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Zhi</surname> <given-names>H</given-names>
</name>
<name>
<surname>Dong</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Hi</surname> <given-names>H</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>S</given-names>
</name>
<etal/>
</person-group>. <article-title>Artificial intelligence in healthcare: past, present and future</article-title>. <source>Stroke Vasc Neurol</source>. (<year>2017</year>) <volume>2</volume>:<page-range>230&#x2013;43</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1136/svn-2017-000101</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liang</surname> <given-names>N</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>C</given-names>
</name>
<name>
<surname>Duan</surname> <given-names>J</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>X</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y</given-names>
</name>
</person-group>. <article-title>Efficacy prediction of noninvasive ventilation failure based on the stacking ensemble algorithm and autoencoder</article-title>. <source>BMC Med Inform Decis Mak</source>. (<year>2022</year>) <volume>22</volume>:<fpage>27</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12911-022-01767-z</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liang</surname> <given-names>M</given-names>
</name>
<name>
<surname>Chang</surname> <given-names>T</given-names>
</name>
<name>
<surname>An</surname> <given-names>B</given-names>
</name>
<name>
<surname>Duan</surname> <given-names>X</given-names>
</name>
<name>
<surname>Du</surname> <given-names>L</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X</given-names>
</name>
<etal/>
</person-group>. <article-title>A stacking ensemble learning framework for genomic prediction</article-title>. <source>Front Genet</source>. (<year>2021</year>) <volume>12</volume>:<elocation-id>600040</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fgene.2021.600040</pub-id>
</citation>
</ref>
<ref id="B18">
<label>18</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Verma</surname> <given-names>AK</given-names>
</name>
<name>
<surname>Pal</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Prediction of skin disease with three different feature selection techniques using stacking ensemble method</article-title>. <source>Appl Biochem Biotechnol</source>. (<year>2020</year>) <volume>191</volume>:<page-range>637&#x2013;56</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s12010-019-03222-8</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gantenberg</surname> <given-names>JR</given-names>
</name>
<name>
<surname>McConeghy</surname> <given-names>KW</given-names>
</name>
<name>
<surname>Howe</surname> <given-names>CJ</given-names>
</name>
<name>
<surname>Steingrimsson</surname> <given-names>J</given-names>
</name>
<name>
<surname>van Aalst</surname> <given-names>R</given-names>
</name>
<name>
<surname>Chit</surname> <given-names>A</given-names>
</name>
<etal/>
</person-group>. <article-title>Predicting seasonal influenza hospitalizations using an ensemble super learner: A simulation study</article-title>. <source>Am J Epidemiol</source>. (<year>2023</year>) <volume>192</volume>:<page-range>1688&#x2013;700</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/aje/kwad113</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname> <given-names>X</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>P</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Kuang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>L</given-names>
</name>
</person-group>. <article-title>Using the Super Learner algorithm to predict risk of major adverse cardiovascular events after percutaneous coronary intervention in patients with myocardial infarction</article-title>. <source>BMC Med Res Methodol</source>. (<year>2024</year>) <volume>24</volume>:<fpage>59</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12874-024-02179-5</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>F</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>J</given-names>
</name>
<name>
<surname>Ge</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Wen</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Yue</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Hayashida</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>Computational prediction and interpretation of both general and specific types of promoters in Escherichia coli by exploiting a stacked ensemble-learning framework</article-title>. <source>Brief Bioinform</source>. (<year>2021</year>) <volume>22</volume>:<page-range>2126&#x2013;40</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bib/bbaa049</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Lundberg</surname> <given-names>SM</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>S-I</given-names>
</name>
</person-group>. (<year>2017</year>). <article-title>A unified approach to interpreting model predictions</article-title>, in: <conf-name>Proceedings of the 31st International Conference on Neural Information Processing Systems</conf-name>, <conf-loc>Curran Associates Inc., Red Hook, NY, USA</conf-loc>. pp. <page-range>4768&#x2013;77</page-range>.</citation>
</ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>W</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>S</given-names>
</name>
<name>
<surname>Ye</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>P</given-names>
</name>
<name>
<surname>Xia</surname> <given-names>X</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>Prediction of lung metastases in thyroid cancer using machine learning based on SEER database</article-title>. <source>Cancer Med</source>. (<year>2022</year>) <volume>11</volume>:<page-range>2503&#x2013;15</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/cam4.4617</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kop</surname> <given-names>R</given-names>
</name>
<name>
<surname>Hoogendoorn</surname> <given-names>M</given-names>
</name>
<name>
<surname>Teije</surname> <given-names>AT</given-names>
</name>
<name>
<surname>B&#xfc;chner</surname> <given-names>FL</given-names>
</name>
<name>
<surname>Slottje</surname> <given-names>P</given-names>
</name>
<name>
<surname>Moons</surname> <given-names>LMG</given-names>
</name>
<etal/>
</person-group>. <article-title>Predictive modeling of colorectal cancer using a dedicated pre-processing pipeline on routine electronic medical records</article-title>. <source>Comput Biol Med</source>. (<year>2016</year>) <volume>76</volume>:<page-range>30&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compbiomed.2016.06.019</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Singal</surname> <given-names>AG</given-names>
</name>
<name>
<surname>Mukherjee</surname> <given-names>A</given-names>
</name>
<name>
<surname>Elmunzer</surname> <given-names>BJ</given-names>
</name>
<name>
<surname>Higgins</surname> <given-names>PDR</given-names>
</name>
<name>
<surname>Lok</surname> <given-names>AS</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>Machine learning algorithms outperform conventional regression models in predicting development of hepatocellular carcinoma</article-title>. <source>Am J Gastroenterol</source>. (<year>2013</year>) <volume>108</volume>:<page-range>1723&#x2013;30</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/ajg.2013.332</pub-id>
</citation>
</ref>
<ref id="B26">
<label>26</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname> <given-names>H</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>C</given-names>
</name>
<name>
<surname>Zhai</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Tian</surname> <given-names>X</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>L</given-names>
</name>
<etal/>
</person-group>. <article-title>Identifying diseases that cause psychological trauma and social avoidance by GCN-Xgboost</article-title>. <source>BMC Bioinf</source>. (<year>2020</year>) <volume>21</volume>:<fpage>504</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12859-020-03847-1</pub-id>
</citation>
</ref>
<ref id="B27">
<label>27</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>van der Laan</surname> <given-names>MJ</given-names>
</name>
<name>
<surname>Polley</surname> <given-names>EC</given-names>
</name>
<name>
<surname>Hubbard</surname> <given-names>AE</given-names>
</name>
</person-group>. <article-title>Super learner</article-title>. <source>Stat Appl Genet Mol Biol</source>. (<year>2007</year>) <volume>6</volume>:<fpage>Article25</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.2202/1544-6115.1309</pub-id>
</citation>
</ref>
<ref id="B28">
<label>28</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Naimi</surname> <given-names>AI</given-names>
</name>
<name>
<surname>Balzer</surname> <given-names>LB</given-names>
</name>
</person-group>. <article-title>Stacked generalization: an introduction to super learning</article-title>. <source>Eur J Epidemiol</source>. (<year>2018</year>) <volume>33</volume>:<page-range>459&#x2013;64</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10654-018-0390-z</pub-id>
</citation>
</ref>
<ref id="B29">
<label>29</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dalal</surname> <given-names>S</given-names>
</name>
<name>
<surname>Onyema</surname> <given-names>EM</given-names>
</name>
<name>
<surname>Malik</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Hybrid XGBoost model with hyperparameter tuning for prediction of liver disease with better accuracy</article-title>. <source>World J Gastroenterol</source>. (<year>2022</year>) <volume>28</volume>:<page-range>6551&#x2013;63</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.3748/wjg.v28.i46.6551</pub-id>
</citation>
</ref>
<ref id="B30">
<label>30</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>&#x160;trumbelj</surname> <given-names>E</given-names>
</name>
<name>
<surname>Kononenko</surname> <given-names>I</given-names>
</name>
</person-group>. <article-title>Explaining prediction models and individual predictions with feature contributions</article-title>. <source>Knowl Inf Syst</source>. (<year>2014</year>) <volume>41</volume>:<page-range>647&#x2013;65</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10115-013-0679-x</pub-id>
</citation>
</ref>
<ref id="B31">
<label>31</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ahmed</surname> <given-names>SH</given-names>
</name>
<name>
<surname>Bose</surname> <given-names>DB</given-names>
</name>
<name>
<surname>Khandoker</surname> <given-names>R</given-names>
</name>
<name>
<surname>Rahman</surname> <given-names>MS</given-names>
</name>
</person-group>. <article-title>StackDPP: a stacking ensemble based DNA-binding protein prediction model</article-title>. <source>BMC Bioinf</source>. (<year>2024</year>) <volume>25</volume>:<fpage>111</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12859-024-05714-9</pub-id>
</citation>
</ref>
<ref id="B32">
<label>32</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Biswas</surname> <given-names>SK</given-names>
</name>
<name>
<surname>Nath Boruah</surname> <given-names>A</given-names>
</name>
<name>
<surname>Saha</surname> <given-names>R</given-names>
</name>
<name>
<surname>Raj</surname> <given-names>RS</given-names>
</name>
<name>
<surname>Chakraborty</surname> <given-names>M</given-names>
</name>
<name>
<surname>Bordoloi</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>Early detection of Parkinson disease using stacking ensemble method</article-title>. <source>Comput Methods Biomech BioMed Engin</source>. (<year>2023</year>) <volume>26</volume>:<page-range>527&#x2013;39</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/10255842.2022.2072683</pub-id>
</citation>
</ref>
<ref id="B33">
<label>33</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kapila</surname> <given-names>R</given-names>
</name>
<name>
<surname>Saleti</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Optimizing fetal health prediction: Ensemble modeling with fusion of feature selection and extraction techniques for cardiotocography data</article-title>. <source>Comput Biol Chem</source>. (<year>2023</year>) <volume>107</volume>:<elocation-id>107973</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compbiolchem.2023.107973</pub-id>
</citation>
</ref>
<ref id="B34">
<label>34</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname> <given-names>H</given-names>
</name>
<name>
<surname>Xin</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Li</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>A diabetes prediction model based on Boruta feature selection and ensemble learning</article-title>. <source>BMC Bioinf</source>. (<year>2023</year>) <volume>24</volume>:<fpage>224</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12859-023-05300-5</pub-id>
</citation>
</ref>
<ref id="B35">
<label>35</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fine-Edelstein</surname> <given-names>JS</given-names>
</name>
<name>
<surname>Wolf</surname> <given-names>PA</given-names>
</name>
<name>
<surname>O&#x2019;Leary</surname> <given-names>DH</given-names>
</name>
<name>
<surname>Poehlman</surname> <given-names>H</given-names>
</name>
<name>
<surname>Belange</surname> <given-names>AJ</given-names>
</name>
<name>
<surname>Kase</surname> <given-names>CS</given-names>
</name>
<etal/>
</person-group>. <article-title>Precursors of extracranial carotid atherosclerosis in the Framingham Study</article-title>. <source>Neurology</source>. (<year>1994</year>) <volume>44</volume>:<page-range>1046&#x2013;50</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1212/WNL.44.6.1046</pub-id>
</citation>
</ref>
<ref id="B36">
<label>36</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fan</surname> <given-names>J</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>M</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>J</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>J</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>S</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>J</given-names>
</name>
<name>
<surname>Yao</surname> <given-names>Q</given-names>
</name>
<etal/>
</person-group>. <article-title>The prediction of asymptomatic carotid atherosclerosis with electronic health records: a comparative study of six machine learning models</article-title>. <source>BMC Med Inform Decis Mak</source>. (<year>2021</year>) <volume>21</volume>:<fpage>115</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12911-021-01480-3</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>