<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Physiol.</journal-id>
<journal-title>Frontiers in Physiology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Physiol.</abbrev-journal-title>
<issn pub-type="epub">1664-042X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1464744</article-id>
<article-id pub-id-type="doi">10.3389/fphys.2024.1464744</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Physiology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>An ensemble model for predicting dyslipidemia using 3-years continuous physical examination data</article-title>
<alt-title alt-title-type="left-running-head">Zhang et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fphys.2024.1464744">10.3389/fphys.2024.1464744</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Naiwen</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Guo</surname>
<given-names>Xiaolong</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yu</surname>
<given-names>Xiaxia</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Tan</surname>
<given-names>Zhen</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/841600/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Cai</surname>
<given-names>Feiyue</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Dai</surname>
<given-names>Ping</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Guo</surname>
<given-names>Jing</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/625312/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Dan</surname>
<given-names>Guo</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/990781/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Biomedical Engineering</institution>, <institution>Shenzhen University Medical School</institution>, <institution>Shenzhen University</institution>, <addr-line>Shenzhen</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Health Management Center, Shenzhen University General Hospital, Shenzhen University Clinical Medical Academy, Shenzhen University</institution>, <addr-line>Shenzhen</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Shenzhen Nanshan District General Practice Alliance</institution>, <addr-line>Shenzhen</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Department of Endocrinology and Metabolism, Shenzhen University General Hospital</institution>, <addr-line>Shenzhen</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/500687/overview">Yunlong Huo</ext-link>, Shanghai Jiao Tong University, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2814907/overview">Yongliang Fan</ext-link>, PKU-HKUST Shenzhen-Hongkong Institution, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2054538/overview">Xu Huang</ext-link>, Nanjing University of Science and Technology, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Guo Dan, <email>danguo@szu.edu.cn</email>; Jing Guo, <email>guojing198564@hotmail.com</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>24</day>
<month>10</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1464744</elocation-id>
<history>
<date date-type="received">
<day>15</day>
<month>07</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>11</day>
<month>10</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Zhang, Guo, Yu, Tan, Cai, Dai, Guo and Dan.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Zhang, Guo, Yu, Tan, Cai, Dai, Guo and Dan</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Background</title>
<p>Dyslipidemia has emerged as a significant clinical risk, with its associated complications, including atherosclerosis and ischemic cerebrovascular disease, presenting a grave threat to human well-being. Hence, it holds paramount importance to precisely predict the onset of dyslipidemia. This study aims to use ensemble technology to establish a machine learning model for the prediction of dyslipidemia.</p>
</sec>
<sec>
<title>Methods</title>
<p>This study included three consecutive years of physical examination data of 2,479 participants, and used the physical examination data of the first two years to predict whether the participants would develop dyslipidemia in the third year. Feature selection was conducted through statistical methods and the analysis of mutual information between features. Five machine learning models, including support vector machine (SVM), logistic regression (LR), random forest (RF), K nearest neighbor (KNN) and extreme gradient boosting (XGBoost), were utilized as base learners to construct the ensemble model. Area under the receiver operating characteristic curve (AUC), calibration curves, and decision curve analysis (DCA) were used to evaluate the model.</p>
</sec>
<sec>
<title>Results</title>
<p>Experimental results show that the ensemble model achieves superior performance across several metrics, achieving an AUC of 0.88 &#xb1; 0.01 (<italic>P</italic> &#x3c; 0.001), surpassing the base learners by margins of 0.04 to 0.20. Calibration curves and DCA exhibited good predictive performance as well. Furthermore, this study explores the minimal necessary feature set for accurate prediction, finding that just the top 12 features were required for dependable outcomes. Among them, HbA1c and CEA are key indicators for model construction.</p>
</sec>
<sec>
<title>Conclusions</title>
<p>Our results suggest that the proposed ensemble model has good predictive performance and has the potential to become an effective tool for personal health management.</p>
</sec>
</abstract>
<kwd-group>
<kwd>dyslipidemia</kwd>
<kwd>prediction</kwd>
<kwd>physical examination data</kwd>
<kwd>machine learning</kwd>
<kwd>ensemble model</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Physiology and Medicine</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Dyslipidemia has been recognized as a major risk factor for cardiovascular disease, which seriously endangers people&#x2019;s health (<xref ref-type="bibr" rid="B9">Hedayatnia et al., 2020</xref>; <xref ref-type="bibr" rid="B42">Zhao et al., 2022</xref>; <xref ref-type="bibr" rid="B6">Doi et al., 2022</xref>). Over the past 3 decades, the global incidence of dyslipidemia has risen markedly, representing a grave threat to public health (<xref ref-type="bibr" rid="B27">Pirillo et al., 2021</xref>). A report from the World Health Organization found that 4.5% of the global mortality rate for people aged 18 and over and 2% of disability-adjusted life years are due to high cholesterol (<xref ref-type="bibr" rid="B25">Organization, 2021</xref>). Research suggests that the incidence density of dyslipidemia in China is as high as 101/1,000, 121/1,000 in men and 69/1,000 in women (<xref ref-type="bibr" rid="B41">Zhang et al., 2019</xref>). Cardiovascular events caused by high cholesterol in China have increased dramatically, and may reach 9.2 million between 2010 and 2030 (<xref ref-type="bibr" rid="B23">Moran et al., 2010</xref>). Dyslipidemia is defined as elevated plasma concentrations of total cholesterol (TC), low-density-lipoprotein-cholesterol (LDL-C), or triglycerides (TG), or a low plasma concentration of high-density-lipoprotein-cholesterol (HDL-C) or a combination of these features (<xref ref-type="bibr" rid="B15">Klop et al., 2013</xref>; <xref ref-type="bibr" rid="B29">Raja et al., 2023</xref>). Its complex pathogenesis, coupled with the absence of conspicuous symptoms in early stages, complicates its detection and often leads to its underestimation. Therefore, the prediction of dyslipidemia occurrence plays an important role in improving its preventive and therapeutic effects.</p>
<p>In recent years, several studies have investigated the primary risk factors associated with dyslipidemia, including body mass index (BMI), waist-to-hip ratio, obesity, and gender, yielding significant insights (<xref ref-type="bibr" rid="B37">Vekic et al., 2019</xref>; <xref ref-type="bibr" rid="B13">Kavey, 2023</xref>; <xref ref-type="bibr" rid="B31">Ruan et al., 2024</xref>). Qi et al. (<xref ref-type="bibr" rid="B28">Qi et al., 2015</xref>) analyzed 5,375 participants aged 18 and older to ascertain the prevalence of dyslipidemia and its associated risk factors. Similarly, Ni et al. (<xref ref-type="bibr" rid="B24">Ni et al., 2015</xref>) employed a multi-stage stratified cluster random sampling approach to survey 1,995 adults, averaging 46.56 years in age. The findings indicate a substantial correlation between dyslipidemia and factors such as age, smoking, hypertension, diabetes, and BMI. Although these studies have helped identify risk factors for dyslipidemia, they do not have the ability to predict the long-term risk of dyslipidemia. Some studies have noted the limitations of these methods and proposed various approaches for prediction using logistic regression or Cox proportional hazards models (<xref ref-type="bibr" rid="B13">Kavey, 2023</xref>; <xref ref-type="bibr" rid="B16">Lai et al., 2022</xref>; <xref ref-type="bibr" rid="B39">Wang J.-S. et al., 2022</xref>; <xref ref-type="bibr" rid="B14">Kim et al., 2021</xref>; <xref ref-type="bibr" rid="B17">Lan et al., 2023</xref>). As comprehension of health outcomes&#x2019; complexity deepens, it becomes evident that traditional models, limited by their inability to account for non-linear associations, fall short of accurately encapsulating health outcome intricacies (<xref ref-type="bibr" rid="B5">De Silva et al., 2020</xref>).</p>
<p>Machine learning (ML) is a powerful computer-assisted data mining and analysis method that can handle large, complex, and diverse data. It has been widely used in healthcare applications, including disease risk prediction and medical diagnosis (<xref ref-type="bibr" rid="B19">Li et al., 2023</xref>; <xref ref-type="bibr" rid="B11">Ibrahim and Abdulazeez, 2021</xref>). ML has powerful nonlinear fitting capabilities and can solve this problem well. Previous studies (<xref ref-type="bibr" rid="B41">Zhang et al., 2019</xref>; <xref ref-type="bibr" rid="B33">Sasagawa et al., 2024</xref>) have developed some prediction models for dyslipidemia using algorithms such as the random survival forest model, demonstrating the ML&#x2019;s potential in predicting dyslipidemia. Despite these advancements, the application of ML methods in dyslipidemia prediction remains underexplored. These studies also have some shortcomings, such as the reliance on a singular prediction model and the lack of comprehensive validation of different models, making it hard to ensure the stability and applicability of the methodology. Ensemble technology uses the excellent integration ability of the meta-learner on the results of the base learners to achieve more effective performance than a single model. It has been successfully applied in prediction tasks (<xref ref-type="bibr" rid="B21">Lu et al., 2024</xref>; <xref ref-type="bibr" rid="B34">Sun et al., 2024</xref>; <xref ref-type="bibr" rid="B40">Zhang et al., 2022</xref>).</p>
<p>Therefore, in this study, we aimed to use ensemble technology to develop a reliable dyslipidemia prediction model. By integrating the advantages of different machine learning models and making full use of 3 years of continuous physical examination data of non-diseased people, an ensemble model that can effectively predict dyslipidemia was constructed.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<sec id="s2-1">
<title>2.1 Participants and data collection</title>
<p>The overall process of the experiment is shown in <xref ref-type="fig" rid="F1">Figure 1</xref>. We used medical examination data provided by Shenzhen University General Hospital, China, covering the period from December 2018 to December 2022. All participants received a medical examination at the hospital. Ethical approval was obtained from the Ethics Committee of Shenzhen University, Shenzhen (approval number: PN-202300093). Informed consent was waived due to the retrospective nature of the study. The research adhered to the principles outlined in the Declaration of Helsinki.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The analysis workflow for prediction of dyslipidemia from EHR data.</p>
</caption>
<graphic xlink:href="fphys-15-1464744-g001.tif"/>
</fig>
<p>The electronic medical records of participants with a history of undergoing multiple years&#x2019; worth of physical examinations were meticulously reviewed. Our inclusion criteria comprised two distinct categories of physical examination subjects, i.e., individuals exhibiting consistent normal blood lipid levels across three consecutive physical examinations, and those with normal blood lipid levels during the initial two physical examinations but displaying abnormal blood lipid levels in the subsequent third examination. By including these two different participants in the study, we aim to comprehensively understand the occurrence and development mechanism of dyslipidemia, and provide a more accurate reference for future intervention and prediction. In accordance with the 2023 China guidelines for lipid management (<xref ref-type="bibr" rid="B12">Jian-Jun et al., 2023</xref>), dyslipidemia was precisely defined as the presence of TC &#x2265; 5.2 mmol/L, TG &#x2265; 1.7 mmol/L, LDL-C &#x2265; 3.4 mmol/L, and/or HDL-C &#x3c; 1.0 mmol/L.</p>
<p>The data content is primarily categorized into two groups: demographic data and laboratory test results. Every physical examination will meticulously document individual demographic information, encompassing age, gender, height, weight, physical examination date, blood pressure, pulse rate, and BMI. Laboratory findings are likewise derived from each physical examination record. The encompassing examination indicators comprise blood cell analysis, urinalysis, tumor markers, blood glucose test, blood lipid test, liver function, kidney function, and thyroid function.</p>
</sec>
<sec id="s2-2">
<title>2.2 Development and validation of machine learning models</title>
<sec id="s2-2-1">
<title>2.2.1 Machine learning models</title>
<p>In this study, five different machine learning models were employed for predictive modeling of dyslipidemia, namely, support vector machine (SVM), logistic regression (LR), random forest (RF), K nearest neighbor (KNN) and extreme gradient boosting (XGBoost).</p>
<p>SVM (<xref ref-type="bibr" rid="B36">Vapnik, 1999</xref>) is a powerful machine learning algorithm that can be used to solve classification problems. The core idea of SVM is to find an optimal decision boundary, which can divide different categories of datapoints in the feature space.</p>
<p>LR (<xref ref-type="bibr" rid="B4">Cox, 1958</xref>) is a statistical method commonly used to solve binary classification problems. Logistic regression models map the output values to a range between 0 and 1 by passing linear combinations of independent variables to logistic functions, which not only provide a prediction of the occurrence of an event, but also account for the effect of independent variables on the probability of an event.</p>
<p>RF (<xref ref-type="bibr" rid="B1">Breiman, 2001</xref>) is a powerful ensemble learning algorithm that is widely used in classification tasks. It works based on the construction of multiple decision trees, each based on a different subset of data and features, which helps to reduce the risk of overfitting and improve the robustness of the model.</p>
<p>KNN (<xref ref-type="bibr" rid="B8">Fix and Hodges, 1989</xref>) is a supervised learning algorithm that is widely used in classification problems by using information from neighbors to make predictions. It is based on the proximity between samples, especially using the labels of the K closest training samples to make predictions.</p>
<p>XGBoost (<xref ref-type="bibr" rid="B3">Chen and Guestrin, 2016</xref>) is a widely used ensemble learning method that performs well in a variety of machine learning tasks. The core principle of XGBoost is to iteratively add new weak models to correct the errors of the model in the previous round of iterations and build an efficient prediction model.</p>
<p>Each of the above machine learning models has been carefully configured to achieve accurate prediction of dyslipidemia. Specifically, we first selected the commonly used hyperparameters and candidate values that need to be optimized for each model, then used grid search to optimize the hyperparameters of each model, and used five-fold cross validation to select the best parameters. The specific parameter selection and optimization results are shown in <xref ref-type="table" rid="T1">Table 1</xref>. In addition, based on the above five machine learning models, ensemble technology will be used to effectively fuse their prediction results to achieve more accurate prediction performance.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Specific parameter selection and optimization results of each model.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model</th>
<th align="center">Model parameter</th>
<th align="center">Range</th>
<th align="center">Parameter after optimization</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="3" align="center">SVM</td>
<td align="center">C</td>
<td align="center">[0.1, 0.5, 1]</td>
<td align="center">0.1</td>
</tr>
<tr>
<td align="center">kernel</td>
<td align="center">[&#x201c;rbf&#x201d;, &#x201c;linear&#x201d;, &#x201c;poly&#x201d;]</td>
<td align="center">&#x201c;linear&#x201d;</td>
</tr>
<tr>
<td align="center">gamma</td>
<td align="center">[0.05, 0.1, 0.15]</td>
<td align="center">0.1</td>
</tr>
<tr>
<td rowspan="3" align="center">LR</td>
<td align="center">C</td>
<td align="center">[50, 100, 150]</td>
<td align="center">100</td>
</tr>
<tr>
<td align="center">max_iter</td>
<td align="center">[1,000, 2000, 3,000]</td>
<td align="center">2,000</td>
</tr>
<tr>
<td align="center">solver</td>
<td align="center">[&#x201c;lbfgs&#x201d;, &#x201c;newton-cholesky&#x201d;, &#x201c;sag&#x201d;]</td>
<td align="center">&#x201c;newton-cholesky&#x201d;</td>
</tr>
<tr>
<td rowspan="3" align="center">RF</td>
<td align="center">n_estimators</td>
<td align="center">[10, 15, 20]</td>
<td align="center">15</td>
</tr>
<tr>
<td align="center">max_depth</td>
<td align="center">[6, 8, 10]</td>
<td align="center">10</td>
</tr>
<tr>
<td align="center">max_features</td>
<td align="center">[&#x201c;sqrt&#x201d;, &#x201c;log2&#x201d;]</td>
<td align="center">&#x201c;sqrt&#x201d;</td>
</tr>
<tr>
<td align="center">KNN</td>
<td align="center">n_neighbors</td>
<td align="center">[20, 30, 40]</td>
<td align="center">40</td>
</tr>
<tr>
<td rowspan="3" align="center">XGBoost</td>
<td align="center">n_estimators</td>
<td align="center">[10, 50, 100]</td>
<td align="center">50</td>
</tr>
<tr>
<td align="center">max_depth</td>
<td align="center">[2, 4, 6]</td>
<td align="center">2</td>
</tr>
<tr>
<td align="center">learning_rate</td>
<td align="center">[0.05, 0.1, 0.15]</td>
<td align="center">0.05</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-2-2">
<title>2.2.2 Feature selection</title>
<p>In this study, we first addressed the heterogeneity in physical examination items across subjects by excluding those for which data were available for less than one-third of the cohort. For the remaining dataset, missing values were imputed using either mean or mode, depending on the nature of the data, thus completing the data preprocessing phase. We then defined the sets of index values for each subject in the first and second years as <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>f</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, respectively, and used <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>f</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and their difference <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>f</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> as features, identifying a total of 204 features. Recognizing the potential for irrelevant or redundant information within these features, the study implemented a two-step feature selection strategy. Firstly, we used t-test or <inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c7;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> test to identify features that showed significant differences between the dyslipidemia group and the non-dyslipidemia group, excluding those with <italic>P</italic>-values above 0.05. We then applied mutual information to remove redundant features, ensuring that the selected features were both statistically significant and independent across the groups. We explored the optimal number of features (N) to minimize redundancy while retaining sufficient discriminatory information. This approach enabled us to investigate the optimal number of features required to maintain model performance, experimenting with N values in increments of two from 2 to 20. This process facilitated effective feature selection, optimizing the efficiency of feature utilization.</p>
</sec>
<sec id="s2-2-3">
<title>2.2.3 Model building and evaluation</title>
<p>To accurately predict dyslipidemia onset utilizing routine physical examination data, this study introduced a stacking ensemble model executed in two stages. The first stage employed five base learners, including LR, SVM, KNN, RF, and XGBoost, to produce preliminary outputs. These outputs, alongside selected key features, serve as inputs for the second stage. The second stage integrated these inputs to train and establish the final predictive model.</p>
<p>In the first stage, to mitigate the risk of overfitting, each base learner underwent training and evaluation employing a five-fold cross-validation approach. This entailed partitioning the data into five subsets, with each subset serving once as the test set while the model trains on the remaining four. The model then predicted outcomes for both the training and test sets, generating sets of predictions for each. Concurrently, key features were identified based on their recurrence, with those appearing more than three times across five folds deemed significant. The outputs from this stage, comprising both the predicted values and the identified key features for the training and test sets, were then forwarded as inputs to the second stage. Moreover, an analysis to assess the impact of varying the number of features on model performance was conducted, aiming to ascertain the most effective feature set for the predictive model.</p>
<p>In the second stage, based on the results of each base learner in the first stage, XGBoost was chosen to develop the ensemble model for final predictions. To ensure robustness and validity, the five-fold cross-validation technique was reapplied. The training dataset encompassed the predictive outcomes and pivotal features from the first stage, generated by the five base learners in the best feature set. Similarly, the test dataset was constituted of analogous predictions and features, also from the same base models. This construction of new training and test datasets addresses and circumvents potential issues of data leakage. Ultimately, the ensemble model, having been thoroughly trained, performs the final predictive analysis on the test dataset.</p>
</sec>
</sec>
<sec id="s2-3">
<title>2.3 Statistical analysis</title>
<p>Clinical factors were analyzed using Student&#x2019;s t-test, Mann-Whitney U test, or Chi-square test according to the data distribution. Multiple criteria, including sensitivity, specificity, accuracy, and the area under the ROC curve (AUC), were used to evaluate the effectiveness of these models. Calibration curve and decision curve analysis were used to evaluate clinical usability. In addition, to facilitate understanding of the contribution of the second stage model input features to the prediction score, we calculated the SHapley Additive exPlanations (SHAP) values and illustrated them graphically. Statistical analyses were performed using Python (version 3.9) or Medcalc (version 22).</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<sec id="s3-1">
<title>3.1 Participant characteristics</title>
<p>The dataset encompasses 7,437 distinct medical examination records from a total of 2,479 participants. Participants were categorized based on outcomes from their third examination into two sets: dyslipidemia, comprising 310 individuals or 12.5% of the study population, and non-dyslipidemia, numbering 2,169 or 87.5% of the total. The characteristics of the dyslipidemia and non-dyslipidemia sets were shown in <xref ref-type="table" rid="T2">Table 2</xref>. As shown, there were differences in the baseline data between dyslipidemia and non-dyslipidemia in some characteristics, indicating that the development of dyslipidemia was traceable. In addition to the physical examination data listed in <xref ref-type="table" rid="T2">Table 2</xref>, there were also blood cell analysis, urinalysis, blood glucose test, liver function, kidney function, and thyroid function. The baseline data of these characteristics were in <xref ref-type="sec" rid="s12">Supplementary Table S1</xref>.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Baseline characteristics of dyslipidemia and non-dyslipidemia participants.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Characteristics</th>
<th align="center">Dyslipidemia (n &#x3d; 310)</th>
<th align="center">Non-dyslipidemia (n &#x3d; 2,169)</th>
<th align="center">
<italic>P</italic>-value</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Age</td>
<td align="center">32.00 (28.00, 37.00)</td>
<td align="center">33.00 (29.00, 38.00)</td>
<td align="center">0.106</td>
</tr>
<tr>
<td align="center">Sex</td>
<td align="left"/>
<td align="left"/>
<td align="center">&#x3c;0.001</td>
</tr>
<tr>
<td align="center">Male</td>
<td align="center">159 (51.29%)</td>
<td align="center">856 (39.47%)</td>
<td align="left"/>
</tr>
<tr>
<td align="center">Female</td>
<td align="center">151 (48.71%)</td>
<td align="center">1,313 (60.53%)</td>
<td align="left"/>
</tr>
<tr>
<td align="center">Height</td>
<td align="center">164.00 (159.00, 170.50)</td>
<td align="center">166.00 (159.00, 172.00)</td>
<td align="center">0.021</td>
</tr>
<tr>
<td align="center">Weight</td>
<td align="center">57.40 (51.70, 65.80)</td>
<td align="center">60.90 (53.30, 70.30)</td>
<td align="center">&#x3c;0.001</td>
</tr>
<tr>
<td align="center">SBP</td>
<td align="center">113.00 (105.00, 122.00)</td>
<td align="center">116.00 (108.00, 127.00)</td>
<td align="center">&#x3c;0.001</td>
</tr>
<tr>
<td align="center">DBP</td>
<td align="center">68.00 (63.00, 75.00)</td>
<td align="center">70.00 (65.00, 77.00)</td>
<td align="center">&#x3c;0.001</td>
</tr>
<tr>
<td align="center">Pulse</td>
<td align="center">79.00 (72.00, 88.00)</td>
<td align="center">79.00 (72.00, 88.00)</td>
<td align="center">0.754</td>
</tr>
<tr>
<td align="center">BMI</td>
<td align="center">21.40 (19.70, 23.40)</td>
<td align="center">21.97 (20.20, 24.50)</td>
<td align="center">&#x3c;0.001</td>
</tr>
<tr>
<td align="center">TC</td>
<td align="center">4.03 (3.70, 4.34)</td>
<td align="center">4.45 (4.17, 4.73)</td>
<td align="center">&#x3c;0.001</td>
</tr>
<tr>
<td align="center">TG</td>
<td align="center">0.79 (0.63, 1.01)</td>
<td align="center">0.99 (0.75, 1.23)</td>
<td align="center">&#x3c;0.001</td>
</tr>
<tr>
<td align="center">HDL-C</td>
<td align="center">1.54 (1.34, 1.75)</td>
<td align="center">1.45 (1.22, 1.78)</td>
<td align="center">0.002</td>
</tr>
<tr>
<td align="center">LDL-C</td>
<td align="center">2.50 (2.17, 2.84)</td>
<td align="center">2.92 (2.63, 3.17)</td>
<td align="center">&#x3c;0.001</td>
</tr>
<tr>
<td align="center">HbA1c</td>
<td align="center">5.30 (5.10, 5.40)</td>
<td align="center">5.30 (5.10, 5.50)</td>
<td align="center">0.314</td>
</tr>
<tr>
<td align="center">CEA</td>
<td align="center">1.39 (0.95, 1.99)</td>
<td align="center">1.56 (1.16, 2.19)</td>
<td align="center">0.001</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>
<italic>P</italic> &#x3c; 0.050 is considered statistical significance. SBP, systolic blood pressure; DBP, diastolic blood pressure; TC, total cholesterol; TG, triglycerides; HDL-C, high-density lipoprotein cholesterol; LDL-C, low-density lipoprotein cholesterol; HbA1c, glycated hemoglobin; CEA, carcinoembryonic antigen. Categorical variables, expressed as frequencies (proportions), line &#x3c7;2 test. Non-normally distributed variables, expressed as median (interquartile range), line Mann&#x2013;Whitney U test.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3-2">
<title>3.2 Feature selection</title>
<p>In the first stage, the construction of models involved the application of various base learners alongside different numbers of feature. The results were shown in <xref ref-type="fig" rid="F2">Figure 2</xref> and the quantitative description of the results was in <xref ref-type="sec" rid="s12">Supplementary Table S2</xref>. In all base learners, the prediction performance improves with the increase in features and then reaches a plateau. When the number of features was 12, the models generally achieved the best performance. Notably, disparities in performance were observed among the base learners, even with identical feature sets. In particular, the XGBoost model (AUC &#x3d; 0.84, number of features was 12) demonstrated superior predictive accuracy compared to the SVM model (AUC &#x3d; 0.68, number of features was 12), which lagged in performance. This result showed that selecting appropriate models and features can effectively improve the accuracy of dyslipidemia prediction.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Predictive performance of different base learners and different number of features.</p>
</caption>
<graphic xlink:href="fphys-15-1464744-g002.tif"/>
</fig>
</sec>
<sec id="s3-3">
<title>3.3 Feature importance</title>
<p>Following the performance evaluation of base learners, we also conducted a deep investigation on feature utilization, specifically focusing on scenarios where the feature number was set to 12. We tallied the frequency with which each feature was selected across the five-fold cross-validation process. This examination&#x2019;s findings were illustrated in <xref ref-type="fig" rid="F3">Figure 3</xref>. The analysis unveiled a notable consistency in feature usage across the various folds: five features (TC and LDL-C at the first examination, TG, TC and LDL-C at the second examination) were employed in all five folds of validation, while three features (Glycated hemoglobin (HbA1c) at the second examination, carcinoembryonic antigen (CEA) at the first examination, and the difference between the two CEA examinations) were utilized in four out of five validations. Additionally, we analyzed the top 12 features in each fold during the five-fold cross-validation. <xref ref-type="sec" rid="s12">Supplementary Figure S3</xref> presents the mutual information scores for these 12 features during feature selection. The features with higher mutual information scores are also those mentioned above. This result showed that the model&#x2019;s robust consistency and stability throughout different segments of validation. Among the frequently utilized indicators, TC, LDL-C, TG, CEA, and HbA1c were distinguished as key features, reflecting their significant role in the model&#x2019;s predictive capability.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>The frequency of features using in the five-fold cross-validation. Among them, the suffix f means the first examination, s means the second examination, and d means the difference between the two examinations.</p>
</caption>
<graphic xlink:href="fphys-15-1464744-g003.tif"/>
</fig>
</sec>
<sec id="s3-4">
<title>3.4 Development and validation of prediction models</title>
<p>Employing the predicted outcomes from the base learners alongside the key features, the inputs for the ensemble model were synthesized, culminating in the final predictive analysis conducted using the XGBoost algorithm. Using ROC curve analysis, we calculated the corresponding AUCs for the different base learners and ensemble model when the number of features was 12 in five-fold cross validation. As can be seen in <xref ref-type="fig" rid="F4">Figure 4</xref>, the AUC scores for the base learners fluctuate between 0.68 &#xb1; 0.05 and 0.84 &#xb1; 0.02, whereas the ensemble model achieved an AUC of 0.88 &#xb1; 0.01 (<italic>P</italic> &#x3c; 0.001), markedly surpassing those of the individual base learners. <xref ref-type="table" rid="T3">Table 3</xref> showed a more detailed average performance comparison. The ensemble model exhibited pronounced superiority in several performance metrics, with accuracy of 0.78 &#xb1; 0.01 and specificity of 0.78 &#xb1; 0.02, both of which were better than other base learners. Additionally, we conducted experiments by adjusting the ratio of dyslipidemia and non-dyslipidemia samples under the same hyperparameters, and the results showed that the model maintained good predictive performance across different sample ratios (<xref ref-type="sec" rid="s12">Supplementary Table S3</xref>). These insights not only highlighted the capability of machine learning techniques in dyslipidemia predictions but also illustrated the profound impact of ensemble learning approach on improving predictive accuracy.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>The AUC performance and average AUC performance of base learners and ensemble model in five-fold cross validation.</p>
</caption>
<graphic xlink:href="fphys-15-1464744-g004.tif"/>
</fig>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Average prediction performance of different machine learning models in five-fold cross validation.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Models</th>
<th align="center">Accuracy (mean &#xb1; SD)</th>
<th align="center">AUC (mean &#xb1; SD)</th>
<th align="center">Sensitivity (mean &#xb1; SD)</th>
<th align="center">Specificity (mean &#xb1; SD)</th>
<th align="center">
<italic>P</italic>-value</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">LR</td>
<td align="center">0.71 &#xb1; 0.01</td>
<td align="center">0.80 &#xb1; 0.02</td>
<td align="center">0.77 &#xb1; 0.07</td>
<td align="center">0.70 &#xb1; 0.01</td>
<td align="center">&#x3c;0.001</td>
</tr>
<tr>
<td align="center">RF</td>
<td align="center">0.72 &#xb1; 0.02</td>
<td align="center">0.83 &#xb1; 0.04</td>
<td align="center">0.77 &#xb1; 0.05</td>
<td align="center">0.71 &#xb1; 0.02</td>
<td align="center">&#x3c;0.001</td>
</tr>
<tr>
<td align="center">KNN</td>
<td align="center">0.71 &#xb1; 0.01</td>
<td align="center">0.79 &#xb1; 0.02</td>
<td align="center">0.74 &#xb1; 0.08</td>
<td align="center">0.70 &#xb1; 0.02</td>
<td align="center">&#x3c;0.001</td>
</tr>
<tr>
<td align="center">SVM</td>
<td align="center">0.61 &#xb1; 0.10</td>
<td align="center">0.68 &#xb1; 0.05</td>
<td align="center">0.64 &#xb1; 0.12</td>
<td align="center">0.61 &#xb1; 0.13</td>
<td align="center">&#x3c;0.001</td>
</tr>
<tr>
<td align="center">XGBoost</td>
<td align="center">0.70 &#xb1; 0.03</td>
<td align="center">0.84 &#xb1; 0.02</td>
<td align="center">
<bold>0.84 &#xb1; 0.07</bold>
</td>
<td align="center">0.68 &#xb1; 0.03</td>
<td align="center">&#x3c;0.001</td>
</tr>
<tr>
<td align="center">Ensemble Model</td>
<td align="center">
<bold>0.78 &#xb1; 0.01</bold>
</td>
<td align="center">
<bold>0.88 &#xb1; 0.01</bold>
</td>
<td align="center">0.80 &#xb1; 0.06</td>
<td align="center">
<bold>0.78 &#xb1; 0.02</bold>
</td>
<td align="center">&#x3c;0.001</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Values in bold indicate best performance</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3-5">
<title>3.5 Clinical usage of the models</title>
<p>To visually demonstrate the clinical usability of the ensemble model, we plotted calibration curves and conducted decision curve analysis (DCA). The calibration curve showed that the actual observations were well consistent with the predictions of the ensemble model (<xref ref-type="fig" rid="F5">Figure 5A</xref>), suggesting that the ensemble model has an excellent predictive value. The DCA curve of the ensemble model also demonstrated good clinical utility, showing preferable positive net benefit (<xref ref-type="fig" rid="F5">Figure 5B</xref>). In addition, similar results were shown in each fold of the five-fold cross validation (<xref ref-type="sec" rid="s12">Supplementary Figures S1, S2</xref>).</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>The calibration curves and decision curve analysis curves of the ensemble model.</p>
</caption>
<graphic xlink:href="fphys-15-1464744-g005.tif"/>
</fig>
</sec>
<sec id="s3-6">
<title>3.6 Model explainability</title>
<p>We visualized the influence of predictor variables on the results based on SHAP plots. <xref ref-type="fig" rid="F6">Figure 6</xref> shows the SHAP summary plot of the second stage model input features in five-fold cross-validation. Specifically, the influence of variables on the results can be intuitively explained by the magnitude of the SHAP value (indicated by color change) and the trend on the horizontal axis of the variable (the probability of an adverse outcome). For example, in the scenario of HbA1c_s, individuals with higher indicators (indicated in red) were more likely to have dyslipidemia (on the right) compared to those with lower HbA1c_s indicators (indicated in blue). Overall, it is evident that the important predictors of these five models have strong consistency, among which XGBoost_prob, LR_prob, HbA1c_s, CEA_d, LDL-C_s, KNN_prob, and TG_s were extracted as important predictors.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>SHapley Additive exPlanations summary plot of the input features in second stage model. XGBoost_prob, LR_prob, KNN_prob, SVM_prob, and RF_prob are the probabilities corresponding to the first stage XGBoost, LR, KNN, SVM, and RF models; HbA1c, glycated hemoglobin; CEA, carcinoembryonic antigen; LDL-C, low-density-lipoprotein-cholesterol; TG, triglycerides; TC, total cholesterol. The suffix f means the first examination, s means the second examination, and d means the difference between the two examinations.</p>
</caption>
<graphic xlink:href="fphys-15-1464744-g006.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<p>Dyslipidemia has become a common disease among patients, posing a significant risk for the development and progression of cardiovascular disease and is one of the most important risk factors for atherosclerotic cardiovascular disease, which accounts for the most deaths worldwide (<xref ref-type="bibr" rid="B32">Sandesara et al., 2019</xref>). Therefore, early risk prediction is particularly important for the prevention and management of dyslipidemia. In this research, we developed an ensemble model tailored to predict dyslipidemia risk in the third year based on data from the initial 2 years&#x2019; physical examinations. The efficacy of this model was corroborated on a dataset encompassing 2,479 participants, where it attained an AUC value of 0.88 &#xb1; 0.01 (<italic>P</italic> &#x3c; 0.001), indicating a high capacity for dyslipidemia prediction. The improved performances of the ensemble model over the base learners were consistent with our assumption that ensemble model performs better than individual machine learning models. Different models are suited to handling different types of data patterns. For instance, LR is well-suited for linear relationships, RF and XGBoost excel at handling nonlinear data, SVM performs well with high-dimensional data, and KNN are effective at capturing local patterns. By combining these algorithms (LR, SVM, RF, KNN, and XGBoost) into an ensemble model, we can leverage the strengths of each algorithm and compensate for their individual weaknesses, leading to significantly improved predictive performance.</p>
<p>Furthermore, the investigation delved into identifying the optimal minimal set of features necessary for accurate predictions. Through rigorous application of statistical analyses and mutual information for feature selection, the study identified that a subset of the top 12 features suffices to achieve reliable predictive outcomes. In the statistical examination of the 12 features utilized in the modeling process by base learners, it was observed that five features were consistently selected across the five-fold cross-validation, whereas an additional three features were chosen in four out of five folds. These eight key features encompass TC, LDL-C, and CEA from the first physical examination; TC, TG, LDL-C, and HbA1c from the second examination; along with the difference in CEA levels between the two examinations.</p>
<p>Notably, TC, TG, and LDL-C were acknowledged as fundamental metrics for assessing blood lipid status, with HbA1c also recognized for its association with lipid concentrations (<xref ref-type="bibr" rid="B18">Li et al., 2022</xref>; <xref ref-type="bibr" rid="B2">Bulut et al., 2017</xref>). Previous study (<xref ref-type="bibr" rid="B7">Feng et al., 2019</xref>) have shown that high LDL-C is the most common component of dyslipidemia, followed by elevated TG. HbA1c was a recognized indicator related to dyslipidemia, and it was significantly correlated with common lipid parameters such as TC, TG, and LDL-C (<xref ref-type="bibr" rid="B26">Ozder, 2014</xref>; <xref ref-type="bibr" rid="B30">Reddy et al., 2014</xref>). Previous study (<xref ref-type="bibr" rid="B10">Huang et al., 2021</xref>) have shown that lowering HbA1c levels may improve blood lipid levels. At the same time, HbA1c was closely related to diabetes, and abnormal lipid metabolism was part of the pathogenesis of diabetes (<xref ref-type="bibr" rid="B35">Sunjaya and Sunjaya, 2018</xref>). Metabolic syndrome was a combination of metabolic abnormalities such as hypertension, obesity, hyperglycemia, and dyslipidemia, which increases the risk of cancer (<xref ref-type="bibr" rid="B22">Mendrick et al., 2018</xref>). CEA was widely considered to be a serological tumor marker, and CEA levels can affect a variety of metabolic diseases (<xref ref-type="bibr" rid="B20">Lu et al., 2018</xref>; <xref ref-type="bibr" rid="B38">Wang C.-H. et al., 2022</xref>). Therefore, CEA levels may have a certain relationship with dyslipidemia, which was consistent with the results of the model. The inclusion of these indicators as key features demonstrates the model&#x2019;s strong clinical relevance and interpretability.</p>
<p>Moreover, the importance of predictors in the ensemble model evaluated using SHAP values was consistent across five-fold cross validation. Among them, the prediction probabilities of the XGBoost, LR, and KNN models in the first stage were important predictors of ensemble models, which proved that the ensemble model can well integrate the advantages of each base learner and achieve better prediction performance. In addition, an interesting phenomenon was observed that HbA1c and CEA were more important than TC, TG, and LDL-C, which were conventionally considered predictors of dyslipidemia. This may be because the ensemble model does not obtain results based on a simple linear relationship, but explores a more complex relationship between predictors and results.</p>
<p>Current research into dyslipidemia predominantly centers on elucidating its risk factors. For example, Qi et al. (<xref ref-type="bibr" rid="B28">Qi et al., 2015</xref>) and Ni et al. (<xref ref-type="bibr" rid="B24">Ni et al., 2015</xref>) undertook analyses using different datasets and statistical methodologies to investigate dyslipidemia prevalence and the differential indicators between affected individuals and the general populace, with the objective of identifying dyslipidemia risk factors. However, such cross-sectional investigations are largely constrained to singular temporal analyses, neglecting the longitudinal progression of dyslipidemia, which curtails their predictive utility. Conversely, our research examines the dynamic evolutions of physiological indicators over time. Through a longitudinal analysis of indicator fluctuations within the same cohort across multiple intervals, we discern patterns indicative of alterations in lipid concentrations, thereby facilitating effective dyslipidemia prediction. While several studies have employed cohort data for dyslipidemia predicting (<xref ref-type="bibr" rid="B33">Sasagawa et al., 2024</xref>), the majority are limited by their reliance on singular predictive model, overlooking the varied data mining emphases inherent to different algorithms. Their performance, as measured by the AUC, usually hovers around 0.83. Our methodology diverges by adopting a multifaceted perspective, substantially augmenting predictive efficacy through the exploitation of diverse model strengths and their integration. This strategy not only elevates the accuracy of our predictive model but also its applicability in real-world settings, furnishing a robust scientific foundation for dyslipidemia&#x2019;s early prevention and management.</p>
<p>The primary application of this model is in health examination centers, where it can be used to predict the risk of dyslipidemia in the following year based on consecutive years of health examination data. Examination centers need to maintain continuous health records for each patient, and by analyzing both historical and current examination data, the model can provide predictions on the likelihood of developing dyslipidemia in the future. This model not only provides early warnings of dyslipidemia for patients, but also reinforces the value of regular health check-ups, thereby encouraging patients to adhere to scheduled examinations. For health examination centers, the model offers more personalized services, enhancing customer satisfaction. Moreover, this modeling approach can be extended to risk prediction for other diseases, showcasing its broad clinical application potential.</p>
<p>The strength of this study lies in the integration of multiple machine learning algorithms to construct an ensemble model for dyslipidemia prediction. In comparison to base learners, including LR, SVM, RF, KNN, and XGBoost, our ensemble model has shown an improvement in the AUC indicator, with the AUC improved by 0.04&#x2013;0.20. In addition, we also conducted a detailed analysis of the features selected by the model to ensure transparency and facilitate the interpretation of the results. This study provides a health management tool that can help identify individuals at risk of dyslipidemia early, potentially reducing its prevalence. However, this study has several limitations. Firstly, the data samples were exclusively sourced from Shenzhen City, Guangdong Province, China, which may impart a regional bias to the findings. It is worth noting that Shenzhen is a city with a large migrant population, resulting in a relatively diverse demographic composition. Therefore, the impact of regional and demographic characteristics on the results may not be as significant as in other areas. Secondly, the median age of the participants is 32 years, predominantly under 50, leading to an underrepresentation of the elderly demographic in the analysis. These limitations underscore the necessity for subsequent studies to encompass a more diverse and representative population sample and to explore alternative methods of feature construction. Such expansions are crucial for augmenting the model&#x2019;s generalizability and enhancing its predictive precision.</p>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>In conclusion, we presented an ensemble learning approach to predict dyslipidemia risk in the third year based on physical examination data from two successive years. The empirical findings substantiate the effectiveness of our proposed methodology in accurately predicting dyslipidemia, with the model also exhibiting notable clinical interpretability. This study also found that HbA1c and CEA could be used as key indicators for assessing blood lipid status. Future directions include refining the model through the inclusion of a more extensive population sample and investigating the potential for more efficient exploitation of existing features or the innovation of new feature engineering strategies to elevate the predictive accuracy for dyslipidemia.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec id="s7">
<title>Ethics statement</title>
<p>The studies involving humans were approved by The Ethics Committee of Shenzhen University, Shenzhen. The studies were conducted in accordance with the local legislation and institutional requirements. Written informed consent for participation was not required from the participants or the participants&#x2019;; legal guardians/next of kin in accordance with the national legislation and institutional requirements.</p>
</sec>
<sec id="s8">
<title>Author contributions</title>
<p>NZ: Investigation, Methodology, Writing&#x2013;original draft, Writing&#x2013;review and editing. XG: Visualization, Writing&#x2013;original draft. XY: Writing&#x2013;review and editing. ZT: Data curation, Funding acquisition, Writing&#x2013;review and editing. FC: Data curation, Funding acquisition, Writing&#x2013;review and editing. PD: Data curation, Writing&#x2013;review and editing. JG: Data curation, Funding acquisition, Supervision, Writing&#x2013;review and editing. GD: Project administration, Supervision, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This work was supported by the Science, Technology and Innovation Commission of Shenzhen Municipality (20231121163750002) and the Shenzhen Nanshan District Science and Technology Plan (NS2022145, NS2023128). Medicine Plus Program of Shenzhen University (No.2024YG011), Shenzhen health elite talents (No.2021XKQ193), Education Reform Project of Guangdong Province (No.2021JD082).</p>
</sec>
<ack>
<p>The authors would like to thank the reviewers for their valuable suggestions.</p>
</ack>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fphys.2024.1464744/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fphys.2024.1464744/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table1.DOCX" id="SM1" mimetype="application/DOCX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Breiman</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Random forests</article-title>. <source>Mach. Learn.</source> <volume>45</volume>, <fpage>5</fpage>&#x2013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1023/a:1010933404324</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bulut</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Demirel</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Metin</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>The prevalence of dyslipidemia and associated factors in children and adolescents with type 1 diabetes</article-title>. <source>J. Pediatr. Endocrinol. Metabolism</source> <volume>30</volume>, <fpage>181</fpage>&#x2013;<lpage>187</lpage>. <pub-id pub-id-type="doi">10.1515/jpem-2016-0111</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Guestrin</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Xgboost: a scalable tree boosting system</article-title>,&#x201d; in <source>Proceedings of the 22nd acm sigkdd international conference on knowledge discovery and data mining</source>, <fpage>785</fpage>&#x2013;<lpage>794</lpage>.</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cox</surname>
<given-names>D. R.</given-names>
</name>
</person-group> (<year>1958</year>). <article-title>The regression analysis of binary sequences</article-title>. <source>J. R. Stat. Soc. Ser. B Stat. Methodol.</source> <volume>20</volume>, <fpage>215</fpage>&#x2013;<lpage>232</lpage>. <pub-id pub-id-type="doi">10.1111/j.2517-6161.1958.tb00292.x</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>De Silva</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>W. K.</given-names>
</name>
<name>
<surname>Forbes</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Demmer</surname>
<given-names>R. T.</given-names>
</name>
<name>
<surname>Barton</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Enticott</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Use and performance of machine learning models for type 2 diabetes prediction in community settings: a systematic review and meta-analysis</article-title>. <source>Int. J. Med. Inf.</source> <volume>143</volume>, <fpage>104268</fpage>. <pub-id pub-id-type="doi">10.1016/j.ijmedinf.2020.104268</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Doi</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Langsted</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Nordestgaard</surname>
<given-names>B. G.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Elevated remnant cholesterol reclassifies risk of ischemic heart disease and myocardial infarction</article-title>. <source>J. Am. Coll. Cardiol.</source> <volume>79</volume>, <fpage>2383</fpage>&#x2013;<lpage>2397</lpage>. <pub-id pub-id-type="doi">10.1016/j.jacc.2022.03.384</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Feng</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Ying</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Exploration of dyslipidemia prevalence and its risk factors in a coastal city of China: a population-based cross-sectional study</article-title>. <source>Int. J. Clin. Exp. Med.</source> <volume>12</volume>, <fpage>2729</fpage>&#x2013;<lpage>2737</lpage>.</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fix</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Hodges</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>1989</year>). <article-title>Discriminatory analysis. Nonparametric discrimination: consistency properties</article-title>. <source>Int. Stat. Review/Revue Int. Stat.</source> <volume>57</volume>, <fpage>238</fpage>&#x2013;<lpage>247</lpage>. <pub-id pub-id-type="doi">10.2307/1403797</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hedayatnia</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Asadi</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zare-Feyzabadi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Yaghooti-Khorasani</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ghazizadeh</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ghaffarian-Zirak</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Dyslipidemia and cardiovascular disease risk among the MASHAD study population</article-title>. <source>Lipids health Dis.</source> <volume>19</volume>, <fpage>42</fpage>&#x2013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1186/s12944-020-01204-y</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Lei</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>The relationship between high-density lipoprotein cholesterol (HDL-C) and glycosylated hemoglobin in diabetic patients aged 20 or above: a cross-sectional study</article-title>. <source>BMC Endocr. Disord.</source> <volume>21</volume>, <fpage>198</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1186/s12902-021-00863-x</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ibrahim</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Abdulazeez</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>The role of machine learning algorithms for diagnosing diseases</article-title>. <source>J. Appl. Sci. Technol. Trends</source> <volume>2</volume>, <fpage>10</fpage>&#x2013;<lpage>19</lpage>. <pub-id pub-id-type="doi">10.38094/jastt20179</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jian-Jun</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Shui-Ping</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Guo-Ping</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Dao-Quan</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Jing</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>2023 China guidelines for lipid management</article-title>. <source>J. Geriatric Cardiol. JGC</source> <volume>20</volume>, <fpage>621</fpage>&#x2013;<lpage>663</lpage>. <pub-id pub-id-type="doi">10.26599/1671-5411.2023.09.008</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kavey</surname>
<given-names>R.-E. W.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Combined dyslipidemia in children and adolescents: a proposed new management approach</article-title>. <source>Curr. Atheroscler. Rep.</source> <volume>25</volume>, <fpage>237</fpage>&#x2013;<lpage>245</lpage>. <pub-id pub-id-type="doi">10.1007/s11883-023-01099-x</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Lim</surname>
<given-names>D. H.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Classification and prediction on the effects of nutritional intake on overweight/obesity, dyslipidemia, hypertension and type 2 diabetes mellitus using deep learning model: 4&#x2013;7th Korea national health and nutrition examination survey</article-title>. <source>Int. J. Environ. Res. Public Health</source> <volume>18</volume>, <fpage>5597</fpage>. <pub-id pub-id-type="doi">10.3390/ijerph18115597</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Klop</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Elte</surname>
<given-names>J. W. F.</given-names>
</name>
<name>
<surname>Castro Cabezas</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Dyslipidemia in obesity: mechanisms and potential targets</article-title>. <source>Nutrients</source> <volume>5</volume>, <fpage>1218</fpage>&#x2013;<lpage>1240</lpage>. <pub-id pub-id-type="doi">10.3390/nu5041218</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lai</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>IL-38 in modulating hyperlipidemia and its related cardiovascular diseases</article-title>. <source>Int. Immunopharmacol.</source> <volume>108</volume>, <fpage>108876</fpage>. <pub-id pub-id-type="doi">10.1016/j.intimp.2022.108876</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Xi</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Development and validation of a simple-to-use nomogram for self-screening the risk of dyslipidemia</article-title>. <source>Sci. Rep.</source> <volume>13</volume>, <fpage>9169</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-023-36281-3</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Nie</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ge</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Prevalence of dyslipidemia, treatment rate and its control among patients with type 2 diabetes mellitus in Northwest China: a cross-sectional study</article-title>. <source>Lipids Health Dis.</source> <volume>21</volume>, <fpage>77</fpage>. <pub-id pub-id-type="doi">10.1186/s12944-022-01691-1</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>CT-based radiomics signature of visceral adipose tissue for prediction of disease progression in patients with crohn&#x27;s disease: a multicentre cohort study</article-title>. <source>EClinicalMedicine</source> <volume>56</volume>, <fpage>101805</fpage>. <pub-id pub-id-type="doi">10.1016/j.eclinm.2022.101805</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>HbA1c is positively associated with serum carcinoembryonic antigen (CEA) in patients with diabetes: a cross-sectional study</article-title>. <source>Diabetes Ther.</source> <volume>9</volume>, <fpage>209</fpage>&#x2013;<lpage>217</lpage>. <pub-id pub-id-type="doi">10.1007/s13300-017-0356-2</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Yin</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X. S.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Ensemble methods of rank-based trees for single sample classification with gene expression profiles</article-title>. <source>J. Transl. Med.</source> <volume>22</volume>, <fpage>140</fpage>. <pub-id pub-id-type="doi">10.1186/s12967-024-04940-2</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mendrick</surname>
<given-names>D. L.</given-names>
</name>
<name>
<surname>Diehl</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Topor</surname>
<given-names>L. S.</given-names>
</name>
<name>
<surname>Dietert</surname>
<given-names>R. R.</given-names>
</name>
<name>
<surname>Will</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>La Merrill</surname>
<given-names>M. A.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Metabolic syndrome and associated diseases: from the bench to the clinic</article-title>. <source>Toxicol. Sci.</source> <volume>162</volume>, <fpage>36</fpage>&#x2013;<lpage>42</lpage>. <pub-id pub-id-type="doi">10.1093/toxsci/kfx233</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moran</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Coxson</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y. C.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>C.-S.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>Future cardiovascular disease in China: markov model and risk factor scenario projections from the coronary heart disease policy model&#x2013;China</article-title>. <source>Circulation Cardiovasc. Qual. Outcomes</source> <volume>3</volume>, <fpage>243</fpage>&#x2013;<lpage>252</lpage>. <pub-id pub-id-type="doi">10.1161/CIRCOUTCOMES.109.910711</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ni</surname>
<given-names>W.-Q.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.-L.</given-names>
</name>
<name>
<surname>Zhuo</surname>
<given-names>Z.-P.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>X.-L.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>J.-P.</given-names>
</name>
<name>
<surname>Chi</surname>
<given-names>H.-S.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Serum lipids and associated factors of dyslipidemia in the adult population in Shenzhen</article-title>. <source>Lipids health Dis.</source> <volume>14</volume>, <fpage>71</fpage>&#x2013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1186/s12944-015-0073-7</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Organization</surname>
<given-names>W. H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Raised cholesterol</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://www.who.int/data/gho/indicator-metadata-registry/imr-details/3236">https://www.who.int/data/gho/indicator-metadata-registry/imr-details/3236</ext-link> (Accessed December 29, 2021)</comment>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ozder</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Lipid profile abnormalities seen in T2DM patients in primary healthcare in Turkey: a cross-sectional study</article-title>. <source>Lipids health Dis.</source> <volume>13</volume>, <fpage>183</fpage>&#x2013;<lpage>186</lpage>. <pub-id pub-id-type="doi">10.1186/1476-511X-13-183</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pirillo</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Casula</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Olmastroni</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Norata</surname>
<given-names>G. D.</given-names>
</name>
<name>
<surname>Catapano</surname>
<given-names>A. L.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Global epidemiology of dyslipidaemias</article-title>. <source>Nat. Rev. Cardiol.</source> <volume>18</volume>, <fpage>689</fpage>&#x2013;<lpage>700</lpage>. <pub-id pub-id-type="doi">10.1038/s41569-021-00541-4</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Prevalence and risk factors associated with dyslipidemia in Chongqing, China</article-title>. <source>Int. J. Environ. Res. public health</source> <volume>12</volume>, <fpage>13455</fpage>&#x2013;<lpage>13465</lpage>. <pub-id pub-id-type="doi">10.3390/ijerph121013455</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Raja</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Aguiar</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Alsayed</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Chibber</surname>
<given-names>Y. S.</given-names>
</name>
<name>
<surname>Elbadawi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ezhov</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Non-HDL-cholesterol in dyslipidemia: review of the state-of-the-art literature and outlook</article-title>. <source>Atherosclerosis</source> <volume>383</volume>, <fpage>117312</fpage>. <pub-id pub-id-type="doi">10.1016/j.atherosclerosis.2023.117312</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reddy</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Meera</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>William</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Correlation between glycemic control and lipid profile in type 2 diabetic patients: HbA1c as an indirect indicator of dyslipidemia</article-title>. <source>Asian J. Pharm. Clin. Res.</source>, <fpage>153</fpage>&#x2013;<lpage>155</lpage>.</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ruan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ran</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.-S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Dyslipidemia versus obesity as predictors of ischemic stroke prognosis: a multi-center study in China</article-title>. <source>Lipids Health Dis.</source> <volume>23</volume>, <fpage>72</fpage>. <pub-id pub-id-type="doi">10.1186/s12944-024-02061-9</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sandesara</surname>
<given-names>P. B.</given-names>
</name>
<name>
<surname>Virani</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Fazio</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shapiro</surname>
<given-names>M. D.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>The forgotten lipids: triglycerides, remnant cholesterol, and atherosclerotic cardiovascular disease risk</article-title>. <source>Endocr. Rev.</source> <volume>40</volume>, <fpage>537</fpage>&#x2013;<lpage>557</lpage>. <pub-id pub-id-type="doi">10.1210/er.2018-00184</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sasagawa</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Inoue</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Futagami</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Nakamura</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Maeda</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Aoki</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Application of deep neural survival networks to the development of risk prediction models for diabetes mellitus, hypertension, and dyslipidemia</article-title>. <source>J. Hypertens.</source> <volume>42</volume>, <fpage>506</fpage>&#x2013;<lpage>514</lpage>. <pub-id pub-id-type="doi">10.1097/HJH.0000000000003626</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Nong</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Architecting the metabolic reprogramming survival risk framework in LUAD through single-cell landscape analysis: three-stage ensemble learning with genetic algorithm optimization</article-title>. <source>J. Transl. Med.</source> <volume>22</volume>, <fpage>353</fpage>. <pub-id pub-id-type="doi">10.1186/s12967-024-05138-2</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sunjaya</surname>
<given-names>A. P.</given-names>
</name>
<name>
<surname>Sunjaya</surname>
<given-names>A. F.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Glycated hemoglobin targets and glycemic control: link with lipid, uric acid and kidney profile</article-title>. <source>Diabetes and Metabolic Syndrome Clin. Res. and Rev.</source> <volume>12</volume>, <fpage>743</fpage>&#x2013;<lpage>748</lpage>. <pub-id pub-id-type="doi">10.1016/j.dsx.2018.04.039</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Vapnik</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>1999</year>). <source>The nature of statistical learning theory</source>. <publisher-name>Springer science and business media</publisher-name>.</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vekic</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zeljkovic</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Stefanovic</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Jelic-Ivanovic</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Spasojevic-Kalimanovska</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Obesity and dyslipidemia</article-title>. <source>Metabolism</source> <volume>92</volume>, <fpage>71</fpage>&#x2013;<lpage>81</lpage>. <pub-id pub-id-type="doi">10.1016/j.metabol.2018.11.005</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>C.-H.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhuang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>L.-H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.-H.</given-names>
</name>
<etal/>
</person-group> (<year>2022a</year>). <article-title>High-normal serum carcinoembryonic antigen levels and increased risk of diabetic peripheral neuropathy in type 2 diabetes</article-title>. <source>Diabetology and Metabolic Syndrome</source> <volume>14</volume>, <fpage>142</fpage>. <pub-id pub-id-type="doi">10.1186/s13098-022-00909-7</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>J.-S.</given-names>
</name>
<name>
<surname>Chiang</surname>
<given-names>H.-Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.-C.</given-names>
</name>
<name>
<surname>Yeh</surname>
<given-names>H.-C.</given-names>
</name>
<name>
<surname>Ting</surname>
<given-names>I.-W.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>C.-C.</given-names>
</name>
<etal/>
</person-group> (<year>2022b</year>). <article-title>Dyslipidemia and coronary artery calcium: from association to development of a risk-prediction nomogram</article-title>. <source>Nutr. Metabolism Cardiovasc. Dis.</source> <volume>32</volume>, <fpage>1944</fpage>&#x2013;<lpage>1954</lpage>. <pub-id pub-id-type="doi">10.1016/j.numecd.2022.05.006</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>You</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Prediction of acute kidney injury after cardiac surgery: model development using a Chinese electronic health record dataset</article-title>. <source>J. Transl. Med.</source> <volume>20</volume>, <fpage>166</fpage>. <pub-id pub-id-type="doi">10.1186/s12967-022-03351-5</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Ji</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Risk prediction of dyslipidemia for Chinese Han adults using random Forest survival model</article-title>. <source>Clin. Epidemiol.</source> <volume>11</volume>, <fpage>1047</fpage>&#x2013;<lpage>1055</lpage>. <pub-id pub-id-type="doi">10.2147/CLEP.S223694</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>The TBK1/IKK&#x3b5; inhibitor amlexanox improves dyslipidemia and prevents atherosclerosis</article-title>. <source>JCI insight</source> <volume>7</volume>, <fpage>e155552</fpage>. <pub-id pub-id-type="doi">10.1172/jci.insight.155552</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>