<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="1.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title-group>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1752811</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2025.1752811</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Original Research</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Coronary heart disease risk prediction based on GAIN imputation and interpretable machine learning</article-title>
<alt-title alt-title-type="left-running-head">Zhao et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2025.1752811">10.3389/fgene.2025.1752811</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Zhao</surname>
<given-names>Shulin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3286582"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing&#x2013;original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Nan</surname>
<given-names>Baoyun</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x26; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/">Writing&#x2013;review and editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Guo</surname>
<given-names>Jun</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing&#x2013;original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xu</surname>
<given-names>Wenkai</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x26; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/">Writing&#x2013;review and editing</role>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Li</surname>
<given-names>Zhen</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3240965"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing&#x2013;original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
</contrib>
</contrib-group>
<aff id="aff1">
<label>1</label>
<institution>The Quzhou Affiliated Hospital of Wenzhou Medical University, Quzhou People&#x2019;s Hospital</institution>, <city>Quzhou</city>, <country country="CN">China</country>
</aff>
<aff id="aff2">
<label>2</label>
<institution>School of Artificial Intelligence, Shenzhen University of Information Technology</institution>, <city>Shenzhen</city>, <country country="CN">China</country>
</aff>
<author-notes>
<corresp id="c001">
<label>&#x2a;</label>Correspondence: Zhen Li, <email xlink:href="mailto:lizhen5000@sziit.edu.cn">lizhen5000@sziit.edu.cn</email>
</corresp>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2026-01-21">
<day>21</day>
<month>01</month>
<year>2026</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1752811</elocation-id>
<history>
<date date-type="received">
<day>24</day>
<month>11</month>
<year>2025</year>
</date>
<date date-type="rev-recd">
<day>18</day>
<month>12</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>26</day>
<month>12</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2026 Zhao, Nan, Guo, Xu and Li.</copyright-statement>
<copyright-year>2026</copyright-year>
<copyright-holder>Zhao, Nan, Guo, Xu and Li</copyright-holder>
<license>
<ali:license_ref start_date="2026-01-21">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Coronary atherosclerotic heart disease (CHD) is a leading cause of morbidity and mortality worldwide, making timely identification critical for improving patient prognosis. However, traditional imaging examinations are limited by high costs and patient selection bias, while existing prediction models often lack interpretability and generalization ability. This study aimed to develop a robust, interpretable machine learning approach to address these challenges.</p>
</sec>
<sec>
<title>Methods</title>
<p>This retrospective study analyzed hospitalized patients at Quzhou People&#x2019;s Hospital from July 2021 to March 2025. Patients diagnosed with CHD were categorized as positive samples, while those without cardiovascular disease served as negative controls. The dataset integrated demographic data, blood biomarkers, and vital signs. A Generative Adversarial Imputation Network (GAIN) was utilized to handle missing values, and multiple machine learning models were constructed and compared for prediction performance.</p>
</sec>
<sec>
<title>Results</title>
<p>Among the evaluated algorithms, the XGBoost model achieved superior performance on the test set with an Area Under the Curve (AUC) of 0.9053. To enhance clinical utility, the integration of SHAP (SHapley Additive exPlanations) values enabled both global and local interpretation of model decisions. Key predictive factors identified included mean respiratory rate during hospitalization, age, high-sensitivity troponin I (hs-cTnI), and hypertension.</p>
</sec>
<sec>
<title>Discussion</title>
<p>The developed model demonstrates robust prediction performance combined with high clinical interpretability. Unlike traditional &#x201C;black box&#x201d; models, this approach clarifies the contribution of specific risk factors. Crucially, the tool is well-suited for dual deployment: serving as an automated screening tool integrated into hospital electronic health records (EHRs) and as a self-monitoring aid for individuals with underlying health conditions via mobile health applications.</p>
</sec>
</abstract>
<kwd-group>
<kwd>coronary heart disease</kwd>
<kwd>disease prediction</kwd>
<kwd>explainable machine learning</kwd>
<kwd>feature fusion</kwd>
<kwd>GAIN</kwd>
</kwd-group>
<funding-group>
<funding-statement>The author(s) declared that financial support was received for this work and/or its publication. This work was supported by Zhejiang Health Information Association Research Program (2023XHSZ-Y06), Quzhou City Competitive Science and Technology Project (2025K021) and Doctoral Initiation Projects of Shenzhen Institute of Information Technology (No. SZIIT2025KJ058). The funders were not involved in the study design, collection, analysis, interpretation of data, the writing of this article, or the decision to submit it for publication.</funding-statement>
</funding-group>
<counts>
<fig-count count="8"/>
<table-count count="6"/>
<equation-count count="6"/>
<ref-count count="36"/>
<page-count count="13"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Genomics</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<label>1</label>
<title>Introduction</title>
<p>Coronary atherosclerotic heart disease (CHD) is one of the most prevalent and deadliest diseases worldwide (<xref ref-type="bibr" rid="B33">Yang et al., 2023</xref>). It is characterized by the narrowing or occlusion of the coronary artery lumen. The deleterious effects of CHD are progressive and potentially lethal, manifesting as a spectrum from arrhythmias and angina pectoris to myocardial infarction and heart failure. CHD significantly compromises patients&#x2019; life expectancy and quality of life while imposing a substantial economic burden on families and society (<xref ref-type="bibr" rid="B3">Colantonio et al., 2017</xref>; <xref ref-type="bibr" rid="B10">Ladak et al., 2020</xref>; <xref ref-type="bibr" rid="B17">Pickles and Keller, 2025</xref>).</p>
<p>Beyond therapeutic management, effective risk prediction is crucial, enabling timely intervention and preventative measures. Disease prediction is a continuous spectrum that includes both anticipation of future patients and screening of patients who are currently ill but have not been detected. For chronic and often insidious conditions like CHD, the latter is particularly important (<xref ref-type="bibr" rid="B6">Koloi et al., 2024</xref>). In hospitalized populations, undetected occult CHD significantly elevates perioperative risks&#x2014;especially during non-cardiac surgeries&#x2014;thereby severely impacting prognosis and exacerbating medical burdens. Additionally, for the general population with underlying conditions such as hypertension and diabetes, the occult nature of CHD makes it difficult to detect through routine self-examinations, often leading to delayed diagnosis until severe cardiovascular events occur, causing patients to miss the critical window for early intervention (<xref ref-type="bibr" rid="B23">Sawaf et al., 2024</xref>; <xref ref-type="bibr" rid="B34">Zaninotto et al., 2024</xref>).</p>
<p>Although imaging techniques such as computed tomography angiography (CTA) and invasive coronary angiography (ICA) can assess the degree of coronary artery stenosis and plaque burden, their widespread clinical application is constrained by high costs, operator dependency, and selection bias (<xref ref-type="bibr" rid="B13">Min et al., 2022</xref>; <xref ref-type="bibr" rid="B31">Xiong et al., 2024</xref>). Usually, only patients with a high clinical suspicion of disease undergo these expensive or radiation-intensive procedures. This means that there is a severe lack of healthy but slightly abnormal samples and atypical symptoms cases in the imaging patient dataset. Conversely, biomarkers derived from routine blood tests offer a non-invasive, cost-effective, and scalable evaluation method accessible at all levels of healthcare (<xref ref-type="bibr" rid="B22">Sanchez-Morillo et al., 2024</xref>). Combining personal basic information (gender, age, etc.) with easily accessible vital sign information (blood pressure, blood oxygen saturation SpO<sub>2</sub>, body temperature, etc.) of smart wearable devices can identify high-risk individuals for diseases earlier and more widely (<xref ref-type="bibr" rid="B8">Kundrick et al., 2025</xref>; <xref ref-type="bibr" rid="B15">Nenova and Shang, 2022</xref>).</p>
<p>Although machine learning or deep learning driven models can improve prediction performance, they often lack interpretability due to their &#x201c;black box&#x201d; nature, which cannot clearly reveal the correlation mechanism between risk factors and disease probability (<xref ref-type="bibr" rid="B25">Topranin et al., 2025</xref>; <xref ref-type="bibr" rid="B11">Li et al., 2021</xref>; <xref ref-type="bibr" rid="B12">Liu et al., 2019</xref>), limiting clinical doctors&#x2019; trust in prediction results and the development of personalized intervention strategies. Although traditional models such as Framingham risk score have some interpretability, they have problems such as insufficient prediction accuracy and weak generalization ability, making it difficult to meet the current needs of precision medicine (<xref ref-type="bibr" rid="B20">Rehman et al., 2025</xref>).</p>
<p>Therefore, this research constructed a specific group of non-cardiovascular disease hospitalized patients as negative samples, combined with their personal basic information, blood biomarkers, and vital sign information, to construct an efficient and stable interpretable model for predicting CHD risk. This model not only predicted the probability of individual disease risk, but also clearly explained the specific contributions of various risk factors to the prediction results. The framework flowchart shown in <xref ref-type="fig" rid="F1">Figure 1</xref> illustrates the comprehensive process from data collection to clinical interpretation. This approach aims to provide intuitive basis for clinical doctors to understand the mechanism of disease association and formulate personalized intervention strategies, and to provide low-cost and easy to promote practical tools for independent heart health monitoring in populations with underlying conditions.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The schematic workflow of the model design.</p>
</caption>
<graphic xlink:href="fgene-16-1752811-g001.tif">
<alt-text content-type="machine-generated">Flowchart illustrating data processing for predicting coronary heart disease (CHD) using electronic health records. Data from 2021.7 to 2025.3 includes 19,690 CHD and 17,765 non-CHD cases, grouped by personal information, blood biomarkers, and vital signs. Outlier handling is performed, removing samples with high missing rates. The predictive model comparison graph shows ROC curves for different algorithms, with XGBoost achieving the highest AUC. Analyses include feature combination, SHAP interpretation, and key feature correlation. The GAIN structure combines data imputation and predictive modeling.</alt-text>
</graphic>
</fig>
</sec>
<sec sec-type="methods" id="s2">
<label>2</label>
<title>Methods</title>
<sec id="s2-1">
<label>2.1</label>
<title>Research population design</title>
<p>We retrospectively enrolled hospitalized patients at Quzhou People&#x2019;s Hospital from July 2021 to March 2025. The condition for positive sample collection is based on patients diagnosed as coronary atherosclerotic heart disease after discharge and whose length of stay is &#x2265; 2, excluding patients with cancer/tumor. A total of 19,690 eligible patients with CHD were included. The negative sample set comprised patients without a discharge diagnosis of cardiovascular-related diseases, hospitalized for &#x2265; 2&#xa0;days, excluding patients with cancer/tumors. Ultimately, 17,765 eligible non-cardiovascular disease (non-CHD) patients were included.</p>
<p>Utilizing a healthy population as a control often causes the model to learn merely the generalized differences between ill patients and healthy individuals, rather than the specific pathological features that distinguish CHD from other diseases. Consequently, applying such a model to patients with existing comorbidities often results in unacceptable false positive rates, diminishing the clinical utility of the predicted results. To achieve the goal of disease screening within medical institutions and self-screening among individuals with underlying conditions, this research innovatively used other hospitalized patients with non-cardiovascular diseases as negative controls.</p>
</sec>
<sec id="s2-2">
<label>2.2</label>
<title>Data variables and preprocessing</title>
<p>The dataset comprises three variable categories: demographic characteristics, blood biomarkers, and vital signs, all extracted from electronic health records (EHR). The basic personal information includes the patient&#x2019;s gender, blood type, and age; lifestyle factors included smoking and drinking status; and comorbidities included diabetes and hypertension. Age was recorded at the time of treatment; smoking and drinking status were obtained from medical history records; and diabetes and hypertension status were derived from discharge diagnoses. Blood biomarkers, derived from initial admission tests, included complete blood counts (CBC), biochemical indicators (e.g., liver and kidney function), and D-dimer levels, among others. The vital sign information included the patient&#x2019;s initial admission temperature, heart rate, respiratory rate, systolic blood pressure (SBP), and diastolic blood pressure (DBP), and SpO<sub>2</sub>. Additionally, the maximum, minimum, and mean values of SBP, DBP, body temperature, respiratory rate and SpO<sub>2</sub> measured during hospitalization were recorded.</p>
<p>In laboratory testing, sample quality issues caused by hemolysis, instrument errors, or other factors can produce extreme outliers. As these outliers do not reflect true physiological or pathological states, rigorous detection and cleaning were performed on the blood biomarker data. We adopted a modified Z-score method to identify outliers, which is more robust to outliers (<xref ref-type="bibr" rid="B9">Kuo et al., 2024</xref>).<disp-formula id="equ1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>D</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>Z</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.6745</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mo>/</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>M</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Among them, <inline-formula id="inf1">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the sample feature value, <inline-formula id="inf2">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the median of the sample feature value, and <italic>MAD</italic> is the median absolute deviation. Values with a Z-score &#x3e; 3.5 were identified as outliers and replaced with null values (NaN). Further screening was conducted on samples with missing values below 40%, while retaining samples with more valid data. Consequently, 12 CHD and 7 non-CHD samples were removed.</p>
<p>The partial continuous value feature names, abbreviations, units, distribution descriptions (mean, standard deviation), and missing rates on the positive and negative sample sets are shown in the <xref ref-type="table" rid="T1">Table 1</xref>. Given the high dimensionality of the dataset, the complete feature table is included in the <xref ref-type="sec" rid="s12">Supplementary Material</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Description of partial features.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left">Feature</th>
<th rowspan="2" align="left">Abbreviation</th>
<th rowspan="2" align="left">Unit</th>
<th colspan="3" align="left">CHD</th>
<th colspan="3" align="left">Non-CHD</th>
</tr>
<tr>
<th align="left">Mean</th>
<th align="left">std</th>
<th align="left">Missing_rate (%)</th>
<th align="left">Mean</th>
<th align="left">std</th>
<th align="left">Missing_rate (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Age</td>
<td align="left">Age</td>
<td align="left">years</td>
<td align="left">70.81</td>
<td align="left">10.95</td>
<td align="left">0</td>
<td align="left">60.96</td>
<td align="left">14.33</td>
<td align="left">0</td>
</tr>
<tr>
<td align="left">D-dimer</td>
<td align="left">D-D</td>
<td align="left">mg/L FEU</td>
<td align="left">0.5</td>
<td align="left">0.38</td>
<td align="left">17.42</td>
<td align="left">0.47</td>
<td align="left">0.37</td>
<td align="left">20.37</td>
</tr>
<tr>
<td align="left">High-sensitivity troponin I</td>
<td align="left">Hs-cTnI</td>
<td align="left">&#xb5;g/L</td>
<td align="left">0.005</td>
<td align="left">0.0045</td>
<td align="left">38.88</td>
<td align="left">0.003</td>
<td align="left">0.0037</td>
<td align="left">9.32</td>
</tr>
<tr>
<td align="left">Hemoglobin</td>
<td align="left">HB</td>
<td align="left">g/L</td>
<td align="left">120.73</td>
<td align="left">21.76</td>
<td align="left">0.09</td>
<td align="left">122.07</td>
<td align="left">21.27</td>
<td align="left">0.24</td>
</tr>
<tr>
<td align="left">White blood cell count</td>
<td align="left">WBC</td>
<td align="left">&#x2a;10&#x5e;9/L</td>
<td align="left">6.25</td>
<td align="left">2.1</td>
<td align="left">3.38</td>
<td align="left">6.25</td>
<td align="left">2.22</td>
<td align="left">4.01</td>
</tr>
<tr>
<td align="left">Lymphocyte percentage</td>
<td align="left">LYM%</td>
<td align="left">%</td>
<td align="left">22.2</td>
<td align="left">9.64</td>
<td align="left">0.38</td>
<td align="left">23.63</td>
<td align="left">10.89</td>
<td align="left">0.46</td>
</tr>
<tr>
<td align="left">Monocyte percentage</td>
<td align="left">MO%</td>
<td align="left">%</td>
<td align="left">8.22</td>
<td align="left">2.63</td>
<td align="left">2.09</td>
<td align="left">7.84</td>
<td align="left">2.64</td>
<td align="left">1.4</td>
</tr>
<tr>
<td align="left">Neutrophil percentage</td>
<td align="left">NE%</td>
<td align="left">%</td>
<td align="left">66.57</td>
<td align="left">11.46</td>
<td align="left">0.5</td>
<td align="left">65.77</td>
<td align="left">12.78</td>
<td align="left">0.57</td>
</tr>
<tr>
<td align="left">Eosinophil percentage</td>
<td align="left">EO%</td>
<td align="left">%</td>
<td align="left">1.99</td>
<td align="left">1.6</td>
<td align="left">3.73</td>
<td align="left">1.8</td>
<td align="left">1.56</td>
<td align="left">3.1</td>
</tr>
<tr>
<td align="left">Basophil percentage</td>
<td align="left">BA%</td>
<td align="left">%</td>
<td align="left">0.41</td>
<td align="left">0.26</td>
<td align="left">0.8</td>
<td align="left">0.4</td>
<td align="left">0.27</td>
<td align="left">0.61</td>
</tr>
<tr>
<td align="left">Neutrophil count</td>
<td align="left">NE</td>
<td align="left">&#x2a;10&#x5e;9/L</td>
<td align="left">4.12</td>
<td align="left">1.72</td>
<td align="left">4.79</td>
<td align="left">4.06</td>
<td align="left">1.86</td>
<td align="left">5.61</td>
</tr>
<tr>
<td align="left">Lymphocyte count</td>
<td align="left">LYM</td>
<td align="left">&#x2a;10&#x5e;9/L</td>
<td align="left">1.32</td>
<td align="left">0.57</td>
<td align="left">0.9</td>
<td align="left">1.4</td>
<td align="left">0.6</td>
<td align="left">1.11</td>
</tr>
<tr>
<td align="left">Monocyte count</td>
<td align="left">MO</td>
<td align="left">&#x2a;10&#x5e;9/L</td>
<td align="left">0.51</td>
<td align="left">0.2</td>
<td align="left">2.69</td>
<td align="left">0.48</td>
<td align="left">0.2</td>
<td align="left">2.44</td>
</tr>
<tr>
<td align="left">Eosinophil count</td>
<td align="left">EO</td>
<td align="left">&#x2a;10&#x5e;9/L</td>
<td align="left">0.11</td>
<td align="left">0.09</td>
<td align="left">4.76</td>
<td align="left">0.1</td>
<td align="left">0.09</td>
<td align="left">3.73</td>
</tr>
<tr>
<td align="left">Basophil count</td>
<td align="left">BA</td>
<td align="left">&#x2a;10&#x5e;9/L</td>
<td align="left">0.02</td>
<td align="left">0.02</td>
<td align="left">2.08</td>
<td align="left">0.02</td>
<td align="left">0.02</td>
<td align="left">1.73</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-3">
<label>2.3</label>
<title>GAIN architecture</title>
<p>Generative adversarial imputation Nets (GAIN) represent a data imputation method based on generative adversarial networks (GANs) (<xref ref-type="bibr" rid="B32">Xu et al., 2025</xref>; <xref ref-type="bibr" rid="B14">Nayak et al., 2024</xref>). By leveraging adversarial training between a generator and a discriminator, GAIN learns the underlying data distribution to generate plausible imputed values. The architecture is illustrated in <xref ref-type="fig" rid="F1">Figure 1</xref>. The generator, which serves as the core component for imputation, utilizes a three-layer fully connected neural network structure. The input consists of the original data tensor <inline-formula id="inf3">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">X</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> containing missing values and the mask tensor <inline-formula id="inf4">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, where 1 denotes observed data and 0 denotes missing data. The discriminator is tasked with distinguishing between observed true values and imputed values produced by the generator; its network structure mirrors that of the generator. The discriminator input comprises a concatenated tensor of the imputed data <inline-formula id="inf5">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">X</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the hint vector <inline-formula id="inf6">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">H</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The hint vector, derived from a randomly generated probability matrix and a mask tensor, provides auxiliary information to assist the discriminator in identifying missingness patterns. The generator loss is composed of a weighted adversarial loss <inline-formula id="inf7">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mrow>
<mml:mtext>loss</mml:mtext>
<mml:mo>_</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and an MSE loss <inline-formula id="inf8">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mrow>
<mml:mtext>loss</mml:mtext>
<mml:mo>_</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, where the adversarial loss is achieved by minimizing the discriminator&#x2019;s recognition accuracy of the generated values, and the MSE loss constrains the generator to not destroy the original information at known data positions. Optimize the discriminative ability of the discriminator by calculating the classification loss <inline-formula id="inf9">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">D</mml:mi>
<mml:mtext>loss</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> between the original real data and the generated imputed data. The entire model gradually learns the inherent distribution pattern of the data through a continuous adversarial game between the generator and discriminator, ultimately generating missing values that are close to the true distribution.</p>
</sec>
<sec id="s2-4">
<label>2.4</label>
<title>Evaluation metrics</title>
<p>To comprehensively evaluate the model&#x2019;s predictive capability and clinical utility, we employed multiple complementary metrics, including accuracy, precision, recall, F1-score and Area Under the Receiver Operating Characteristic Curve (AUC) (<xref ref-type="bibr" rid="B1">Chen et al., 2025a</xref>; <xref ref-type="bibr" rid="B7">Kumar et al., 2024</xref>; <xref ref-type="bibr" rid="B21">Rimal and Sharma, 2023</xref>; <xref ref-type="bibr" rid="B2">Chen et al., 2025b</xref>; <xref ref-type="bibr" rid="B18">Qiao et al., 2024</xref>). These are the core metrics for evaluating the performance of binary classification models, calculated based on four fundamental values in the confusion matrix: true positive cases (TP), false positive cases (FP), true negative cases (TN), and false negative cases (FN) (<xref ref-type="bibr" rid="B35">Zeng et al., 2025</xref>; <xref ref-type="bibr" rid="B36">Zulfiqar et al., 2024</xref>; <xref ref-type="bibr" rid="B19">Qiao et al., 2025</xref>; <xref ref-type="bibr" rid="B30">Xie et al., 2025</xref>; <xref ref-type="bibr" rid="B28">Wang et al., 2025</xref>; <xref ref-type="bibr" rid="B27">Wang et al., 2024</xref>).<disp-formula id="equ3">
<mml:math id="m12">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mo>/</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ4">
<mml:math id="m13">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mo>/</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ5">
<mml:math id="m14">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mo>/</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ6">
<mml:math id="m15">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mo>/</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Discussion and results</title>
<sec id="s3-1">
<label>3.1</label>
<title>Analysis of data distribution and feature missing</title>
<p>
<xref ref-type="fig" rid="F2">Figure 2</xref> showed the visualization results of positive and negative samples using t-distributed stochastic neighbor embedding (t-SNE), which was used to display the distribution patterns of CHD and non-CHD samples in a high-dimensional feature space, intuitively presenting the feature differences and clustering patterns of the two groups (<xref ref-type="bibr" rid="B16">Nollmann et al., 2024</xref>).</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>t-SNE visualization scatter plot and density profile plot of CHD and non-CHD samples.</p>
</caption>
<graphic xlink:href="fgene-16-1752811-g002.tif">
<alt-text content-type="machine-generated">Two t-SNE visualizations compare data points for CHD (in red) and non-CHD (in blue). The left scatter plot shows dense clusters. The right plot includes density contours illustrating data concentration, enhancing cluster visibility.</alt-text>
</graphic>
</fig>
<p>In the scatter plot, blue (non-CHD) samples form a core cluster and two independent small clusters, indicating that the characteristics of the non-CHD population have strong concentration. The red (CHD) samples are interspersed in the form of scattered dots within and at the edges of the blue clusters, with only mild clustering in local areas and a relatively scattered overall distribution. In the density contour map, the density contour of non-CHD samples covers most of the areas in the map, and the core area has a high density, further verifying the concentration of non-CHD population characteristics and the stability of subgroup structure. The density profile of CHD samples highly overlaps with non-CHD, with only weak independent trends in local areas, indicating that the characteristic boundaries between CHD and non-CHD are blurred and overlap is high. At the same time, the heterogeneity of CHD features, such as different disease courses, subtypes, and comorbidities leads to their scattered distribution.</p>
<p>
<xref ref-type="fig" rid="F3">Figure 3</xref> showed the heatmap of feature missing rates for the CHD and non-CHD groups. The missing rates for most variables were similar in both groups, but significant differences existed in key clinical indicators. The CHD group had significantly higher missing rates for hs-cTnI (38.88%) and HbA1c (33.12%) than the non-CHD group (9.32% and 5.53%, respectively). This difference may reflect insufficient detection of these important diagnostic and monitoring indicators in patients with CHD in clinical practice. In contrast, the non-CHD group had slightly higher missing rates for indicators such as D-dimer (20.37%) and hs-CRP (26.16%). The missing rates for most routine laboratory indicators and vital signs remained below 10% in both groups, indicating relatively complete basic clinical data collection. However, key indicators with high missing rates require appropriate missing data processing strategies in subsequent analyses.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Heatmap of missing feature rates for CHD and non-CHD samples.</p>
</caption>
<graphic xlink:href="fgene-16-1752811-g003.tif">
<alt-text content-type="machine-generated">Heatmap comparing missing values for CHD and non-CHD sample types. The vertical axis lists variables such as &#x22;blood type,&#x22; &#x22;gender,&#x22; and others. The horizontal axis displays missing value percentages, ranging from 0 to over 35 percent, with colors from purple (low) to yellow (high). Notable variables with higher missing rates include &#x22;hs-Ctnt&#x22; and &#x22;D-D,&#x22; with differences between sample types.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3-2">
<label>3.2</label>
<title>Performance analysis of imputation models</title>
<p>Although incomplete recorded data may be reasonable in clinical practice, the performance of machine learning algorithms is often affected by biased and incomplete data. Medical record data is extremely valuable for disease research. If partially missing samples are directly removed and models are constructed using non missing samples, although this approach is simple, it wastes a lot of available information.</p>
<p>This research comprehensively compared and analyzed the performance of traditional imputation algorithms (KNN, MICE) (<xref ref-type="bibr" rid="B26">Varol et al., 2025</xref>), deep learning autoencoder series (AE, DAE, VAE) (<xref ref-type="bibr" rid="B5">Gautier et al., 2024</xref>; <xref ref-type="bibr" rid="B24">Shi et al., 2024</xref>), and GAIN in medical record data. Due to the lack of real data references, the performance of the downstream tasks is generally taken as the standard. <xref ref-type="table" rid="T2">Table 2</xref> compared the 5-fold cross validation (5-cv) performance of the imputation algorithms, and <xref ref-type="table" rid="T3">Table 3</xref> compared its performance on the independent test set. The performance metrics of traditional imputation algorithms were significantly lower than those of deep learning methods, and they were limited to being unable to adapt to the complex nonlinear correlations between features in the data, resulting in insufficient expression ability in high-dimensional medical data scenarios.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Comparison of 5-cv performance of imputation algorithms.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Method</th>
<th align="center">Accuracy</th>
<th align="center">Precision</th>
<th align="center">Recall</th>
<th align="center">F1</th>
<th align="center">AUC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">KNN</td>
<td align="right">0.8208</td>
<td align="right">0.8206</td>
<td align="right">0.8198</td>
<td align="right">0.8201</td>
<td align="right">0.9016</td>
</tr>
<tr>
<td align="center">MICE</td>
<td align="right">0.8265</td>
<td align="right">0.8263</td>
<td align="right">0.8256</td>
<td align="right">0.8259</td>
<td align="right">0.9078</td>
</tr>
<tr>
<td align="center">AE</td>
<td align="right">0.8322</td>
<td align="right">0.8321</td>
<td align="right">0.8313</td>
<td align="right">0.8316</td>
<td align="right">0.9146</td>
</tr>
<tr>
<td align="center">DAE</td>
<td align="right">0.8350</td>
<td align="right">0.8349</td>
<td align="right">0.8339</td>
<td align="right">0.8343</td>
<td align="right">0.9157</td>
</tr>
<tr>
<td align="center">VAE</td>
<td align="right">0.8388</td>
<td align="right">0.8385</td>
<td align="right">0.8382</td>
<td align="right">0.8383</td>
<td align="right">0.9196</td>
</tr>
<tr>
<td align="center">GAIN</td>
<td align="right">0.8343</td>
<td align="right">0.8342</td>
<td align="right">0.8333</td>
<td align="right">0.8336</td>
<td align="right">0.9154</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Comparison of independent testing performance of imputation algorithms.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Method</th>
<th align="center">Accuracy</th>
<th align="center">Precision</th>
<th align="center">Recall</th>
<th align="center">F1</th>
<th align="center">AUC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">KNN</td>
<td align="right">0.8096</td>
<td align="right">0.8145</td>
<td align="right">0.8261</td>
<td align="right">0.8202</td>
<td align="right">0.8946</td>
</tr>
<tr>
<td align="center">MICE</td>
<td align="right">0.8234</td>
<td align="right">0.8256</td>
<td align="right">0.8420</td>
<td align="right">0.8337</td>
<td align="right">0.9043</td>
</tr>
<tr>
<td align="center">AE</td>
<td align="right">0.8285</td>
<td align="right">0.8281</td>
<td align="right">0.8503</td>
<td align="right">0.8390</td>
<td align="right">0.9095</td>
</tr>
<tr>
<td align="center">DAE</td>
<td align="right">0.8317</td>
<td align="right">0.8342</td>
<td align="right">0.8486</td>
<td align="right">0.8413</td>
<td align="right">0.9135</td>
</tr>
<tr>
<td align="center">VAE</td>
<td align="right">0.8350</td>
<td align="right">0.8381</td>
<td align="right">0.8504</td>
<td align="right">0.8442</td>
<td align="right">0.9165</td>
</tr>
<tr>
<td align="center">GAIN</td>
<td align="right">0.8354</td>
<td align="right">0.8405</td>
<td align="right">0.8477</td>
<td align="right">0.8441</td>
<td align="right">0.9156</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>As a basic autoencoder, AE achieved a 5-cv AUC of 0.9146 and an independent test AUC of 0.9095, which preliminarily demonstrated the modeling ability of deep learning for complex data. DAE enhanced robustness through noise reduction mechanism, further improving performance (5-cv AUC 0.9157, independent test AUC 0.9135), and had a higher tolerance for data noise. After introducing variational inference, VAE had better flexibility in distribution modeling, with a 5-cv AUC 0.9196 and independent testing AUC 0.9165, ranking among the top in multiple performance metrics. As a generative model, GAIN not only considered the distribution of individual features in adversarial learning, but also comprehensively considers the complex correlations between all other features. Its 5-cv AUC 0.9154 and independent test AUC 0.9156 were slightly lower than VAE.</p>
<p>The primary goal of imputation in medical research is to faithfully preserve the original data distribution and minimize bias. We further observed the fitting degree of each imputation algorithm on the data distribution, and selected hs-cTnI and Jaun features with high missing rates. The feature density curves before and after imputation were shown in <xref ref-type="fig" rid="F4">Figures 4</xref>, <xref ref-type="fig" rid="F5">5</xref>. Compared to VAE, the data distribution after GAIN imputation has a higher degree of fit with the original data in terms of morphology, which can avoid additional bias caused by distribution offset due to imputation and ensure the authenticity and effectiveness of feature information in subsequent analysis. To strictly quantify this observation, we used Kullback-Leibler (KL) divergence and Kolmogorov-Smirnov (KS) test statistics for the imputed data of VAE and GAIN on independent test sets. A lower KL or KS value indicates a distribution closer to the ground truth., GAIN achieved the lowest mean KL divergence (0.158) and mean KS statistic (0.067) outperforming VAE (KL: 0.210; KS: 0.068). This statistical evidence demonstrates that GAIN is superior in capturing the complex underlying probability distribution of the real data, avoiding the distributional shifts often introduced by variational inference in VAEs. Consequently, considering both the robust downstream performance and the superior data fidelity, GAIN was selected as the optimal imputation algorithm for this research.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Density curve of feature hs-cTnI distribution before and after imputation.</p>
</caption>
<graphic xlink:href="fgene-16-1752811-g004.tif">
<alt-text content-type="machine-generated">Density plot comparing original hs-cTnI data with multiple imputation methods. Original data is shown in blue, with imputed methods in various colors: KNN (red), MICE (green), AE (orange), DAE (purple), VAE (brown), and GAIN (pink). X-axis represents hs-cTnI values and Y-axis represents density.</alt-text>
</graphic>
</fig>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Density curve of feature Jaun distribution before and after imputation.</p>
</caption>
<graphic xlink:href="fgene-16-1752811-g005.tif">
<alt-text content-type="machine-generated">Density plot comparing original data and multiple imputation methods for &#x22;Jaun&#x22; values. The methods include KNN, MICE, AE, DAE, VAE, and GAIN. Each method has a distinct style in the legend.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3-3">
<label>3.3</label>
<title>Comparison of prediction algorithm performance</title>
<p>In this section, we conducted a performance comparison analysis of XGBoost, random forest, logistic regression, SVM, KNN, AdaBoost, and ANN algorithms. The prediction metrics of training set in <xref ref-type="table" rid="T4">Table 4</xref> showed that XGBoost (AUC &#x3d; 0.9184, Accuracy &#x3d; 0.8345) and ANN (AUC &#x3d; 0.9182, Accuracy &#x3d; 0.8381) have the most outstanding comprehensive performance. In the test set prediction performance in <xref ref-type="table" rid="T5">Table 5</xref>, XGBoost exhibited excellent generalization stability, while ANN&#x2019;s generalization ability is significantly insufficient. <xref ref-type="fig" rid="F6">Figure 6</xref> showed the comparison of ROC curves of different prediction algorithms on the training and testing sets, which intuitively proves that the XGBoost model had the strongest ability to distinguish positive and negative samples and excellent generalization.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Comparison of performance metrics of different prediction algorithms on the training set.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model</th>
<th align="center">Accuracy</th>
<th align="center">Precision</th>
<th align="center">Recall</th>
<th align="center">F1</th>
<th align="center">AUC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">XGBoost</td>
<td align="center">0.8345</td>
<td align="center">0.8346</td>
<td align="center">0.8345</td>
<td align="center">0.8344</td>
<td align="center">0.9184</td>
</tr>
<tr>
<td align="center">Random forest</td>
<td align="center">0.7999</td>
<td align="center">0.8019</td>
<td align="center">0.7999</td>
<td align="center">0.7989</td>
<td align="center">0.8874</td>
</tr>
<tr>
<td align="center">Logistic regression</td>
<td align="center">0.7789</td>
<td align="center">0.7788</td>
<td align="center">0.7789</td>
<td align="center">0.7787</td>
<td align="center">0.8582</td>
</tr>
<tr>
<td align="center">SVM</td>
<td align="center">0.8108</td>
<td align="center">0.8113</td>
<td align="center">0.8108</td>
<td align="center">0.8104</td>
<td align="center">0.8923</td>
</tr>
<tr>
<td align="center">KNN</td>
<td align="center">0.7963</td>
<td align="center">0.8004</td>
<td align="center">0.7963</td>
<td align="center">0.7963</td>
<td align="center">0.8844</td>
</tr>
<tr>
<td align="center">AdaBoost</td>
<td align="center">0.8079</td>
<td align="center">0.8078</td>
<td align="center">0.8079</td>
<td align="center">0.8078</td>
<td align="center">0.8911</td>
</tr>
<tr>
<td align="center">ANN</td>
<td align="center">0.8381</td>
<td align="center">0.8382</td>
<td align="center">0.8381</td>
<td align="center">0.8379</td>
<td align="center">0.9182</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Comparison of performance metrics of different prediction algorithms on the test set.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model</th>
<th align="center">Accuracy</th>
<th align="center">Precision</th>
<th align="center">Recall</th>
<th align="center">F1</th>
<th align="center">AUC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">XGBoost</td>
<td align="center">0.8246</td>
<td align="center">0.8247</td>
<td align="center">0.8246</td>
<td align="center">0.8246</td>
<td align="center">0.9053</td>
</tr>
<tr>
<td align="center">Random forest</td>
<td align="center">0.7868</td>
<td align="center">0.7888</td>
<td align="center">0.7868</td>
<td align="center">0.7868</td>
<td align="center">0.8732</td>
</tr>
<tr>
<td align="center">Logistic regression</td>
<td align="center">0.7750</td>
<td align="center">0.7749</td>
<td align="center">0.7750</td>
<td align="center">0.7750</td>
<td align="center">0.8517</td>
</tr>
<tr>
<td align="center">SVM</td>
<td align="center">0.7919</td>
<td align="center">0.7922</td>
<td align="center">0.7919</td>
<td align="center">0.7919</td>
<td align="center">0.8744</td>
</tr>
<tr>
<td align="center">KNN</td>
<td align="center">0.7356</td>
<td align="center">0.7411</td>
<td align="center">0.7356</td>
<td align="center">0.7356</td>
<td align="center">0.8177</td>
</tr>
<tr>
<td align="center">AdaBoost</td>
<td align="center">0.7976</td>
<td align="center">0.7976</td>
<td align="center">0.7976</td>
<td align="center">0.7976</td>
<td align="center">0.8811</td>
</tr>
<tr>
<td align="center">ANN</td>
<td align="center">0.8035</td>
<td align="center">0.8034</td>
<td align="center">0.8035</td>
<td align="center">0.8035</td>
<td align="center">0.8854</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Comparison of ROC curves of different prediction algorithms on the training and testing sets.</p>
</caption>
<graphic xlink:href="fgene-16-1752811-g006.tif">
<alt-text content-type="machine-generated">Two ROC curve charts compare model performance on train and test sets. The left chart depicts curves for XGBoost, Random Forest, Logistic Regression, SVM, KNN, AdaBoost, and ANN with AUC ranging from 0.858 to 0.918. The right chart presents test set results with AUC ranging from 0.818 to 0.905. A diagonal dashed line represents a random guess baseline.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3-4">
<label>3.4</label>
<title>Feature combination analysis</title>
<p>In many research, derived indicators based on blood biomarkers have shown excellent performance. We have established 8 derived indicators according to the obtained blood markers, namely, plasma atherogenic index (AIP &#x3d; log<sub>10</sub> [TG/HDL]), uric acid/high-density lipoprotein ratio (UHR &#x3d; UA/(18 &#x2a; HDL)), neutrophil/lymphocyte ratio (NLR &#x3d; NE/LYM), platelet/lymphocyte ratio (PLR &#x3d; PLT/LYM), monocyte/lymphocyte ratio (MLR &#x3d; MO/LYM), and systemic immune inflammation index (SII &#x3d; PLT &#xd7; NE/LYM), systemic inflammatory response index (SIRI &#x3d; NE &#xd7; MO/LYM), and systemic inflammatory composite index (AISI &#x3d; NE &#xd7; MO &#xd7; PLT/LYM) (<xref ref-type="bibr" rid="B29">Wu et al., 2023</xref>; <xref ref-type="bibr" rid="B4">E et al., 2025</xref>).</p>
<p>In this section, we focused on different types of features, such as basic information (BI), blood biomarkers (BB), vital signs information (VSI), derivative indicators (DI). Feature combinations analysis was conducted, and its performance on the test set is shown in <xref ref-type="table" rid="T6">Table 6</xref>. From the view of single category features, blood biomarkers (BB) demonstrated core prediction value, reflecting the direct correlation of blood biomarkers in the pathological mechanisms of CHD such as lipid metabolism and inflammatory response. In multi class feature combinations, the performance of three class feature fusion (BI&#x2b;BB&#x2b;VSI) reached its peak, with accuracy 0.8218 and AUC 0.9047 being the best among all combinations. The integration of basic information, blood biomarkers, and vital signs has constructed a complete CHD risk profile from three dimensions, clinical phenotype, biochemical mechanisms, and physiological status, maximizing the complementarity between features. Although the feature combination performance of DI is theoretically guaranteed, it does not exceed BI&#x2b;BB&#x2b;VSI. This may be because DI introduced redundant information, which slightly interferes with the model&#x2019;s generalization. This also proved the prediction algorithm&#x2019;s ability to mine the cross-complementarity of feature.</p>
<table-wrap id="T6" position="float">
<label>TABLE 6</label>
<caption>
<p>Performance comparison of different feature combinations on the test set.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Feature set</th>
<th align="center">Accuracy</th>
<th align="center">Precision</th>
<th align="center">Recall</th>
<th align="center">F1</th>
<th align="center">AUC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">BI</td>
<td align="center">0.7078</td>
<td align="center">0.7081</td>
<td align="center">0.7078</td>
<td align="center">0.7065</td>
<td align="center">0.7776</td>
</tr>
<tr>
<td align="center">BB</td>
<td align="center">0.7510</td>
<td align="center">0.7510</td>
<td align="center">0.7510</td>
<td align="center">0.7505</td>
<td align="center">0.8312</td>
</tr>
<tr>
<td align="center">VSI</td>
<td align="center">0.6988</td>
<td align="center">0.7008</td>
<td align="center">0.6988</td>
<td align="center">0.6990</td>
<td align="center">0.7786</td>
</tr>
<tr>
<td align="center">DI</td>
<td align="center">0.5922</td>
<td align="center">0.5922</td>
<td align="center">0.5922</td>
<td align="center">0.5844</td>
<td align="center">0.6212</td>
</tr>
<tr>
<td align="center">BI&#x2b;BB</td>
<td align="center">0.7823</td>
<td align="center">0.7829</td>
<td align="center">0.7823</td>
<td align="center">0.7817</td>
<td align="center">0.8634</td>
</tr>
<tr>
<td align="center">BI&#x2b;VSI</td>
<td align="center">0.7868</td>
<td align="center">0.7869</td>
<td align="center">0.7868</td>
<td align="center">0.7865</td>
<td align="center">0.8704</td>
</tr>
<tr>
<td align="center">BI&#x2b;DI</td>
<td align="center">0.7160</td>
<td align="center">0.7168</td>
<td align="center">0.716</td>
<td align="center">0.7144</td>
<td align="center">0.7849</td>
</tr>
<tr>
<td align="center">BB&#x2b;VSI</td>
<td align="center">0.7979</td>
<td align="center">0.7978</td>
<td align="center">0.7979</td>
<td align="center">0.7979</td>
<td align="center">0.8812</td>
</tr>
<tr>
<td align="center">BB&#x2b;DI</td>
<td align="center">0.7532</td>
<td align="center">0.7532</td>
<td align="center">0.7532</td>
<td align="center">0.7527</td>
<td align="center">0.8308</td>
</tr>
<tr>
<td align="center">VSI&#x2b;DI</td>
<td align="center">0.7105</td>
<td align="center">0.7116</td>
<td align="center">0.7105</td>
<td align="center">0.7107</td>
<td align="center">0.7945</td>
</tr>
<tr>
<td align="center">BI&#x2b;BB&#x2b;VSI</td>
<td align="center">
<bold>0.8218</bold>
</td>
<td align="center">
<bold>0.8219</bold>
</td>
<td align="center">
<bold>0.8218</bold>
</td>
<td align="center">
<bold>0.8216</bold>
</td>
<td align="center">
<bold>0.9047</bold>
</td>
</tr>
<tr>
<td align="center">BI&#x2b;BB&#x2b;DI</td>
<td align="center">0.7786</td>
<td align="center">0.7792</td>
<td align="center">0.7786</td>
<td align="center">0.7780</td>
<td align="center">0.8607</td>
</tr>
<tr>
<td align="center">BI&#x2b;VSI&#x2b;DI</td>
<td align="center">0.7866</td>
<td align="center">0.7865</td>
<td align="center">0.7866</td>
<td align="center">0.7863</td>
<td align="center">0.8709</td>
</tr>
<tr>
<td align="center">BB&#x2b;VSI&#x2b;DI</td>
<td align="center">0.7984</td>
<td align="center">0.7984</td>
<td align="center">0.7984</td>
<td align="center">0.7984</td>
<td align="center">0.8813</td>
</tr>
<tr>
<td align="center">BI&#x2b;BB&#x2b;VSI&#x2b;DI</td>
<td align="center">0.8208</td>
<td align="center">0.8208</td>
<td align="center">0.8208</td>
<td align="center">0.8205</td>
<td align="center">0.9044</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold text denoted the best performance among different feature combinations.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3-5">
<label>3.5</label>
<title>SHAP based model interpretation and key feature correlation analysis</title>
<p>The highest mean absolute SHAP value of mean-RR in <xref ref-type="fig" rid="F6">Figure 6</xref> indicated that it has the most significant global influence on CHD prediction in the model, followed by age, hs-cTnI, and hypertension, which collectively constitute the core drivers of model decision-making. We further analyzed the SHAP dependency plots for key features in <xref ref-type="fig" rid="F7">Figure 7</xref>. The mean-RR dependency plot showed that a low respiratory rate is weighted as a positive contributor to CHD risk. In hospitalized patients or underlying disease populations, shortness of breath is an extremely common non-specific symptom with various causes, such as pain, anemia, anxiety, etc. Through data-driven analysis, the model identified high RR as strongly correlated with non-CHD hospitalization causes. Consequently, a relatively lower RR served as a distinguishing signal for occult CHD within this specific patient population. Regarding age, SHAP values increased monotonically, confirming age as a robust risk factor. The observed plateauing effect in the elderly suggested a deceleration in risk accumulation, consistent with established clinical knowledge regarding the progression of coronary atherosclerosis. For hs-cTnI, when hs-cTnI exceeded 0.0024, the SHAP value rapidly turned positive and remained at a high level, even within the clinical normal reference range, indicating that an increase in hs-cTnI has significantly increased the risk of CHD. This highlighted the sensitivity of high-sensitivity troponin in early myocardial injury and risk prediction. The role of SHAP analysis is limited to reflecting specific behavioral patterns of the model and cannot be used to infer causal relationships. At the same time, the numerical values of its results will also vary with the distribution of data and the structure of the model.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>The left sub figure: the top 20 features with mean absolute SHAP values. The right three sub figures: SHAP dependency plots for the top 3 key features including SHAP value fitting curves and feature boundary values when SHAP &#x3d; 0.</p>
</caption>
<graphic xlink:href="fgene-16-1752811-g007.tif">
<alt-text content-type="machine-generated">Bar chart and scatter plots show SHAP analysis for a medical dataset. The bar chart ranks features by their mean SHAP values, with &#x27;mean-RR&#x27;, &#x27;age&#x27;, and &#x27;hs-cTnI&#x27; being most influential. Scatter plots depict SHAP value dependencies on &#x27;mean-RR&#x27;, &#x27;age&#x27;, and &#x27;hs-cTnI&#x27; with observed trends and highlighted thresholds.</alt-text>
</graphic>
</fig>
<p>However, the control group may include individuals with acute respiratory conditions, potentially introducing a confounding bias where elevated respiratory rates reflect the pathology of the control group rather than a direct risk factor for CHD. To address this concern and verify the model&#x2019;s robustness, we conducted a sensitivity analysis by excluding all respiratory-related features (mean-RR, max-RR, min-RR, and 1st-RR) and retraining the XGBoost model. The results showed that while the AUC on the test set experienced a moderate decline from 0.9053 to 0.8693, it remained within a clinically excellent range. This performance retention confirmed that although respiratory rate significantly contributes to discrimination, the model&#x2019;s prediction power is fundamentally driven by the comprehensive integration of multiparametric features, rather than solely relying on distinguishing respiratory-related anomalies in the control group.</p>
<p>To further explore feature interactions, we calculated the Spearman rank correlation coefficients for the top 20 features ranked by SHAP importance, as visualized in <xref ref-type="fig" rid="F8">Figure 8</xref>. The absolute value of the correlation coefficient was represented by the radius of a circle. The larger the radius, the stronger the correlation. The color represented the direction of correlation, with red indicating positive correlation and blue indicating negative correlation. The &#x201c;X&#x201d; in the upper triangle indicated that the feature pair is not statistically significant, while the specific correlation coefficient values were labeled in the lower triangle. The diagonal represented the autocorrelation of the feature. In addition to RR series features, the correlation coefficients of lipid metabolism features Ch, ApoB, and LDL are 0.79, 0.77, and 0.76, respectively, reflecting their synergistic effects in the process of lipid transport and atherosclerosis. This type of strong correlation prompt required attention to the joint effect of feature groups when interpreting model decisions, rather than the independent contribution of a single feature. However, features without statistically significant associations, such as age and some metabolic indicators, smoke and LDL, had no statistically significant association at the SHAP level (P &#x3e; 0.05), indicating that the weight allocation of these features by the model was relatively independent and can reduce the interference of multicollinearity on model stability.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Spearman correlation of TOP20 features.</p>
</caption>
<graphic xlink:href="fgene-16-1752811-g008.tif">
<alt-text content-type="machine-generated">Correlation matrix displaying the Spearman correlation coefficients for the top 20 SHAP values features in a test set. Larger red circles indicate stronger positive correlations, while smaller blue circles or crosses denote weaker or negative correlations. The strongest correlations are between hypertension and mean respiratory rate, max-RR, and 1st-RR, which are all above 0.5. The color gradient bar on the right visually represents the correlation coefficient scale from blue (negative) to red (positive).</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3-6">
<label>3.6</label>
<title>Clinical implementation and limitations</title>
<p>To facilitate the clinical translation of this low-cost screening tool, a dual approach can be implemented: integration into in-hospital EHR systems to generate automated risk alerts for non-cardiology departments, and the development of mobile health applications to enable self-monitoring for individuals with underlying conditions. However, widespread deployment necessitates addressing critical barriers, including strict adherence to data privacy regulations, the necessity of dynamic model updating to counter concept drift, and the challenge of fostering clinician trust&#x2014;which is partially mitigated by the SHAP interpretability framework employed in this research. Furthermore, the interpretation of findings must be tempered by the limitations of a retrospective, single-center design. While the current model incorporates age and comorbidities as features, rigorous validation in future multi-center prospective studies is required to ensure fairness, robustness, and equitable healthcare outcomes across diverse subpopulations.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s4">
<label>4</label>
<title>Conclusion</title>
<p>This research successfully developed a machine learning based CHD risk prediction model, effectively improving its generalization ability and practicality in complex clinical backgrounds. By using GAIN imputation method to process missing data, combined with XGBoost algorithm to achieve high-precision prediction, and utilizing SHAP method to reveal the contribution of key features to the prediction results, the interpretability of the model is enhanced. The research results indicate that the model has significant potential in identifying hidden CHD, which can assist clinical doctors in early intervention and personalized management, and provide a low-cost and easy to promote self-monitoring method for the population with underlying diseases. Although the model cannot replace the gold-standard diagnosis, it has important practical value in the connection between public health prevention and clinical diagnosis and treatment, which helps to reduce medical burden.</p>
<p>This approach aims to provide intuitive basis for clinical doctors to understand the mechanism of disease association and formulate personalized intervention strategies, and to provide low-cost and easy to promote practical tools for independent heart health monitoring in populations with underlying conditions.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s12">Supplementary Material</xref>, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec sec-type="ethics-statement" id="s6">
<title>Ethics statement</title>
<p>The studies involving humans were approved by Ethics Committee of Quzhou People&#x2019;s Hospital. The studies were conducted in accordance with the local legislation and institutional requirements.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>SZ: Writing &#x2013; original draft, Methodology. BN: Validation, Writing &#x2013; review and editing, Supervision, Conceptualization. JG: Methodology, Writing &#x2013; original draft, Data curation. WX: Writing &#x2013; review and editing. ZL: Writing &#x2013; original draft, Methodology.</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The author(s) declared that this work was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The author(s) declared that generative AI was not used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s12">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2025.1752811/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2025.1752811/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table1.docx" id="SM1" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<fn-group>
<fn fn-type="custom" custom-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/637149/overview">Chunyu Wang</ext-link>, Harbin Institute of Technology, China</p>
</fn>
<fn fn-type="custom" custom-type="reviewed-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/351544/overview">Ran Su</ext-link>, Tianjin University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/876620/overview">Yongqing Zhang</ext-link>, Chengdu University of Information Technology, China</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yue</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2025a</year>). <article-title>MoSViT: a lightweight vision transformer framework for efficient disease detection <italic>via</italic> precision attention mechanism</article-title>. <source>Front. Artif. Intell.</source> <volume>8</volume>, <fpage>1498025</fpage>. <pub-id pub-id-type="doi">10.3389/frai.2025.1498025</pub-id>
<pub-id pub-id-type="pmid">40206703</pub-id>
</mixed-citation>
</ref>
<ref id="B2">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z. A.</given-names>
</name>
<etal/>
</person-group> (<year>2025b</year>). <article-title>Self-supervised learning in drug discovery</article-title>. <source>Sci. China Inf. Sci.</source> <volume>68</volume> (<issue>7</issue>), <fpage>170103</fpage>. <pub-id pub-id-type="doi">10.1007/s11432-024-4453-4</pub-id>
</mixed-citation>
</ref>
<ref id="B3">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Colantonio</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gamboa</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Richman</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Levitan</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Soliman</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Howard</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Black-white differences in incident fatal, nonfatal, and total coronary heart disease</article-title>. <source>Circulation</source> <volume>136</volume> (<issue>2</issue>), <fpage>152</fpage>&#x2013;<lpage>166</lpage>. <pub-id pub-id-type="doi">10.1161/CIRCULATIONAHA.116.025848</pub-id>
<pub-id pub-id-type="pmid">28696265</pub-id>
</mixed-citation>
</ref>
<ref id="B4">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>E</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ge</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Huo</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Development and validation of a machine learning model for predicting vulnerable carotid plaques using routine blood biomarkers and derived indicators: insights into sex-related risk patterns</article-title>. <source>Cardiovasc. Diabetol.</source> <volume>24</volume> (<issue>1</issue>), <fpage>326</fpage>. <pub-id pub-id-type="doi">10.1186/s12933-025-02867-6</pub-id>
<pub-id pub-id-type="pmid">40784899</pub-id>
</mixed-citation>
</ref>
<ref id="B5">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gautier</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Bousse</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sureau</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Comtat</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Maxim</surname>
<given-names>S. B.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Bimodal PET/MRI generative reconstruction based on VAE architectures</article-title>. <source>Phys. Med. Biol.</source> <volume>69</volume> (<issue>24</issue>), <fpage>245019</fpage>. <pub-id pub-id-type="doi">10.1088/1361-6560/ad9133</pub-id>
<pub-id pub-id-type="pmid">39527911</pub-id>
</mixed-citation>
</ref>
<ref id="B6">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Koloi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Loukas</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Hourican</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Sakellarios</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Quax</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Mishra</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Predicting early-stage coronary artery disease using machine learning and routine clinical biomarkers improved by augmented virtual data</article-title>. <source>Eur. Heart J. - Digital Health</source> <volume>5</volume> (<issue>5</issue>), <fpage>542</fpage>&#x2013;<lpage>550</lpage>. <pub-id pub-id-type="doi">10.1093/ehjdh/ztae049</pub-id>
<pub-id pub-id-type="pmid">39318697</pub-id>
</mixed-citation>
</ref>
<ref id="B7">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kumar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dhanka</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Khan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Maini</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Hybrid machine learning techniques based on genetic algorithm for heart disease detection</article-title>. <source>Innovation Emerg. Technol.</source> <volume>11</volume>, <fpage>2450008</fpage>. <pub-id pub-id-type="doi">10.1142/s2737599424500087</pub-id>
</mixed-citation>
</ref>
<ref id="B8">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kundrick</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Naniwadekar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Singla</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Kancharla</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Bhonsale</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Voigt</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Machine learning applied to wearable fitness tracker data and the risk of hospitalizations and cardiovascular events</article-title>. <source>Am. J. Prev. Cardiol.</source> <volume>22</volume>, <fpage>101006</fpage>. <pub-id pub-id-type="doi">10.1016/j.ajpc.2025.101006</pub-id>
<pub-id pub-id-type="pmid">40496758</pub-id>
</mixed-citation>
</ref>
<ref id="B9">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kuo</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Novel multiple Z-score models for detection of coronary artery dilation: application in Kawasaki disease</article-title>. <source>Pediatr. Rheumatol.</source> <volume>22</volume> (<issue>1</issue>), <fpage>108</fpage>. <pub-id pub-id-type="doi">10.1186/s12969-024-01040-9</pub-id>
<pub-id pub-id-type="pmid">39709497</pub-id>
</mixed-citation>
</ref>
<ref id="B10">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ladak</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gallagher</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Hasan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Awais</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Abdullah</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gullick</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Health-related quality of life in adult CHD surgical patients in a low middle-income country: a mixed-methods study</article-title>. <source>Cardiol. Young</source> <volume>30</volume> (<issue>8</issue>), <fpage>1126</fpage>&#x2013;<lpage>1137</lpage>. <pub-id pub-id-type="doi">10.1017/S1047951120001663</pub-id>
<pub-id pub-id-type="pmid">32633708</pub-id>
</mixed-citation>
</ref>
<ref id="B11">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Pang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>BioSeq-BLM: a platform for analyzing DNA, RNA, and protein sequences based on biological language models</article-title>. <source>Nucleic Acids Res.</source> <volume>49</volume> (<issue>22</issue>), <fpage>e129</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkab829</pub-id>
<pub-id pub-id-type="pmid">34581805</pub-id>
</mixed-citation>
</ref>
<ref id="B12">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>BioSeq-Analysis2.0: an updated platform for analyzing DNA, RNA and protein sequences at sequence level and residue level based on machine learning approaches</article-title>. <source>Nucleic Acids Res.</source> <volume>47</volume> (<issue>20</issue>), <fpage>e127</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkz740</pub-id>
<pub-id pub-id-type="pmid">31504851</pub-id>
</mixed-citation>
</ref>
<ref id="B13">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Min</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Andreini</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Pontone</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Guglielmo</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bax</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Coronary CTA plaque volume severity stages according to invasive coronary angiography and FFR</article-title>. <source>J. Cardiovasc. Comput. Tomogr.</source> <volume>16</volume> (<issue>5</issue>), <fpage>415</fpage>&#x2013;<lpage>422</lpage>. <pub-id pub-id-type="doi">10.1016/j.jcct.2022.03.001</pub-id>
<pub-id pub-id-type="pmid">35379596</pub-id>
</mixed-citation>
</ref>
<ref id="B14">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nayak</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Venugopala</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Ashwini</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>A systematic review on generative adversarial network (GAN): challenges and future directions</article-title>. <source>Archives Comput. Methods Eng.</source> <volume>31</volume> (<issue>8</issue>), <fpage>4739</fpage>&#x2013;<lpage>4772</lpage>. <pub-id pub-id-type="doi">10.1007/s11831-024-10119-1</pub-id>
</mixed-citation>
</ref>
<ref id="B15">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nenova</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Shang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Chronic disease progression prediction: leveraging case-based reasoning and big data analytics</article-title>. <source>Prod. Operations Manag.</source> <volume>31</volume> (<issue>1</issue>), <fpage>259</fpage>&#x2013;<lpage>280</lpage>. <pub-id pub-id-type="doi">10.1111/poms.13532</pub-id>
</mixed-citation>
</ref>
<ref id="B16">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nollmann</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Moskorz</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wimmenauer</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>J&#xe4;ger</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Cadeddu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Timm</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Characterization of CD34&#x2b; cells from patients with acute myeloid leukemia (AML) and myelodysplastic syndromes (MDS) using a t-Distributed stochastic neighbor embedding (t-SNE) protocol</article-title>. <source>Cancers</source> <volume>16</volume> (<issue>7</issue>), <fpage>1320</fpage>. <pub-id pub-id-type="doi">10.3390/cancers16071320</pub-id>
<pub-id pub-id-type="pmid">38610998</pub-id>
</mixed-citation>
</ref>
<ref id="B17">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pickles</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Keller</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>The economic burden of complex CHD in the United States</article-title>. <source>Cardiol. Young</source> <volume>35</volume> (<issue>9</issue>), <fpage>1751</fpage>&#x2013;<lpage>1758</lpage>. <pub-id pub-id-type="doi">10.1017/S1047951125109256</pub-id>
<pub-id pub-id-type="pmid">40908927</pub-id>
</mixed-citation>
</ref>
<ref id="B18">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qiao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Towards retraining-free RNA modification prediction with incremental learning</article-title>. <source>Inf. Sci.</source> <volume>660</volume>, <fpage>120105</fpage>. <pub-id pub-id-type="doi">10.1016/j.ins.2024.120105</pub-id>
</mixed-citation>
</ref>
<ref id="B19">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qiao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Teng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>A self-conformation-aware pre-training framework for molecular property prediction with substructure interpretability</article-title>. <source>Nat. Commun.</source> <volume>16</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>16</lpage>. <pub-id pub-id-type="doi">10.1038/s41467-025-59634-0</pub-id>
<pub-id pub-id-type="pmid">40355450</pub-id>
</mixed-citation>
</ref>
<ref id="B20">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rehman</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Naseem</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Butt</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mahmood</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Khan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Khan</surname>
<given-names>I.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Predicting coronary heart disease with advanced machine learning classifiers for improved cardiovascular risk assessment</article-title>. <source>Sci. Rep.</source> <volume>15</volume> (<issue>1</issue>), <fpage>13361</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-025-96437-1</pub-id>
<pub-id pub-id-type="pmid">40247042</pub-id>
</mixed-citation>
</ref>
<ref id="B21">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rimal</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sharma</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Hyperparameter optimization: a comparative machine learning model analysis for enhanced heart disease prediction accuracy</article-title>. <source>Multimedia Tools Appl.</source> <volume>83</volume>, <fpage>55091</fpage>&#x2013;<lpage>55107</lpage>. <pub-id pub-id-type="doi">10.1007/s11042-023-17273-x</pub-id>
</mixed-citation>
</ref>
<ref id="B22">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sanchez-Morillo</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Le&#xf3;n-Jim&#xe9;nez</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Guerrero-Chanivet</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Jim&#xe9;nez-G&#xf3;mez</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Hidalgo-Molina</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Campos-Caro</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Integrating routine blood biomarkers and artificial intelligence for supporting diagnosis of silicosis in engineered stone workers</article-title>. <source>Bioeng. and Transl. Med.</source> <volume>9</volume> (<issue>6</issue>), <fpage>e10694</fpage>. <pub-id pub-id-type="doi">10.1002/btm2.10694</pub-id>
<pub-id pub-id-type="pmid">39545094</pub-id>
</mixed-citation>
</ref>
<ref id="B23">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sawaf</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Swed</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Alibrahim</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Bohsas</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Dave</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Nasif</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Triglyceride-glucose index as predictor for hypertension, CHD and STROKE risk among non-diabetic patients: a NHANES cross-sectional study 2001-2020</article-title>. <source>J. Epidemiol. Glob. Health</source> <volume>14</volume> (<issue>3</issue>), <fpage>1152</fpage>&#x2013;<lpage>1166</lpage>. <pub-id pub-id-type="doi">10.1007/s44197-024-00269-7</pub-id>
<pub-id pub-id-type="pmid">38954387</pub-id>
</mixed-citation>
</ref>
<ref id="B24">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Qin</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>DAE-Net: dual attention mechanism and edge supervision network for image manipulation detection and localization</article-title>. <source>Ieee Trans. Instrum. Meas.</source> <volume>73</volume>, <fpage>1</fpage>&#x2013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.1109/tim.2024.3451570</pub-id>
</mixed-citation>
</ref>
<ref id="B25">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Topranin</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Wiig-Fisketjon</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Botten</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Dalen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Langaas</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bye</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Sex-specific cardiovascular disease risk prediction using statistical learning and explainable artificial intelligence: the HUNT study</article-title>. <source>Eur. J. Prev. Cardiol.</source>, <fpage>zwaf135</fpage>. <pub-id pub-id-type="doi">10.1093/eurjpc/zwaf135</pub-id>
<pub-id pub-id-type="pmid">40063873</pub-id>
</mixed-citation>
</ref>
<ref id="B26">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Varol</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Omurlu</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Ture</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Simulation comparison of the effects of missing data imputation methods on classification performance in high dimensional data</article-title>. <source>Commun. Statistics-Simulation Comput.</source>, <fpage>1</fpage>&#x2013;<lpage>20</lpage>. <pub-id pub-id-type="doi">10.1080/03610918.2025.2524548</pub-id>
</mixed-citation>
</ref>
<ref id="B27">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhai</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>SBSM-Pro: support bio-sequence machine for proteins</article-title>. <source>Sci. China Inf. Sci.</source> <volume>67</volume> (<issue>11</issue>), <fpage>212106</fpage>. <pub-id pub-id-type="doi">10.1007/s11432-024-4171-9</pub-id>
</mixed-citation>
</ref>
<ref id="B28">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Qian</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Structured sparse regularization-based deep fuzzy networks for RNA N6-Methyladenosine sites prediction</article-title>. <source>IEEE Trans. Fuzzy Syst.</source> <volume>33</volume> (<issue>1</issue>), <fpage>131</fpage>&#x2013;<lpage>144</lpage>. <pub-id pub-id-type="doi">10.1109/tfuzz.2024.3428402</pub-id>
</mixed-citation>
</ref>
<ref id="B29">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Quorum sensing-based interactions among drugs, microbes, and diseases</article-title>. <source>Sci. China-Life Sci.</source> <volume>66</volume> (<issue>1</issue>), <fpage>137</fpage>&#x2013;<lpage>151</lpage>. <pub-id pub-id-type="doi">10.1007/s11427-021-2121-0</pub-id>
<pub-id pub-id-type="pmid">35933489</pub-id>
</mixed-citation>
</ref>
<ref id="B30">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xie</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Qian</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Methyl-GP: accurate generic DNA methylation prediction based on a language model and representation learning</article-title>. <source>Nucleic Acids Res.</source> <volume>53</volume> (<issue>6</issue>), <fpage>gkaf223</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkaf223</pub-id>
<pub-id pub-id-type="pmid">40156859</pub-id>
</mixed-citation>
</ref>
<ref id="B31">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xiong</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Ku</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Diagnostic performance of coronary computed tomography angiography stenosis score for coronary stenosis</article-title>. <source>BMC Med. Imaging</source> <volume>24</volume> (<issue>1</issue>), <fpage>39</fpage>. <pub-id pub-id-type="doi">10.1186/s12880-024-01213-8</pub-id>
<pub-id pub-id-type="pmid">38336622</pub-id>
</mixed-citation>
</ref>
<ref id="B32">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xing</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Self-supervised multi-level generative adversarial network data imputation algorithm</article-title>. <source>Int. J. Approx. Reason.</source> <volume>187</volume>, <fpage>109553</fpage>. <pub-id pub-id-type="doi">10.1016/j.ijar.2025.109553</pub-id>
</mixed-citation>
</ref>
<ref id="B33">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>Y. M.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>C. Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>T. Y.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>X. L.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>A gender specific risk assessment of coronary heart disease based on physical examination data</article-title>. <source>NPJ Digital Medicine</source> <volume>6</volume> (<issue>1</issue>), <fpage>136</fpage>. <pub-id pub-id-type="doi">10.1038/s41746-023-00887-8</pub-id>
<pub-id pub-id-type="pmid">37524859</pub-id>
</mixed-citation>
</ref>
<ref id="B34">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zaninotto</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Steptoe</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shim</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>CVD incidence and mortality among people with diabetes and/or hypertension: results from the English longitudinal study of ageing</article-title>. <source>Plos One</source> <volume>19</volume> (<issue>5</issue>), <fpage>e0303306</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0303306</pub-id>
<pub-id pub-id-type="pmid">38820248</pub-id>
</mixed-citation>
</ref>
<ref id="B35">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zeng</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Colorectal liver metastasis pathomics model (CLMPM): integrating single cell and spatial transcriptome analysis with pathomics for predicting liver metastasis in colorectal cancer</article-title>. <source>Mod. Pathology</source> <volume>38</volume>, <fpage>100805</fpage>. <pub-id pub-id-type="doi">10.1016/j.modpat.2025.100805</pub-id>
<pub-id pub-id-type="pmid">40473111</pub-id>
</mixed-citation>
</ref>
<ref id="B36">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zulfiqar</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ahmad</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Ahmed</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Deep-STP: a deep learning-based approach to predict snake toxin proteins by using word embeddings</article-title>. <source>Front. Med.</source> <volume>10</volume>, <fpage>1291352</fpage>. <pub-id pub-id-type="doi">10.3389/fmed.2023.1291352</pub-id>
<pub-id pub-id-type="pmid">38298505</pub-id>
</mixed-citation>
</ref>
</ref-list>
</back>
</article>