<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Endocrinol.</journal-id>
<journal-title>Frontiers in Endocrinology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Endocrinol.</abbrev-journal-title>
<issn pub-type="epub">1664-2392</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fendo.2024.1376220</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Endocrinology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Identifying diagnostic indicators for type 2 diabetes mellitus from physical examination using interpretable machine learning approach</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Lv</surname>
<given-names>Xiang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2639997"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Luo</surname>
<given-names>Jiesi</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1704429"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Huang</surname>
<given-names>Wei</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/974259"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Guo</surname>
<given-names>Hui</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Bai</surname>
<given-names>Xue</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yan</surname>
<given-names>Pijun</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1822708"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Jiang</surname>
<given-names>Zongzhe</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2158928"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Yonglin</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1612741"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Jing</surname>
<given-names>Runyu</given-names>
</name>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/709478"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Chen</surname>
<given-names>Qi</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="aff" rid="aff7">
<sup>7</sup>
</xref>
<xref ref-type="aff" rid="aff8">
<sup>8</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Li</surname>
<given-names>Menglong</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/731380"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>College of Chemistry, Sichuan University</institution>, <addr-line>Chengdu</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Basic Medical College, Southwest Medical University</institution>, <addr-line>Luzhou</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Endocrinology and Metabolism, The Affiliated Hospital of Southwest Medical University</institution>, <addr-line>Luzhou</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Metabolic Vascular Disease Key Laboratoryof Sichuan Province, The Affiliated Hospital of Southwest Medical University</institution>, <addr-line>Luzhou</addr-line>, <country>China</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Department of Pharmacy, The Affiliated Hospital of North Sichuan Medical College</institution>, <addr-line>Nanchong</addr-line>, <country>China</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>School of Cyber Science and Engineering, Sichuan University</institution>, <addr-line>Chengdu</addr-line>, <country>China</country>
</aff>
<aff id="aff7">
<sup>7</sup>
<institution>Department of Nursing, The Affiliated Hospital of Southwest Medical University</institution>, <addr-line>Luzhou</addr-line>, <country>China</country>
</aff>
<aff id="aff8">
<sup>8</sup>
<institution>School of Nursing, Southwest Medical University</institution>, <addr-line>Luzhou</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Yun Shen, Pennington Biomedical Research Center, United States</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Md. Mehedi Hassan, Khulna University, Bangladesh</p>
<p>Lixia Zhang, Pennington Biomedical Research Center, United States</p>
<p>Laboni Akter, Khulna University of Engineering &amp; Technology, Bangladesh</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Runyu Jing, <email xlink:href="mailto:jingry@scu.edu.cn">jingry@scu.edu.cn</email>; Qi Chen, <email xlink:href="mailto:qichen@swmu.edu.cn">qichen@swmu.edu.cn</email>; Menglong Li, <email xlink:href="mailto:liml@scu.edu.cn">liml@scu.edu.cn</email>
</p>
</fn>
<fn fn-type="equal" id="fn003">
<p>&#x2020;These authors have contributed equally to this work</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>18</day>
<month>03</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1376220</elocation-id>
<history>
<date date-type="received">
<day>25</day>
<month>01</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>29</day>
<month>02</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Lv, Luo, Huang, Guo, Bai, Yan, Jiang, Zhang, Jing, Chen and Li</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Lv, Luo, Huang, Guo, Bai, Yan, Jiang, Zhang, Jing, Chen and Li</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Background</title>
<p>Identification of patients at risk for type 2 diabetes mellitus (T2DM) can not only prevent complications and reduce suffering but also ease the health care burden. While routine physical examination can provide useful information for diagnosis, manual exploration of routine physical examination records is not feasible due to the high prevalence of T2DM.</p>
</sec>
<sec>
<title>Objectives</title>
<p>We aim to build interpretable machine learning models for T2DM diagnosis and uncover important diagnostic indicators from physical examination, including age- and sex-related indicators.</p>
</sec>
<sec>
<title>Methods</title>
<p>In this study, we present three weighted diversity density (WDD)-based algorithms for T2DM screening that use physical examination indicators, the algorithms are highly transparent and interpretable, two of which are missing value tolerant algorithms.</p>
</sec>
<sec>
<title>Patients</title>
<p>Regarding the dataset, we collected 43 physical examination indicator data from 11,071 cases of T2DM patients and 126,622 healthy controls at the Affiliated Hospital of Southwest Medical University. After data processing, we used a data matrix containing 16004 EHRs and 43 clinical indicators for modelling.</p>
</sec>
<sec>
<title>Results</title>
<p>The indicators were ranked according to their model weights, and the top 25% of indicators were found to be directly or indirectly related to T2DM. We further investigated the clinical characteristics of different age and sex groups, and found that the algorithms can detect relevant indicators specific to these groups. The algorithms performed well in T2DM screening, with the highest area under the receiver operating characteristic curve (AUC) reaching 0.9185.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>This work utilized the interpretable WDD-based algorithms to construct T2DM diagnostic models based on physical examination indicators. By modeling data grouped by age and sex, we identified several predictive markers related to age and sex, uncovering characteristic differences among various groups of T2DM patients.</p>
</sec>
</abstract>
<kwd-group>
<kwd>diabetes</kwd>
<kwd>diabetes diagnosis</kwd>
<kwd>diabetic prediction</kwd>
<kwd>diagnostic indicator</kwd>
<kwd>health informatics</kwd>
<kwd>interpretable machine learning</kwd>
</kwd-group>
<counts>
<fig-count count="6"/>
<table-count count="2"/>
<equation-count count="22"/>
<ref-count count="50"/>
<page-count count="15"/>
<word-count count="7720"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Clinical Diabetes</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Type 2 diabetes mellitus (T2DM) is the most common type of diabetes mellitus (DM), whose pathogenesis is that the cells in the body are not sensitive to insulin, meaning they do not respond to insulin (<xref ref-type="bibr" rid="B1">1</xref>). A longer disease duration of diabetes often leads to a variety of complications, such as retinopathy (<xref ref-type="bibr" rid="B2">2</xref>), cardiovascular disease, stroke (<xref ref-type="bibr" rid="B3">3</xref>, <xref ref-type="bibr" rid="B4">4</xref>), and diabetic foot (<xref ref-type="bibr" rid="B5">5</xref>). It is estimated that about half of T2DM patients do not know they have diabetes (44.7%) (<xref ref-type="bibr" rid="B6">6</xref>). Therefore, screening for T2DM is essential to prevent or delay complications, avoid premature death, and improve quality of life.</p>
<p>Manual review of a large amount of clinical data is time-consuming and laborious, and missed diagnosis will be inevitable (<xref ref-type="bibr" rid="B6">6</xref>, <xref ref-type="bibr" rid="B7">7</xref>). Thus, leveraging machine learning for T2DM screening has emerged as a notable approach in auxiliary diagnostics, enhancing both the accuracy and efficiency of diagnoses. Currently, machine learning models such as the random forest (RF) (<xref ref-type="bibr" rid="B8">8</xref>&#x2013;<xref ref-type="bibr" rid="B10">10</xref>), support vector machine (SVM) (<xref ref-type="bibr" rid="B8">8</xref>, <xref ref-type="bibr" rid="B11">11</xref>), logistic regression (LR) (<xref ref-type="bibr" rid="B11">11</xref>&#x2013;<xref ref-type="bibr" rid="B13">13</xref>), and eXtreme gradient boosting (XGBoost) (<xref ref-type="bibr" rid="B9">9</xref>, <xref ref-type="bibr" rid="B14">14</xref>) have been developed for constructing accurate system of T2DM prediction. Some studies have also employed machine learning techniques to identify indicators associated with T2DM, such as the white blood cell (WBC) (<xref ref-type="bibr" rid="B15">15</xref>), urinary and dietary metal exposure (<xref ref-type="bibr" rid="B16">16</xref>) and serum calcium (<xref ref-type="bibr" rid="B17">17</xref>). These works demonstrate the effectiveness of machine&#xa0;learning in predicting T2DM and identifying relevant indicator information.</p>
<p>For the construction of T2DM diagnostic models, the existing problems are as follows: (I) The effective extraction of T2DM diagnostic indicators through machine learning often relies on their interpretability (<xref ref-type="bibr" rid="B10">10</xref>, <xref ref-type="bibr" rid="B12">12</xref>, <xref ref-type="bibr" rid="B18">18</xref>&#x2013;<xref ref-type="bibr" rid="B23">23</xref>). However, some of the current work lacks evaluation of important indicators, and some rely on third-party tools such as Shapley Additive exPlanations (SHAP) and Local Interpretable Model-agnostic Explanations (LIME) (<xref ref-type="bibr" rid="B8">8</xref>, <xref ref-type="bibr" rid="B9">9</xref>, <xref ref-type="bibr" rid="B11">11</xref>), which may bring potential deviation in clinical understanding (<xref ref-type="bibr" rid="B24">24</xref>). (II) The clinical indicators and datasets are critical to whether a model can be used in practice. At present, the frequently used diabetes dataset public like PIMA Indian dataset contains only 8 clinical indicators (<xref ref-type="bibr" rid="B25">25</xref>), and unconventional indicators using in some works are often difficult to obtain in community hospitals. Therefore, it is valuable to use physical examination indicators for T2DM prediction. (III) The problem of missing values in EHRs is unavoidable during data analysis. Currently, methods based on data imputation often require exhaustive searching (<xref ref-type="bibr" rid="B26">26</xref>). How to handle these missing values reasonably and efficiently is a matter that needs consideration.</p>
<p>To this end, we introduced three weighted diversity density (WDD)-based algorithms with a focus on intrinsic interpretability, which two of the algorithms could &#x2018;tolerate&#x2019; missing value by adding penalty terms. By applying these algorithms to physical examination data for T2DM, we identified several clinical indicators related to T2DM diagnosis, including age-related markers like glomerular filtration rate (GFR) and triglycerides (TG). Additionally, by analyzing the model&#x2019;s internal parameters, we can gain a better understanding of the clinical indicators the model relies on for predictions, without the need for third-party interpretability tools.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>Dataset summary</title>
<p>All electronic health record (EHR) data came from the Affiliated Hospital of Southwest Medical University. A total of 16,004 EHRs and 43 usable physical examination indicators were screened out, of which half explicitly contained information about a confirmed T2DM diagnosis (<xref ref-type="fig" rid="f1">
<bold>Figures&#xa0;1</bold>
</xref>, <xref ref-type="fig" rid="f2">
<bold>2</bold>
</xref>; <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>). In order to capture characteristics of early-stage T2DM, the EHRs of T2DM patients were limited to their first record in the hospital system. The physical examination indicators could be divided into three categories: routine urine indicators (9 indicators), blood cell analysis indicators (24 indicators), and biochemical indicators (10 indicators) (<xref ref-type="fig" rid="f1">
<bold>Figures&#xa0;1</bold>
</xref>, <xref ref-type="fig" rid="f2">
<bold>2</bold>
</xref>), the name and the abbreviation of the indicators were shown in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S1</bold>
</xref>.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Overview of the study design. <bold>(A)</bold> The workflow of this work. <bold>(B)</bold> The first two steps in A, 11071 T2DM electronic health records (EHRs) and 126622 physical examination EHRs were collected. After preprocessing, 16004 EHRs were selected to build models for T2DM prediction. <bold>(C)</bold> The last two steps in A and the basic principle of the weighted diversity density method.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fendo-15-1376220-g001.tif"/>
</fig>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Flowchart of inclusion and exclusion criteria for the study populations of patients with type 2 diabetes mellitus (T2DM) and the physical examination population. We only use the first electronic health records for each patient in the hospital system.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fendo-15-1376220-g002.tif"/>
</fig>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Characteristics of the study population.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left"/>
<th valign="middle" align="left">Normal people</th>
<th valign="middle" align="left">T2DM patients</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">TC, mmol/L</td>
<td valign="middle" align="left">4.96 (0.932)</td>
<td valign="middle" align="left">4.63 (1.393)</td>
</tr>
<tr>
<td valign="middle" align="left">AST, U/L</td>
<td valign="middle" align="left">24.58 (12.877)</td>
<td valign="middle" align="left">25.20 (37.516)</td>
</tr>
<tr>
<td valign="middle" align="left">GFR, mL/min</td>
<td valign="middle" align="left">94.01 (14.660)</td>
<td valign="middle" align="left">87.98 (29.298)</td>
</tr>
<tr>
<td valign="middle" align="left">Crea, umol/L</td>
<td valign="middle" align="left">68.48 (20.157)</td>
<td valign="middle" align="left">86.93 (80.094)</td>
</tr>
<tr>
<td valign="middle" align="left">HDL-C, mmol/L</td>
<td valign="middle" align="left">1.40 (0.368)</td>
<td valign="middle" align="left">1.14 (0.369)</td>
</tr>
<tr>
<td valign="middle" align="left">TG, mmol/L</td>
<td valign="middle" align="left">1.61 (1.313)</td>
<td valign="middle" align="left">2.28 (2.377)</td>
</tr>
<tr>
<td valign="middle" align="left">LDL-C, mmol/L</td>
<td valign="middle" align="left">3.09 (0.900)</td>
<td valign="middle" align="left">2.71 (1.014)</td>
</tr>
<tr>
<td valign="middle" align="left">ALT, U/L</td>
<td valign="middle" align="left">24.78 (19.111)</td>
<td valign="middle" align="left">26.82 (32.369)</td>
</tr>
<tr>
<td valign="middle" align="left">GGT, U/L</td>
<td valign="middle" align="left">30.62 (40.086)</td>
<td valign="middle" align="left">49.49 (110.088)</td>
</tr>
<tr>
<td valign="middle" align="left">AST/ALT</td>
<td valign="middle" align="left">1.19 (0.547)</td>
<td valign="middle" align="left">1.18 (0.969)</td>
</tr>
<tr>
<td valign="middle" align="left">MUCUS,/uL</td>
<td valign="middle" align="left">12.88 (18.193)</td>
<td valign="middle" align="left">2.75 (12.213)</td>
</tr>
<tr>
<td valign="middle" align="left">BACT,/uL</td>
<td valign="middle" align="left">21.38 (366.085)</td>
<td valign="middle" align="left">1998.54 (7639.814)</td>
</tr>
<tr>
<td valign="middle" align="left">EC,/uL</td>
<td valign="middle" align="left">6.81 (25.506)</td>
<td valign="middle" align="left">5.28 (14.011)</td>
</tr>
<tr>
<td valign="middle" align="left">BLD</td>
<td valign="middle" align="left">0.31 (0.628)</td>
<td valign="middle" align="left">0.27 (0.603)</td>
</tr>
<tr>
<td valign="middle" align="left">U-SG</td>
<td valign="middle" align="left">1.02 (0.006)</td>
<td valign="middle" align="left">1.02 (0.008)</td>
</tr>
<tr>
<td valign="middle" align="left">U-pH</td>
<td valign="middle" align="left">5.88 (0.729)</td>
<td valign="middle" align="left">5.73 (0.719)</td>
</tr>
<tr>
<td valign="middle" align="left">Crystal,/uL</td>
<td valign="middle" align="left">6.09 (37.158)</td>
<td valign="middle" align="left">23.45 (178.737)</td>
</tr>
<tr>
<td valign="middle" align="left">RBC-Urine,/uL</td>
<td valign="middle" align="left">7.29 (165.853)</td>
<td valign="middle" align="left">20.96 (267.325)</td>
</tr>
<tr>
<td valign="middle" align="left">WBC-Urine,/uL</td>
<td valign="middle" align="left">14.43 (140.372)</td>
<td valign="middle" align="left">61.44 (640.552)</td>
</tr>
<tr>
<td valign="middle" align="left">NEU, 10E9/L</td>
<td valign="middle" align="left">3.64 (1.290)</td>
<td valign="middle" align="left">5.50 (3.591)</td>
</tr>
<tr>
<td valign="middle" align="left">NEU-R, %</td>
<td valign="middle" align="left">59.31 (8.511)</td>
<td valign="middle" align="left">68.59 (11.639)</td>
</tr>
<tr>
<td valign="middle" align="left">PCT, %</td>
<td valign="middle" align="left">0.22 (0.051)</td>
<td valign="middle" align="left">0.24 (0.086)</td>
</tr>
<tr>
<td valign="middle" align="left">PDW, %</td>
<td valign="middle" align="left">16.15 (1.010)</td>
<td valign="middle" align="left">15.48 (2.733)</td>
</tr>
<tr>
<td valign="middle" align="left">PLT, 10E9/L</td>
<td valign="middle" align="left">210.83 (56.276)</td>
<td valign="middle" align="left">206.96 (80.247)</td>
</tr>
<tr>
<td valign="middle" align="left">MPV, fL</td>
<td valign="middle" align="left">10.74 (1.412)</td>
<td valign="middle" align="left">11.44 (1.400)</td>
</tr>
<tr>
<td valign="middle" align="left">HGB, g/L</td>
<td valign="middle" align="left">142.27 (15.014)</td>
<td valign="middle" align="left">126.97 (21.740)</td>
</tr>
<tr>
<td valign="middle" align="left">EOS, 10E9/L</td>
<td valign="middle" align="left">0.16 (0.161)</td>
<td valign="middle" align="left">0.13 (0.172)</td>
</tr>
<tr>
<td valign="middle" align="left">EOS-R, %</td>
<td valign="middle" align="left">2.62 (2.317)</td>
<td valign="middle" align="left">1.99 (2.336)</td>
</tr>
<tr>
<td valign="middle" align="left">BASO, 10E9/L</td>
<td valign="middle" align="left">0.03 (0.018)</td>
<td valign="middle" align="left">0.02 (0.020)</td>
</tr>
<tr>
<td valign="middle" align="left">BASO-R, %</td>
<td valign="middle" align="left">0.53 (0.283)</td>
<td valign="middle" align="left">0.24 (0.258)</td>
</tr>
<tr>
<td valign="middle" align="left">LYM, 10E9/L</td>
<td valign="middle" align="left">1.88 (0.617)</td>
<td valign="middle" align="left">1.57 (0.665)</td>
</tr>
<tr>
<td valign="middle" align="left">LYM-R, %</td>
<td valign="middle" align="left">31.65 (7.991)</td>
<td valign="middle" align="left">23.10 (10.179)</td>
</tr>
<tr>
<td valign="middle" align="left">HCT</td>
<td valign="middle" align="left">0.44 (0.041)</td>
<td valign="middle" align="left">0.39 (0.063)</td>
</tr>
<tr>
<td valign="middle" align="left">RDW-SD, fL</td>
<td valign="middle" align="left">43.18 (2.994)</td>
<td valign="middle" align="left">42.91 (4.282)</td>
</tr>
<tr>
<td valign="middle" align="left">RDW-CV, %</td>
<td valign="middle" align="left">13.15 (0.975)</td>
<td valign="middle" align="left">13.34 (1.305)</td>
</tr>
<tr>
<td valign="middle" align="left">RBC, 10E12/L</td>
<td valign="middle" align="left">4.66 (0.489)</td>
<td valign="middle" align="left">4.33 (0.732)</td>
</tr>
<tr>
<td valign="middle" align="left">MCHC, g/L</td>
<td valign="middle" align="left">325.53 (8.441)</td>
<td valign="middle" align="left">328.30 (14.113)</td>
</tr>
<tr>
<td valign="middle" align="left">MCH, pg</td>
<td valign="middle" align="left">30.62 (2.417)</td>
<td valign="middle" align="left">29.40 (2.533)</td>
</tr>
<tr>
<td valign="middle" align="left">MCV, fL</td>
<td valign="middle" align="left">94.00 (6.445)</td>
<td valign="middle" align="left">89.55 (6.784)</td>
</tr>
<tr>
<td valign="middle" align="left">MONO, 10E9/L</td>
<td valign="middle" align="left">0.35 (0.122)</td>
<td valign="middle" align="left">0.45 (0.257)</td>
</tr>
<tr>
<td valign="middle" align="left">MONO-R, %</td>
<td valign="middle" align="left">5.88 (1.491)</td>
<td valign="middle" align="left">6.12 (2.300)</td>
</tr>
<tr>
<td valign="middle" align="left">P-LCR</td>
<td valign="middle" align="left">31.37 (9.845)</td>
<td valign="middle" align="left">36.20 (10.018)</td>
</tr>
<tr>
<td valign="middle" align="left">WBC, 10E9/L</td>
<td valign="middle" align="left">6.08 (2.075)</td>
<td valign="middle" align="left">7.68 (3.860)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Data are mean (standard deviation).</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>For these datasets divided according to the physical examination items, to facilitate their description here, we chose some standardized abbreviations: Whole physical examination indicators dataset (PEI dataset), Blood cell analysis dataset (BCA dataset), Urinalysis dataset (Uri dataset), Biochemical dataset (BioChem dataset).</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Data preprocessing</title>
<p>On the collected data, three steps were performed before modelling:</p>
<list list-type="simple">
<list-item>
<p>(1) First, we retained only the physical examination indicators for exploring the association of these indicators with T2DM diagnosis. In total, 181 features, of which 52 are physical examination features, remained afterwards.</p>
</list-item>
<list-item>
<p>(2) The features were normalized by <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:msubsup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the average and standard deviation of the <inline-formula>
<mml:math display="inline" id="im4">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula> th record, respectively. To avoid the influence from outliers, a featurewise box-plot analysis was performed to remove the features located outside of <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1.5</mml:mn>
<mml:mi>I</mml:mi>
<mml:mi>Q</mml:mi>
<mml:mi>R</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>Q</mml:mi>
<mml:mn>3</mml:mn>
<mml:mo>+</mml:mo>
<mml:mn>1.5</mml:mn>
<mml:mi>I</mml:mi>
<mml:mi>Q</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> when calculating <inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. These outliers were replaced by null values. Any feature with a standard deviation of 0 (that is, the feature value is the same in all samples) was deleted, leaving 43 dimensions of physical examination features.</p>
</list-item>
<list-item>
<p>(3) When using the WDD-KNN algorithm for modelling, we used the K-nearest neighbour (KNN) imputer to impute the missing values of the data (n_neighbours = 15) and then normalized the processed data again in step (2).</p>
</list-item>
</list>
<p>The missing rate of each feature is shown in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S3</bold>
</xref>. When using the MVT-WDD-DI and MVT-WDD-BF algorithms for modelling, we kept the missing value of each dimension feature of positive and negative samples consistent to eliminate bias. (Details in the <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Note 4</bold>
</xref>: Biased distribution misleading model using features).</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Preliminary of weighted diversity density algorithm</title>
<p>The WDD algorithm, initially proposed by Maron for multi-instance learning problems (<xref ref-type="bibr" rid="B27">27</xref>), utilizes radial basis distance metrics to measure classification probabilities. We developed three new algorithms based on WDD in this work. Its non-linear nature and transparent framework make it particularly suitable for the medical field, where high interpretability in models is essential. Building upon the WDD framework, we have made enhancements to adapt it for medical classification problems, ensuring it can effectively handle missing values. Additionally, another reason for selecting the WDD method is its capability to provide a risk score for each sample, similar to the diagnostic approach of a clinical physician. This is achieved through its internal distance function, rather than merely outputting a categorical label. By decomposing the distance function at the feature level, we are further able to extract the feature-based criteria on which the model relies. This significantly enhances the transparency of the model. Unlike traditional models that offer limited insight into their decision-making process, WDD allows for a deeper understanding of how and why certain diagnostic conclusions are reached. This alignment with clinical practices not only aids in the interpretability of the model but also fosters greater trust and reliability in its application in medical settings. The ability to dissect the model&#x2019;s reasoning at a feature level offers invaluable insights into the diagnostic criteria, bridging the gap between machine learning outputs and clinical decision-making.</p>
<p>The major principle of the WDD algorithm (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1C</bold>
</xref>) is to find an optimal point <inline-formula>
<mml:math display="inline" id="im8">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>f</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> in the data space that maximizes the probability density that positive samples (<inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> = <inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>+</mml:mo>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mo>+</mml:mo>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mo>+</mml:mo>
</mml:msubsup>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>) are near this point and minimizes the probability density that negative samples (<inline-formula>
<mml:math display="inline" id="im11">
<mml:mrow>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> = <inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
</mml:msubsup>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>) are near it, where <inline-formula>
<mml:math display="inline" id="im13">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula> is the index of the sample, <inline-formula>
<mml:math display="inline" id="im14">
<mml:mi>f</mml:mi>
</mml:math>
</inline-formula> is the number of features. The modified WDD algorithm can be represented as:</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:mi>arg</mml:mi>
<mml:munder>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:mi>x</mml:mi>
</mml:munder>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mo>&#x220f;</mml:mo>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi>Pr</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>|</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mo>+</mml:mo>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
<mml:msub>
<mml:mo>&#x220f;</mml:mo>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi>Pr</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>|</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</disp-formula>
<p>
<inline-formula>
<mml:math display="inline" id="im15">
<mml:mi>t</mml:mi>
</mml:math>
</inline-formula> is the target point, <inline-formula>
<mml:math display="inline" id="im16">
<mml:mrow>
<mml:mi>Pr</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>|</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is the probability density that positive samples are near this point, and <inline-formula>
<mml:math display="inline" id="im17">
<mml:mrow>
<mml:mi>Pr</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>|</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is the probability density of negative samples. The implicit functions could be given as:</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mi>Pr</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>|</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mstyle mathvariant="normal" mathsize="normal">
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:mi>Pr</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>|</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mstyle mathvariant="normal" mathsize="normal">
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where distance function (Dist) for both positive and negative samples is defined as the sum of weighted squares:</p>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:mstyle mathvariant="normal" mathsize="normal">
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mstyle mathvariant="normal" mathsize="normal">
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
</mml:mstyle>
<mml:msub>
<mml:mstyle mathvariant="normal" mathsize="normal">
<mml:mi>t</mml:mi>
</mml:mstyle>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:mstyle mathvariant="normal" mathsize="normal">
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
</mml:mstyle>
<mml:msub>
<mml:mstyle mathvariant="normal" mathsize="normal">
<mml:mi>t</mml:mi>
</mml:mstyle>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where k is the index of the feature.</p>
<p>The formulas yield a function to measure an optimal point based on both positive and negative samples, and the position of the point can be optimized during a deep learning process.</p>
<p>In this work, based on the idea of Maron&#x2019;s work, we modified the distance function to achieve our predictive goals. One of the algorithms first imputes the data with missing values by k-nearest neighbour (KNN) and then uses WDD for prediction. The other two algorithms, named MVT-WDD-DI and MVT-WDD-BF, do not need to impute data but use penalty mechanisms for missing value tolerance.</p>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Our proposed modified WDD algorithm</title>
<p>As mentioned above, we modified the WDD algorithm proposed in Maron&#x2019;s work to improve the algorithms&#x2019; ability to handle data containing missing values. The modifications are introduced in the following subsections:</p>
<sec id="s2_4_1">
<label>2.4.1</label>
<title>WDD-KNN</title>
<p>The WDD-KNN algorithm uses data imputation by k-nearest neighbor (KNN) as input and then uses the modified WDD algorithm for modelling. Since all the missing values are imputed by KNN, we only modified <xref ref-type="disp-formula" rid="eq2">Equations (2)</xref> and <xref ref-type="disp-formula" rid="eq3">(3)</xref> by adding a hyperparameter <inline-formula>
<mml:math display="inline" id="im18">
<mml:mi>&#x3b3;</mml:mi>
</mml:math>
</inline-formula> to make the calculation more flexible:</p>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:mi>Pr</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>|</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mo mathvariant="bold-italic">+</mml:mo>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mstyle mathvariant="normal" mathsize="normal">
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mo mathvariant="bold-italic">+</mml:mo>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq7">
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:mi>Pr</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>|</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mo mathvariant="bold-italic">&#x2212;</mml:mo>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mstyle mathvariant="normal" mathsize="normal">
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mo mathvariant="bold-italic">&#x2212;</mml:mo>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Using WDD-KNN, we can classify a dataset containing missing values, but it still needs a step for missing value filling. This step limits the process of prediction. When given a new sample that contains a missing value, we need a dataset for imputing the missing value more than the trained model parameters. Therefore, we developed two missing value tolerant (MVT) algorithms for our predictive goals.</p>
</sec>
<sec id="s2_4_2">
<label>2.4.2</label>
<title>MVT-WDD-DI</title>
<p>We developed MVT-WDD-DI by adding penalty term for <xref ref-type="disp-formula" rid="eq6">Equations (6)</xref> and <xref ref-type="disp-formula" rid="eq7">(7)</xref> using division (DI) to handle missing values and finally obtain <xref ref-type="disp-formula" rid="eq8">Equations (8)</xref> and <xref ref-type="disp-formula" rid="eq9">(9)</xref>:</p>
<disp-formula id="eq8">
<label>(8)</label>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:mi>Pr</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>|</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mo mathvariant="bold-italic">+</mml:mo>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mstyle mathvariant="normal" mathsize="normal">
<mml:mi>&#x3b3;</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mstyle mathvariant="normal" mathsize="normal">
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mo mathvariant="bold-italic">+</mml:mo>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq9">
<label>(9)</label>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:mi>Pr</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>|</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mo mathvariant="bold-italic">&#x2212;</mml:mo>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mstyle mathvariant="normal" mathsize="normal">
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mo mathvariant="bold-italic">&#x2212;</mml:mo>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im19">
<mml:mi>f</mml:mi>
</mml:math>
</inline-formula> is the number of features and <inline-formula>
<mml:math display="inline" id="im20">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the number of missing values. Since the modification will influence the calculation of <xref ref-type="disp-formula" rid="eq6">Equations (6)</xref> and <xref ref-type="disp-formula" rid="eq7">(7)</xref>, we added a rule to <xref ref-type="disp-formula" rid="eq5">Equation (5)</xref>: <inline-formula>
<mml:math display="inline" id="im21">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>g</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. We also added a hyperparameter <inline-formula>
<mml:math display="inline" id="im22">
<mml:mi>&#x3b4;</mml:mi>
</mml:math>
</inline-formula> for adjusting the value of the penalty term. The concept behind this modification is to diminish the influence of samples containing missing values. Specifically, when a sample has numerous missing values, indicating lower data quality, we increase the penalty term to reduce its distance metric value. This approach, during the optimization process, results in the target point relying less on these data segments. The penalty term can reduce the diversity density according to the number of missing features and will not influence a sample containing all the features.</p>
</sec>
<sec id="s2_4_3">
<label>2.4.3</label>
<title>MVT-WDD-BF</title>
<p>In addition to using division to penalize missing values, we also tried to design another method by ignoring the missing feature (BF) of a sample. To make the idea work, we modified <xref ref-type="disp-formula" rid="eq2">Equations (2)</xref> through <xref ref-type="disp-formula" rid="eq5">(5)</xref> of the original WDD algorithm. First, <xref ref-type="disp-formula" rid="eq4">Equations (4)</xref> and <xref ref-type="disp-formula" rid="eq5">(5)</xref> were modified to</p>
<disp-formula id="eq11">
<label>(10)</label>
<mml:math display="block" id="M10">
<mml:mrow>
<mml:mstyle mathvariant="normal" mathsize="normal">
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
</mml:mstyle>
<mml:msup>
<mml:mstyle mathvariant="normal" mathsize="normal">
<mml:mi>t</mml:mi>
</mml:mstyle>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mstyle mathvariant="normal" mathsize="normal">
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
</mml:mstyle>
<mml:msubsup>
<mml:mstyle mathvariant="normal" mathsize="normal">
<mml:mi>t</mml:mi>
</mml:mstyle>
<mml:mi>k</mml:mi>
<mml:msup>
<mml:mo>&#x2032;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
</mml:msup>
</mml:msubsup>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq12">
<label>(11)</label>
<mml:math display="block" id="M11">
<mml:mrow>
<mml:mstyle mathvariant="normal" mathsize="normal">
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
</mml:mstyle>
<mml:msubsup>
<mml:mstyle mathvariant="normal" mathsize="normal">
<mml:mi>t</mml:mi>
</mml:mstyle>
<mml:mi>k</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In <xref ref-type="disp-formula" rid="eq11">Equation (11)</xref>, <inline-formula>
<mml:math display="inline" id="im23">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>g</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>exp</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. This modification allows the algorithm to &#x2018;ignore&#x2019; a missing feature when calculating. However, a reduced number of features will increase the diversity density based on the analysis of monotonicity. We modified <xref ref-type="disp-formula" rid="eq2">Equations (2)</xref> and <xref ref-type="disp-formula" rid="eq3">(3)</xref> to solve this problem:</p>
<disp-formula id="eq13">
<label>(12)</label>
<mml:math display="block" id="M12">
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi>Pr</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>|</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mo mathvariant="bold-italic">+</mml:mo>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>Pr</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>|</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mo mathvariant="bold-italic">+</mml:mo>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq14">
<label>(13)</label>
<mml:math display="block" id="M13">
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi>Pr</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>|</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mo mathvariant="bold-italic">&#x2212;</mml:mo>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>Pr</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>|</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mo mathvariant="bold-italic">&#x2212;</mml:mo>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In <xref ref-type="disp-formula" rid="eq12">Equations (12)</xref> and <xref ref-type="disp-formula" rid="eq13">(13)</xref>, we changed the monotonicity by removing the minuend  Equations &#x2018;1&#x2019; in the production term and added Euler&#x2019;s number as the base for scaling an excessively large negative number to the range <inline-formula>
<mml:math display="inline" id="im24">
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> after a practical test.</p>
</sec>
</sec>
<sec id="s2_5">
<label>2.5</label>
<title>Model setup</title>
<p>Since the goal of WDD is to find a point to maximize the diversity density, we used a deep learning framework for implementation. We used the Adam optimizer to solve the optimal problem, and the loss function of the three algorithms is shown in <xref ref-type="disp-formula" rid="eq14">Equation (14)</xref>:</p>
<disp-formula id="eq15">
<label>(14)</label>
<mml:math display="block" id="M14">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>=</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mstyle mathvariant="normal" mathsize="normal">
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mo>&#x220f;</mml:mo>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi>Pr</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>|</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mo mathvariant="bold-italic">+</mml:mo>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
<mml:msub>
<mml:mo>&#x220f;</mml:mo>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi>Pr</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="bold-italic">t</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>|</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mo mathvariant="bold-italic">&#x2212;</mml:mo>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</sec>
<sec id="s2_6">
<label>2.6</label>
<title>Model inference</title>
<p>After training, we obtained the optimal position <inline-formula>
<mml:math display="inline" id="im25">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> from the dataset, which optimized <xref ref-type="disp-formula" rid="eq1">Equation (1)</xref>. Thus, all the samples could be classified by calculating the maximum diversity density from its instances:</p>
<disp-formula id="eq16">
<label>(15)</label>
<mml:math display="block" id="M15">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>Dist</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Using <xref ref-type="disp-formula" rid="eq15">Equation (15)</xref> calculated diversity density, a sample could be assigned a label by setting a threshold. In this work, we used the method of analysing the ROC curve from the training set. First, we calculated all the diversity density values of the samples in the training dataset. Then, we calculated the false positive rate (FPR) and true positive rate (TPR), also called recall, under several different cut-off points and selected the best cut-off as the threshold when TPR-FPR reached its maximum The details are given in <xref ref-type="disp-formula" rid="eq16">Equations (16)</xref> - <xref ref-type="disp-formula" rid="eq18">(18)</xref>:</p>
<disp-formula id="eq17">
<label>(16)</label>
<mml:math display="block" id="M16">
<mml:mrow>
<mml:mtext>threshold&#xa0;</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mi>arg</mml:mi>
<mml:munder>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>R</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>R</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq18">
<label>(17)</label>
<mml:math display="block" id="M17">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>R</mml:mi>
<mml:mo>=</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo>+</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq19">
<label>(18)</label>
<mml:math display="block" id="M18">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>R</mml:mi>
<mml:mo>=</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mfrac>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo>+</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Similar to many other works, we employed the area under the receiver operating characteristic curve (AUC), accuracy (ACC), precision, recall, and F1 score as the metrics, calculated as <xref ref-type="disp-formula" rid="eq19">Equations (19)</xref> - <xref ref-type="disp-formula" rid="eq22">(22)</xref>:</p>
<disp-formula id="eq20">
<label>(19)</label>
<mml:math display="block" id="M19">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>=</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo>+</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq21">
<label>(20)</label>
<mml:math display="block" id="M20">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>=</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo>+</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq22">
<label>(21)</label>
<mml:math display="block" id="M21">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mo>=</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo>+</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq23">
<label>(22)</label>
<mml:math display="block" id="M22">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>=</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mn>2</mml:mn>
<mml:mo>&#x2217;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where TP, TN, FP and FN represent the number of true positives, true negatives, false positives and false negatives, respectively.</p>
<p>Additional details about the method, such as parameter tuning and training process, are provided in Supplemental information.</p>
</sec>
<sec id="s2_7">
<label>2.7</label>
<title>Basic workflow</title>
<p>In this study, we optimized the WDD model using gradient descent to identify an optimal point that is close to the distribution center of data from T2DM patients and far from the distribution center of data from normal individuals (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1C</bold>
</xref>). The <inline-formula>
<mml:math display="inline" id="im27">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> function is engineered to directly reflect the T2DM risk score, enabling the model to predict an input sample as T2DM if its risk score exceeds the risk threshold which is learned by the model. Through the analysis of learnable parameters within the <inline-formula>
<mml:math display="inline" id="im28">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> function, we have identified key features. Utilizing this characteristic, we have been able to uncover significant features across different age and gender groups, enhancing our understanding of T2DM risk factors.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Result</title>
<p>In the results section, we first introduce the performance scores of the model, confirming the consistency of its performance by repeating the modeling process 1000 times. More importantly, our discussion centers on the model&#x2019;s transparency and interpretability, aimed at extracting effective clinical information internally. This includes the identification of key diagnostic indicators and the interpretation of the associations between model parameters and prediction result. These sections together demonstrate the model&#x2019;s transparency and potential clinical utility, contributing useful perspectives for T2DM prediction and diagnosis.</p>
<sec id="s3_1">
<label>3.1</label>
<title>Performance of the prediction model</title>
<p>To the above four datasets, we applied three weighted diversity density (WDD)-based algorithms to construct diagnostic prediction models. WDD-KNN refers to an algorithm using k-nearest neighbor (KNN) for imputing missing values. The two MVT-WDD algorithms denote missing value tolerant (MVT) algorithms with penalty terms, where &#x2018;DI&#x2019; and &#x2018;BF&#x2019; represent two different methods of penalization. A total of 12 models were obtained. The models were evaluated by 10-fold cross-validation and independent test with 1000 repetitions (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>, cross-validation set: independent test set = 8:2, the independent test dataset was consistent for each repetition, details in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Note 1</bold>
</xref>). The results are shown in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>, the best AUC achieved 0.9185 ( &#xb1; 0.0035) on the whole PEI dataset, which proves the accuracy of the model.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Performance of algorithms on each dataset: Mean (Standard) of 1000 repetitions.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" colspan="6" align="left">10-fold cross-validation</th>
</tr>
</thead>
<tbody>
<tr>
<th valign="middle" colspan="6" align="center">PEI dataset</th>
</tr>
<tr>
<th valign="middle" align="center"/>
<th valign="middle" align="center">AUC</th>
<th valign="middle" align="center">ACC</th>
<th valign="middle" align="center">Precision</th>
<th valign="middle" align="center">Recall</th>
<th valign="middle" align="center">F1 score</th>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;WDD-KNN</td>
<td valign="middle" align="center">0.9185 (0.0034)</td>
<td valign="middle" align="center">0.8439 (0.0042)</td>
<td valign="middle" align="center">0.8761 (0.0049)</td>
<td valign="middle" align="center">0.8014 (0.0073)</td>
<td valign="middle" align="center">0.8368 (0.0047)</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;MVT-WDD-DI</td>
<td valign="middle" align="center">0.9130 (0.0088)</td>
<td valign="middle" align="center">0.8404 (0.0103)</td>
<td valign="middle" align="center">0.8726 (0.0106)</td>
<td valign="middle" align="center">0.7973 (0.0130)</td>
<td valign="middle" align="center">0.8329 (0.0111)</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;MVT-WDD-BF</td>
<td valign="middle" align="center">0.8882 (0.0096)</td>
<td valign="middle" align="center">0.8138 (0.0093)</td>
<td valign="middle" align="center">0.8291 (0.0096)</td>
<td valign="middle" align="center">0.7910 (0.0135)</td>
<td valign="middle" align="center">0.8091 (0.0103)</td>
</tr>
<tr>
<th valign="middle" colspan="6" align="center">BCA dataset</th>
</tr>
<tr>
<th valign="middle" align="center"/>
<th valign="middle" align="center">AUC</th>
<th valign="middle" align="center">ACC</th>
<th valign="middle" align="center">Precision</th>
<th valign="middle" align="center">Recall</th>
<th valign="middle" align="center">F1 score</th>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;WDD-KNN</td>
<td valign="middle" align="center">0.8770 (0.0018)</td>
<td valign="middle" align="center">0.7992 (0.0021)</td>
<td valign="middle" align="center">0.8386 (0.0039)</td>
<td valign="middle" align="center">0.7418 (0.0058)</td>
<td valign="middle" align="center">0.7868 (0.0028)</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;MVT-WDD-DI</td>
<td valign="middle" align="center">0.8530 (0.0045)</td>
<td valign="middle" align="center">0.7763 (0.0044)</td>
<td valign="middle" align="center">0.8097 (0.0062)</td>
<td valign="middle" align="center">0.7233 (0.0093)</td>
<td valign="middle" align="center">0.7635 (0.0054)</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;MVT-WDD-BF</td>
<td valign="middle" align="center">0.8910 (0.0094)</td>
<td valign="middle" align="center">0.8156 (0.0089)</td>
<td valign="middle" align="center">0.8379 (0.0081)</td>
<td valign="middle" align="center">0.7829 (0.0152)</td>
<td valign="middle" align="center">0.8087 (0.0109)</td>
</tr>
<tr>
<th valign="middle" colspan="6" align="center">Uri dataset</th>
</tr>
<tr>
<th valign="middle" align="center"/>
<th valign="middle" align="center">AUC</th>
<th valign="middle" align="center">ACC</th>
<th valign="middle" align="center">Precision</th>
<th valign="middle" align="center">Recall</th>
<th valign="middle" align="center">F1 score</th>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;WDD-KNN</td>
<td valign="middle" align="center">0.8442 (0.0069)</td>
<td valign="middle" align="center">0.7768 (0.0096)</td>
<td valign="middle" align="center">0.7621 (0.0133)</td>
<td valign="middle" align="center">0.8133 (0.0240)</td>
<td valign="middle" align="center">0.7837 (0.0109)</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;MVT-WDD-DI</td>
<td valign="middle" align="center">0.8985 (0.0074)</td>
<td valign="middle" align="center">0.8580 (0.0096)</td>
<td valign="middle" align="center">0.8463 (0.0109)</td>
<td valign="middle" align="center">0.8763 (0.0126)</td>
<td valign="middle" align="center">0.8604 (0.0101)</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;MVT-WDD-BF</td>
<td valign="middle" align="center">0.6414 (0.0175)</td>
<td valign="middle" align="center">0.6321 (0.0129)</td>
<td valign="middle" align="center">0.6993 (0.0203)</td>
<td valign="middle" align="center">0.4795 (0.0355)</td>
<td valign="middle" align="center">0.5580 (0.0265)</td>
</tr>
<tr>
<th valign="middle" colspan="6" align="center">BioChem dataset</th>
</tr>
<tr>
<th valign="middle" align="center"/>
<th valign="middle" align="center">AUC</th>
<th valign="middle" align="center">ACC</th>
<th valign="middle" align="center">Precision</th>
<th valign="middle" align="center">Recall</th>
<th valign="middle" align="center">F1 score</th>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;WDD-KNN</td>
<td valign="middle" align="center">0.7360 (0.0016)</td>
<td valign="middle" align="center">0.6766 (0.0021)</td>
<td valign="middle" align="center">0.6900 (0.0040)</td>
<td valign="middle" align="center">0.6432 (0.0092)</td>
<td valign="middle" align="center">0.6651 (0.0039)</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;MVT-WDD-DI</td>
<td valign="middle" align="center">0.7395 (0.0128)</td>
<td valign="middle" align="center">0.6801 (0.0089)</td>
<td valign="middle" align="center">0.7075 (0.0084)</td>
<td valign="middle" align="center">0.6137 (0.0237)</td>
<td valign="middle" align="center">0.6540 (0.0185)</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;MVT-WDD-BF</td>
<td valign="middle" align="center">0.6127 (0.0212)</td>
<td valign="middle" align="center">0.5987 (0.0136)</td>
<td valign="middle" align="center">0.6172 (0.0258)</td>
<td valign="middle" align="center">0.4924 (0.0430)</td>
<td valign="middle" align="center">0.5389 (0.0376)</td>
</tr>
<tr>
<th valign="middle" colspan="6" align="center">Independent test</th>
</tr>
<tr>
<th valign="middle" colspan="6" align="center">PEI dataset</th>
</tr>
<tr>
<th valign="middle" align="center"/>
<th valign="middle" align="center">AUC</th>
<th valign="middle" align="center">ACC</th>
<th valign="middle" align="center">Precision</th>
<th valign="middle" align="center">Recall</th>
<th valign="middle" align="center">F1 score</th>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;WDD-KNN</td>
<td valign="middle" align="center">0.9276 (0.0089)</td>
<td valign="middle" align="center">0.8554 (0.0112)</td>
<td valign="middle" align="center">0.8888 (0.0142)</td>
<td valign="middle" align="center">0.8128 (0.0198)</td>
<td valign="middle" align="center">0.8489 (0.0125)</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;MVT-WDD-DI</td>
<td valign="middle" align="center">0.9194 (0.0254)</td>
<td valign="middle" align="center">0.8475 (0.0310)</td>
<td valign="middle" align="center">0.8826 (0.0299)</td>
<td valign="middle" align="center">0.8014 (0.0411)</td>
<td valign="middle" align="center">0.8398 (0.0340)</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;MVT-WDD-BF</td>
<td valign="middle" align="center">0.9071 (0.0214)</td>
<td valign="middle" align="center">0.8296 (0.0235)</td>
<td valign="middle" align="center">0.8429 (0.0280)</td>
<td valign="middle" align="center">0.8110 (0.0298)</td>
<td valign="middle" align="center">0.8263 (0.0243)</td>
</tr>
<tr>
<th valign="middle" colspan="6" align="center">BCA dataset</th>
</tr>
<tr>
<th valign="middle" align="center"/>
<th valign="middle" align="center">AUC</th>
<th valign="middle" align="center">ACC</th>
<th valign="middle" align="center">Precision</th>
<th valign="middle" align="center">Recall</th>
<th valign="middle" align="center">F1 score</th>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;WDD-KNN</td>
<td valign="middle" align="center">0.8870 (0.0057)</td>
<td valign="middle" align="center">0.8106 (0.0072)</td>
<td valign="middle" align="center">0.8469 (0.0098)</td>
<td valign="middle" align="center">0.7551 (0.0200)</td>
<td valign="middle" align="center">0.7982 (0.0099)</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;MVT-WDD-DI</td>
<td valign="middle" align="center">0.8625 (0.0151)</td>
<td valign="middle" align="center">0.7851 (0.0152)</td>
<td valign="middle" align="center">0.8198 (0.0192)</td>
<td valign="middle" align="center">0.7273 (0.0294)</td>
<td valign="middle" align="center">0.7704 (0.0184)</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;MVT-WDD-BF</td>
<td valign="middle" align="center">0.8988 (0.0326)</td>
<td valign="middle" align="center">0.8226 (0.0301)</td>
<td valign="middle" align="center">0.8467 (0.0309)</td>
<td valign="middle" align="center">0.7850 (0.0482)</td>
<td valign="middle" align="center">0.8140 (0.0362)</td>
</tr>
<tr>
<th valign="middle" colspan="6" align="center">Uri dataset</th>
</tr>
<tr>
<th valign="middle" align="center"/>
<th valign="middle" align="center">AUC</th>
<th valign="middle" align="center">ACC</th>
<th valign="middle" align="center">Precision</th>
<th valign="middle" align="center">Recall</th>
<th valign="middle" align="center">F1 score</th>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;WDD-KNN</td>
<td valign="middle" align="center">0.8224 (0.0207)</td>
<td valign="middle" align="center">0.7569 (0.0252)</td>
<td valign="middle" align="center">0.7621 (0.0426)</td>
<td valign="middle" align="center">0.7406 (0.0893)</td>
<td valign="middle" align="center">0.7463 (0.0359)</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;MVT-WDD-DI</td>
<td valign="middle" align="center">0.8893 (0.0230)</td>
<td valign="middle" align="center">0.8499 (0.0291)</td>
<td valign="middle" align="center">0.8355 (0.0348)</td>
<td valign="middle" align="center">0.8635 (0.0356)</td>
<td valign="middle" align="center">0.8488 (0.0303)</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;MVT-WDD-BF</td>
<td valign="middle" align="center">0.6304 (0.0581)</td>
<td valign="middle" align="center">0.6306 (0.0396)</td>
<td valign="middle" align="center">0.6863 (0.0649)</td>
<td valign="middle" align="center">0.4670 (0.1243)</td>
<td valign="middle" align="center">0.5433 (0.0891)</td>
</tr>
<tr>
<th valign="middle" colspan="6" align="center">BioChem dataset</th>
</tr>
<tr>
<th valign="middle" align="center"/>
<th valign="middle" align="center">AUC</th>
<th valign="middle" align="center">ACC</th>
<th valign="middle" align="center">Precision</th>
<th valign="middle" align="center">Recall</th>
<th valign="middle" align="center">F1 score</th>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;WDD-KNN</td>
<td valign="middle" align="center">0.7482 (0.0045)</td>
<td valign="middle" align="center">0.6885 (0.0065)</td>
<td valign="middle" align="center">0.6975 (0.0141)</td>
<td valign="middle" align="center">0.6588 (0.0256)</td>
<td valign="middle" align="center">0.6771 (0.0098)</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;MVT-WDD-DI</td>
<td valign="middle" align="center">0.7464 (0.0456)</td>
<td valign="middle" align="center">0.6862 (0.0300)</td>
<td valign="middle" align="center">0.7098 (0.0255)</td>
<td valign="middle" align="center">0.6186 (0.0841)</td>
<td valign="middle" align="center">0.6573 (0.0684)</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2003;MVT-WDD-BF</td>
<td valign="middle" align="center">0.6290 (0.0653)</td>
<td valign="middle" align="center">0.6104 (0.0442)</td>
<td valign="middle" align="center">0.6245 (0.0738)</td>
<td valign="middle" align="center">0.5192 (0.1228)</td>
<td valign="middle" align="center">0.5594 (0.1050)</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>From the <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>, we could see that the three algorithms had their own advantages on different sub-datasets. However, it was notable that the WDD-KNN algorithm had a KNN imputation step that the other 2 missing value adaptation algorithms did not have. Generally, imputation is limited by the template dataset. Large template datasets are often owned by a few large institutions and are difficult to share for reasons such as ethical review. To their advantage, the 2 missing value adaptation algorithms can skip this step when preprocessing dataset, and the built model does not need a template dataset for prediction, which will be beneficial in practical situations.</p>
<p>After ensuring the reliability of our modeling results through model scores, we delved deeper into the model&#x2019;s internal attributes and parameters in the following sections. This deeper analysis allowed us to extract valuable information pertinent to T2DM prediction, further validating the utility and interpretability of our algorithms.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Model scores provide auxiliary information other than blood glucose</title>
<p>The first aspect of our model&#x2019;s transparency is reflected in how the distance function illustrates the model and feature contributions to predict T2DM. In physical examinations, clinicians often use blood glucose, sometimes with urine glucose as a reference, to initially assess whether a person may have T2DM. For WDD, every sample was given a risk score (Dist for each algorithm, see Method details) by the models and classified according to a risk threshold. We compared the risk stratification through blood glucose and the risk scores (Dist) from our models (<xref ref-type="fig" rid="f3">
<bold>Figures&#xa0;3A&#x2013;C</bold>
</xref>).</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Model interpretability reflected by model scores. <bold>(A-C)</bold> Scatter-density heat maps of model score versus blood glucose of 3 models trained by whole PEI dataset. <bold>(D-F)</bold> Histograms of model score distribution of 3 models using whole PEI dataset. <bold>(G-I)</bold> The raincloud plots of distance scores (<inline-formula>
<mml:math display="inline" id="im29">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) of the three models using the whole PEI dataset.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fendo-15-1376220-g003.tif"/>
</fig>
<p>With the PEI dataset (<xref ref-type="fig" rid="f3">
<bold>Figures&#xa0;3A&#x2013;F</bold>
</xref>), we saw that in the models WDD-KNN, MVT-WDD-DI, and MVT-WDD-BF gave scores of confirmed T2DM patients that clustered in the ranges of 0.5-0.75, 0.4-0.82, and 0.4-3, respectively, while the physical examination population was clustered in the score ranges of 0.3-0.53, 0.1-0.5, and 0-1, respectively. All three models performed well in distinguishing the two populations, with WDD-KNN working best. However, it was difficult to completely distinguish the two groups if they were separated only by the level of blood glucose (<xref ref-type="fig" rid="f3">
<bold>Figures&#xa0;3A&#x2013;C</bold>
</xref>), and many people with T2DM still had the same blood glucose levels as normal people. Also, we analysed the risk scores on the three sub-datasets in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Note 6</bold>
</xref>.</p>
<p>To provide more information on the importance of the EHR features to every patient, we also calculated the <inline-formula>
<mml:math display="inline" id="im30">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>Dist</mml:mtext>
</mml:mrow>
<mml:mtext>k</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> score (see Method details) for each feature between the patients and normal people (<xref ref-type="fig" rid="f3">
<bold>Figures&#xa0;3G&#x2013;I</bold>
</xref>; <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S3</bold>
</xref>). As we can see, the selected important features mostly had different scores by each model. For example, in the model built on the PEI dataset using the MVT-WDD-DI algorithm, the <inline-formula>
<mml:math display="inline" id="im31">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>Dist</mml:mtext>
</mml:mrow>
<mml:mtext>k</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> scores of selected important features (<xref ref-type="fig" rid="f3">
<bold>Figures&#xa0;3H</bold>
</xref>, <xref ref-type="fig" rid="f4">
<bold>4</bold>
</xref>) such as Bact, BLD, Baso, Baso-R, and MCV were much higher than those of other features. T2DM patients and normal people could be well distinguished by the <inline-formula>
<mml:math display="inline" id="im32">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>Dist</mml:mtext>
</mml:mrow>
<mml:mtext>k</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> score of these important features. The result indicates that our model assigns higher significance to features with more pronounced differences, identifying them as important and thus selecting them as effective indicators for T2DM screening.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Important feature weights from different algorithms and datasets. Heat map based on the normalized feature weight values. The summation of a column (an algorithm) is 1. Darker colors represent larger weight values. The top 25% features in every column are framed by black rectangle.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fendo-15-1376220-g004.tif"/>
</fig>
<p>The results show that our model effectively differentiated T2DM patients from healthy individuals in the PEI dataset, as shown in the scatter density heat map. Visualized model scores underscored performance differences, revealing potential for early T2DM detection. Notably, normal blood glucose levels don&#x2019;t rule out T2DM, highlighting our model&#x2019;s diagnostic value. Additionally, by analyzing the model&#x2019;s distance function, we gained a deeper understanding of the mechanism behind the model&#x2019;s selection of important features, enhancing our comprehension of the model.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>The important indicators for T2DM diagnosis selected from different models</title>
<p>After understanding the scoring mechanism of the model and the mechanism for selecting important features, in this section and <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Notes 3</bold>
</xref> and <xref ref-type="supplementary-material" rid="SM1">
<bold>7</bold>
</xref>, we identified crucial diagnostic indicators for T2DM and analyzed the significance of the selected indicators for T2DM combining clinical knowledge.</p>
<p>The feature&#x2019;s significance is determined by its weight within the models, identifying the indicators most associated with T2DM. Important features were defined as those ranking in the top 25% by weight across the 12 models. We visually represented this distribution of relative feature weights with a histogram and the specific details of these crucial features are detailed in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>. We employed the Mann-Whitney U test to evaluate the level of feature differences between the T2DM and normal groups, finding that the selected important features exhibited significant differences (P value&lt; 0.0001) (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S6</bold>
</xref>). In addition, we compared the important features selected by using the internal weights of WDD with least absolute shrinkage and selection operator (LASSO) regression and SHAP framework. The important features showed certain consistency (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Note 9</bold>
</xref>, <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Information Figures S14, S15</bold>
</xref>).</p>
<p>When the three algorithms were applied each dataset, the selected important features intersected. For example, on the PEI dataset, the indicators judged as important features by all three models were BASO, BASO-R, and HCT, and the features given high weights by two of the three models were HDL-C, MUCUS, BACT, BLD, LYM-R, MCH, and MCV. In this dataset, the AUCs of all three models were higher than 0.88, so these indicators were selected as having great significance for T2DM prediction (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>). On the BCA dataset, the important features selected by the three algorithms are the same, which further proves the potential diagnostic value of these indicators. Moreover, we analysed the same and different important features extracted by the three algorithms, details are shown in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Note 7</bold>
</xref>.</p>
<p>We conducted a literature review to integrate our clinical expertise with published research findings and investigate the clinical correlations between these important features and T2DM. Our analysis revealed that most of these important features are shown to have direct or indirect associations with T2DM. For example, urinary tract infections are known to be correlated with diabetes (<xref ref-type="bibr" rid="B28">28</xref>) and some indicators associated with urinary tract infections, such as haematuria and bacteria in urine, have been selected as important biomarkers. The details are collated in the <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Note 3</bold>
</xref>.</p>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Multi-model analysis reveals characteristics among different age and sex groups</title>
<p>Leveraging our model&#x2019;s transparency and feature extraction capabilities, we conducted group modeling for populations with varying demographic characteristics to unearth the diagnostic value of indicators across different groups. To mitigate the potential model bias introduced by data imbalance, we ensured basic balance in the sample volume of each age and sex category for both T2DM patients and normal individuals, as illustrated in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S5</bold>
</xref> and <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S2</bold>
</xref>.</p>
<p>In most cases, age and sex will be correlated with T2DM incidence, which indicates that T2DM in different age and sex groups might have different characteristics. To further explore the importance of each feature for T2DM prediction in different age and sex groups, we tried three additional ways to divide the PEI Dataset: i. by age; ii. by sex; and iii. by age and sex. The number and proportion of people in different age and sex groups are shown in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S6</bold>
</xref> and <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S2</bold>
</xref>. After the division of the datasets, 26 sub-datasets (8 ages + 2 sexes + 8 ages &#xd7; 2 sexes) were generated, and 78 additional models (26 datasets &#xd7; 3 models) were built. The performance of each model is shown in <xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5A, B</bold>
</xref>. The WDD-KNN and MVT-WDD-DI algorithms performed well on each group of datasets, with AUC values mostly above 0.9, while the performance of MVT-WDD-BF was not as good. Therefore, in the subsequent analysis, only the weights of the first two algorithms were taken into consideration.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Model performance of 10-fold cross validation and feature importance in different age and sex groups. <bold>(A)</bold> The AUC values of the three algorithms when modeling male, female and both sexes of different ages. Error bars were generated by 10-fold cross validation (error bar represents standard deviation). <bold>(B)</bold> The AUC values of the three algorithms when modeling male and female of all ages. <bold>(C)</bold> Heat map of normalized feature weight values extracted from the model for male and female of different ages, &#x2018;M&#x2019; represents male, &#x2018;F&#x2019; represents female.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fendo-15-1376220-g005.tif"/>
</fig>
<p>The results showed that the importance of the clinical indicators varied in different age and sex groups (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5C</bold>
</xref>; <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S6</bold>
</xref>). We not only integrated the results of these models using the WDD-KNN and MVT-WDD-DI algorithms but also analysed the distribution of their measured values (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figures S7-13</bold>
</xref>) to explain the various importance of these indicators for T2DM diagnosis in the different groups.</p>
<p>The significance of glomerular filtration rate (GFR) in T2DM diagnosis diminishes with age, showing greater importance in the 5-49 age group (<xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5C</bold>
</xref>, <xref ref-type="fig" rid="f6">
<bold>6B</bold>
</xref>). Elevated GFR in T2DM patients aged 5-49 distinguishes them from normal groups, particularly in the 5-39 (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>). This aligns with studies linking diabetes and GFR, where early diabetic kidney disease (DKD) phases show increased GFR due to various changes in ultrastructural, vascular, and tubular factors (<xref ref-type="bibr" rid="B29">29</xref>). As renal health declines, GFR decreases (<xref ref-type="bibr" rid="B29">29</xref>, <xref ref-type="bibr" rid="B30">30</xref>). Our findings suggest that GFR&#x2019;s diagnostic value for T2DM varies across ages, particularly useful for early screening in younger populations, extending beyond its role in DKD.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Distribution of GFR values in different groups. <bold>(A)</bold> Different age and sex groups. <bold>(B)</bold> Different age groups. <bold>(C)</bold> Different sex groups. All the GFR values were from origin EHRs.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fendo-15-1376220-g006.tif"/>
</fig>
<p>Triglycerides (TG) showed greater significance in the 5-39 age group compared to others (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5C</bold>
</xref>; <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S6</bold>
</xref>), with notable differences in TG distribution between T2DM patients and normal individuals in this age range (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S7</bold>
</xref>). This variation is attributed to age-related dietary and metabolic differences and a genetic link identified by Saxena, R. et&#xa0;al. (<xref ref-type="bibr" rid="B31">31</xref>) High TG levels in T2DM patients are associated with increased cardiovascular risks (<xref ref-type="bibr" rid="B32">32</xref>) and metabolic changes (<xref ref-type="bibr" rid="B33">33</xref>). Our model emphasizes TG&#x2019;s importance in T2DM, especially in younger age groups, aligning with current research trends.</p>
<p>Haemoglobin (HGB) was more important in the 55- to 95-year-old group (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figures S5C, S6A</bold>
</xref>). In T2DM patients, as age increased, their lower HGB compared to that in normal people became more pronounced (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S12</bold>
</xref>). Based on our knowledge and experience, anaemia is diagnosed by HGB decline, so the association between anaemia and T2DM might be the use of metformin. There are reports supporting that long-term metformin use in T2DM patients can cause anaemia (<xref ref-type="bibr" rid="B34">34</xref>, <xref ref-type="bibr" rid="B35">35</xref>), and our EHR included patients who used metformin since metformin has been a commonly prescribed drug for T2DM patients for decades. Similarly, in diabetic patients with chronic kidney disease (CKD), some factors cause iron-deficiency anaemia, such as low intestinal absorption and gastrointestinal bleeding (<xref ref-type="bibr" rid="B36">36</xref>). In addition, erythropoietin deficiency and hyporesponsiveness can lead to anaemia in diabetic patients with CKD (<xref ref-type="bibr" rid="B36">36</xref>&#x2013;<xref ref-type="bibr" rid="B38">38</xref>). Nephrotic syndrome, characterized by oedema, hypoalbuminaemia, dyslipidaemia, and increased transferrin catabolism, contributes to anaemia due to iron and erythropoietin deficiency (<xref ref-type="bibr" rid="B36">36</xref>, <xref ref-type="bibr" rid="B39">39</xref>, <xref ref-type="bibr" rid="B40">40</xref>). Long-term administration of angiotensin-converting enzyme (ACE) inhibitors and angiotensin receptor antagonists in diabetic patients also leads to a reversible decrease in HGB through a direct blockade of the proerythropoietic effects of angiotensin II on red cell&#xa0;precursors, degradation of physiological inhibitors of haematopoiesis, and suppression of IGF-I (<xref ref-type="bibr" rid="B36">36</xref>, <xref ref-type="bibr" rid="B41">41</xref>). Thus, based on many studies and reports, taking HGB as an important feature will be a useful indicator for older patients with longer duration of diabetes, so HGB was selected after modelling the EHR data. In other words, HGB decline might be a marker of T2DM or T2DM-correlated disease, but it might be interfered with by some confounding factors, so its use for early diagnosis might be limited. This limitation is caused by the lack of medication information in our EHR data. Despite our meticulous selection of the patient&#x2019;s first record within the hospital system, we cannot guarantee that they have not undergone therapeutic interventions at other institutions. To address this shortcoming, cohort studies with long-term follow-up are needed.</p>
<p>The Neutrophils (NEU), neutrophil rate (NEU-R), lymphocyte rate (LYM), and lymphocyte rate (LYM-R) have also been observed to correlate with age or sex, as discussed in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Note 8</bold>
</xref>.</p>
<p>The result of this section demonstrates that through group modeling and the model&#x2019;s feature extraction capability, we identified several age and sex-related biomarkers for T2DM prediction. Integrating insights from the results, it&#x2019;s evident that our model can glean valuable auxiliary diagnostic information from internal parameters like feature weights and distance functions. This effectively showcases the algorithm&#x2019;s transparency throughout the modeling process, highlighting its capacity to provide interpretable insights crucial for clinical application.</p>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<label>4</label>
<title>Discussion</title>
<p>Here, we developed WDD framework-based models for the prediction of T2DM using physical examination features in EHRs. Using the model parameters, the importance of the features can be measured by the distance between the sample and optimal point. Based on our investigation, the top 25% of indicators were found to be directly or indirectly related to T2DM, offering potential value as diagnostic markers for T2DM.</p>
<p>In our analysis, a variety of white blood cells&#x2014;neutrophils, basophils, eosinophils, and lymphocytes&#x2014;emerged as significant features. This aligns with existing literature, which indicates that T2DM patients experiencing concurrent infections may exhibit inflammation, leading to an altered white blood cell count (<xref ref-type="bibr" rid="B42">42</xref>, <xref ref-type="bibr" rid="B43">43</xref>). While current research posits that the count or ratio levels of these white cells alone do not suffice as diabetes risk factors, a multifaceted approach is often necessary. For instance, the neutrophil-lymphocyte ratio is recognized as an independent predictor of T2DM (<xref ref-type="bibr" rid="B44">44</xref>). The inclusion of these white blood cells as important indicators by our model is consistent with current research insights, further affirming the potential of combining these white blood cell levels for aiding T2DM diagnosis. Moreover, this&#xa0;demonstrates our model&#x2019;s proficiency in capturing the complex interrelationships among indicators, highlighting its diagnostic relevance.</p>
<p>Indicators related to red blood cells and platelets, such as mean platelet volume, plateletcrit, hematocrit, coefficient of variation of red cell distribution width, mean corpuscular volume, and mean corpuscular haemoglobin, were also identified as significant by our model. In T2DM patients experiencing insulin resistance and metabolic syndrome, the adverse metabolic conditions&#x2014;including hyperglycemia, hypertension, dyslipidemia, inflammation, and impaired fibrinolysis&#x2014;elevate the risk of atherosclerosis and lead to microvascular complications like diabetic retinopathy, nephropathy, and neuropathy (<xref ref-type="bibr" rid="B45">45</xref>, <xref ref-type="bibr" rid="B46">46</xref>). Additionally, atherosclerosis, which may result from increased platelet adhesion and hypercoagulability in T2DM patients, is a key pathological mechanism behind macrovascular complications (<xref ref-type="bibr" rid="B46">46</xref>). These vascular complications can cause abnormalities in red blood cells and platelets. Therefore, the aforementioned indicators are linked to the common microvascular and macrovascular complications in diabetics, suggesting their potential as diagnostic markers for T2DM.</p>
<p>Urinalysis-related indicators, such as haematuria, leukocytes in urine, mucinous filaments, bacteria in urine, epithelial cells in urine, urine pH, and specific gravity, have been selected as significant markers by our model. These indicators are primarily associated with conditions prevalent among individuals with diabetes, such as urinary tract infections (<xref ref-type="bibr" rid="B28">28</xref>), which are notably common and can lead to haematuria or abnormal quantities of cells and bacteria in the urine. Moreover, the inflammation caused by these infections may result in an abnormal number of white blood cells, further validating the model&#x2019;s ability to discern potential relationships between indicators. The combination of increased net acid excretion and reduced use of ammonia buffers in individuals with diabetes leads to lower urine pH (<xref ref-type="bibr" rid="B47">47</xref>, <xref ref-type="bibr" rid="B48">48</xref>). A lower urine pH heightens the risk of nephrolithiasis, including uric acid stones (<xref ref-type="bibr" rid="B47">47</xref>, <xref ref-type="bibr" rid="B49">49</xref>). Diabetic nephropathy may manifest through abnormal urine specific gravity, where a lower-than-normal urinary specific gravity, along with increased polyuria, signals diabetes insipidus (<xref ref-type="bibr" rid="B50">50</xref>). These findings underscore the interconnectivity of urinary markers with diabetes-related infections and complications, emphasizing their potential diagnostic relevance.</p>
<p>While individual physical examination indicators often cannot serve as standalone diagnostic criteria for T2DM, our model successfully integrates multiple indicators to construct a diagnostic model for T2DM. Leveraging the model&#x2019;s high interpretability, we can determine the importance of each indicator in the diagnosis, enhancing its capability to aid in the auxiliary diagnosis of T2DM. This approach not only harnesses the collective diagnostic potential of various indicators but also provides valuable insights into their diagnostic significance, offering a refined perspective on T2DM diagnosis.</p>
<p>Besides, based on our clinical knowledge, the diagnosis of T2DM often correlates with demographic factors. Therefore, we segmented the data by age and gender, utilizing the feature weights provided by our model. Through this process, combined with a literature search, we identified several biomarkers related to age or sex, such as glomerular filtration rate, triglycerides, and haemoglobin. This further validates our model&#x2019;s efficacy in extracting medically valuable information and illustrates that different indicators may require attention when diagnosing diabetes in patients of varying ages and sexes. This approach not only enriches the diagnostic model with nuanced clinical insights but also underscores the importance of personalized medicine in the management and treatment of T2DM.</p>
<p>The results show that our algorithm boasts a high degree of internal interpretability, enabling the extraction of key indicators for T2DM diagnosis without the need for third-party tools. Furthermore, by analyzing the model&#x2019;s parameters (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>), we can comprehend the mechanism behind the selection of important indicators, thereby providing reliable auxiliary diagnostic information. Additionally, our model possesses a distinct advantage as mentioned in the Methods section: benefiting from the transparency of WDD, its internal distance function is easily modifiable. Instead of imputing missing values, our approach involves &#x2018;tolerating&#x2019; them by incorporating penalty terms into two of the algorithms. This strategy diminishes the necessity for exhaustive searches for high-quality template data during the imputation process, proving WDD to be a highly transparent and interpretable auxiliary diagnostic algorithm.</p>
<p>This work suggests that machine learning could extend beyond predictive accuracy to include interpretative insights, which might be useful in clinical settings. Such insights have the potential to aid clinicians in understanding the basis of diagnostic suggestions given by the model. This could lead to a more cooperative relationship between machine learning and healthcare professionals. However, the integration of these technologies in clinical practice requires careful consideration and ongoing evaluation.</p>
</sec>
<sec id="s5" sec-type="conclusions">
<label>5</label>
<title>Conclusion</title>
<p>Overall, we developed three WDD-based interpretability algorithms and built T2DM diagnostic models, identifying several relevant diagnostic indicators with potential utility in assisting T2DM diagnosis. However, it is crucial to acknowledge that the mechanisms of interaction among these indicators, as well as their causal connections with T2DM, cannot be directly deduced from the current model information. In our future work, leveraging the transparency of WDD, we plan to incorporate knowledge of causal probabilities to enhance our model further, uncover the complex relationships between indicators and T2DM.</p>
</sec>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>The code used for modelling in this work is available at: <ext-link ext-link-type="uri" xlink:href="https://github.com/Lvxiang713/WDD_T2DMprediction">https://github.com/Lvxiang713/WDD_T2DMprediction</ext-link>. Rearchers should contact the corresponding authors for approval to obtain and use the source data. The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material</bold>
</xref>. Further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s7" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>The studies involving humans were approved by This study protocol was approved by the ethics committee of the Affiliated Hospital of Southwest Medical University, China (KY2022266) and the Chinese Clinical Trial Registry (ChiCTR2200064435). The studies were conducted in accordance with the local legislation and institutional requirements. Written informed consent for participation was not required from the participants or the participants' legal guardians/next of kin in accordance with the national legislation and institutional requirements.</p>
</sec>
<sec id="s8" sec-type="author-contributions">
<title>Author contributions</title>
<p>XL: Writing &#x2013; review &amp; editing, Writing &#x2013; original draft, Visualization, Methodology, Formal analysis. JL: Writing &#x2013; original draft, Investigation, Funding acquisition, Formal analysis, Data curation. HW: Writing &#x2013; review &amp; editing, Investigation, Data curation. HG: Writing &#x2013; review &amp; editing, Visualization, Formal analysis. XB: Writing &#x2013; original draft, Investigation, Formal analysis, Data curation. PY: Writing &#x2013; original draft, Data curation. ZJ: Writing &#x2013; review &amp; editing, Data curation. YZ: Writing &#x2013; original draft, Investigation, Formal analysis. RJ: Writing &#x2013; review &amp; editing, Writing &#x2013; original draft, Validation, Methodology, Funding acquisition, Data curation, Conceptualization. QC: Writing &#x2013; original draft, Supervision, Funding acquisition, Data curation, Conceptualization. ML: Writing &#x2013; review &amp; editing, Supervision, Funding acquisition, Conceptualization.</p>
</sec>
</body>
<back>
<sec id="s9" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This work has been supported by the National Natural Science Foundation of China (No. 22203057, No. 82200904 and No. 22173065), Science &amp; Technology Department of Sichuan province (No.24NSFSC1621, No. 2022YF0617, No.2022YFS0612 and No.2022YFS0617-B2), Joint project of Luzhou Municipal People&#x2019;s Government and Southwest Medical University (No. 2020LZXNYDJ30 and No. 2020LZXNYDJ39).</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>The numerical calculations in this paper have been done on Hefei advanced computing center.</p>
</ack>
<sec id="s10" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s11" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fendo.2024.1376220/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fendo.2024.1376220/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet_1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Donath</surname> <given-names>MY</given-names>
</name>
<name>
<surname>Shoelson</surname> <given-names>SE</given-names>
</name>
</person-group>. <article-title>Type 2 diabetes as an inflammatory disease</article-title>. <source>Nat Rev Immunol</source>. (<year>2011</year>) <volume>11</volume>:<fpage>98</fpage>&#x2013;<lpage>107</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nri2925</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wong</surname> <given-names>TY</given-names>
</name>
<name>
<surname>Cheung</surname> <given-names>CMG</given-names>
</name>
<name>
<surname>Larsen</surname> <given-names>M</given-names>
</name>
<name>
<surname>Sharma</surname> <given-names>DO</given-names>
</name>
<name>
<surname>Sim&#xf3;</surname> <given-names>R</given-names>
</name>
</person-group>. <article-title>Diabetic retinopathy</article-title>. <source>Nat Rev Dis Primers</source>. (<year>2016</year>) <volume>2</volume>:<fpage>1</fpage>&#x2013;<lpage>17</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nrdp.2016.12</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Matheus AS de</surname> <given-names>M</given-names>
</name>
<name>
<surname>Tannus</surname> <given-names>LRM</given-names>
</name>
<name>
<surname>Cobas</surname> <given-names>RA</given-names>
</name>
<name>
<surname>Palma</surname> <given-names>CCS</given-names>
</name>
<name>
<surname>Negrato</surname> <given-names>CA</given-names>
</name>
<name>
<surname>Gomes M de</surname> <given-names>B</given-names>
</name>
</person-group>. <article-title>Impact of diabetes on cardiovascular disease: an update</article-title>. <source>Int J hypertension</source>. (<year>2013</year>) <volume>2013</volume>
<fpage>:653789</fpage>. doi: <pub-id pub-id-type="doi">10.1155/2013/653789</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Carson</surname> <given-names>AP</given-names>
</name>
<name>
<surname>Muntner</surname> <given-names>P</given-names>
</name>
<name>
<surname>Kissela</surname> <given-names>BM</given-names>
</name>
<name>
<surname>Kleindorfer</surname> <given-names>DO</given-names>
</name>
<name>
<surname>Howard</surname> <given-names>VJ</given-names>
</name>
<name>
<surname>Meschia</surname> <given-names>JF</given-names>
</name>
<etal/>
</person-group>. <article-title>Association of prediabetes and diabetes with stroke symptoms: the REasons for Geographic and Racial Differences in Stroke (REGARDS) study</article-title>. <source>Diabetes Care</source>. (<year>2012</year>) <volume>35</volume>:<page-range>1845&#x2013;52</page-range>. doi: <pub-id pub-id-type="doi">10.2337/dc11-2140</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rathur</surname> <given-names>HM</given-names>
</name>
<name>
<surname>Boulton</surname> <given-names>AJ</given-names>
</name>
</person-group>. <article-title>The neuropathic diabetic foot</article-title>. <source>Nat Rev Endocrinol</source>. (<year>2007</year>) <volume>3</volume>:<fpage>14</fpage>&#x2013;<lpage>25</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/ncpendmet0347</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6</label>
<citation citation-type="web">
<article-title>IDF Diabetes Atlas</article-title>. Available online at: <uri xlink:href="https://diabetesatlas.org/">https://diabetesatlas.org/</uri> (Accessed <access-date>September 22, 2022</access-date>).</citation>
</ref>
<ref id="B7">
<label>7</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cao</surname> <given-names>X</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>M</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>XB</given-names>
</name>
<name>
<surname>Tan</surname> <given-names>XL</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Huo</surname> <given-names>N</given-names>
</name>
<etal/>
</person-group>. <article-title>Prevalence and rates of new diagnosis and missed diagnosis of diabetes mellitus among 35&#x2013;74-year-old residents in urban communities in Southwest China</article-title>. <source>Biomed Environ Sci</source>. (<year>2019</year>) <volume>32</volume>:<page-range>704&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.3967/bes2019.089</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tasin</surname> <given-names>I</given-names>
</name>
<name>
<surname>Nabil</surname> <given-names>TU</given-names>
</name>
<name>
<surname>Islam</surname> <given-names>S</given-names>
</name>
<name>
<surname>Khan</surname> <given-names>R</given-names>
</name>
</person-group>. <article-title>Diabetes prediction using machine learning and explainable AI techniques</article-title>. <source>Healthcare Tech Lett</source>. (<year>2023</year>) <volume>10</volume>:<fpage>1</fpage>&#x2013;<lpage>10</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1049/htl2.12039</pub-id>
</citation>
</ref>
<ref id="B9">
<label>9</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ejiyi</surname> <given-names>CJ</given-names>
</name>
<name>
<surname>Qin</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Amos</surname> <given-names>J</given-names>
</name>
<name>
<surname>Ejiyi</surname> <given-names>MB</given-names>
</name>
<name>
<surname>Nnani</surname> <given-names>A</given-names>
</name>
<name>
<surname>Ejiyi</surname> <given-names>TU</given-names>
</name>
<etal/>
</person-group>. <article-title>A robust predictive diagnosis model for diabetes mellitus using Shapley-incorporated machine learning algorithms</article-title>. <source>Healthcare Analytics</source>. (<year>2023</year>) <volume>3</volume>:<elocation-id>100166</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.health.2023.100166</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Patel</surname> <given-names>R</given-names>
</name>
<name>
<surname>Sivaiah</surname> <given-names>B</given-names>
</name>
<name>
<surname>Patel</surname> <given-names>P</given-names>
</name>
<name>
<surname>Sahoo</surname> <given-names>B</given-names>
</name>
</person-group>. <article-title>Supervised Learning Approaches on the Prediction of Diabetic Disease in Healthcare</article-title>. In: <person-group person-group-type="editor">
<name>
<surname>Udgata</surname> <given-names>SK</given-names>
</name>
<name>
<surname>Sethi</surname> <given-names>S</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>X-Z</given-names>
</name>
</person-group>, editors. <source>Intelligent Systems</source>. <publisher-name>Springer Nature</publisher-name>, <publisher-loc>Singapore</publisher-loc> (<year>2024</year>). p. <page-range>157&#x2013;68</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-981-99-3932-9_15</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Chaw</surname> <given-names>JK</given-names>
</name>
<name>
<surname>Ang</surname> <given-names>MC</given-names>
</name>
<name>
<surname>Daud</surname> <given-names>MM</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>L</given-names>
</name>
</person-group>. <article-title>A Diabetes Prediction Model with Visualized Explainable Artificial Intelligence (XAI) Technology</article-title>. In: <person-group person-group-type="editor">
<name>
<surname>Badioze Zaman</surname> <given-names>H</given-names>
</name>
<name>
<surname>Robinson</surname> <given-names>P</given-names>
</name>
<name>
<surname>Smeaton</surname> <given-names>AF</given-names>
</name>
<name>
<surname>De Oliveira</surname> <given-names>RL</given-names>
</name>
<name>
<surname>J&#xf8;rgensen</surname> <given-names>BN</given-names>
</name>
<name>
<surname>K. Shih</surname> <given-names>T</given-names>
</name>
<name>
<surname>Abdul Kadir</surname> <given-names>R</given-names>
</name>
<name>
<surname>Mohamad</surname> <given-names>UH</given-names>
</name>
<name>
<surname>Ahmad</surname> <given-names>MN</given-names>
</name>
</person-group>, editors. <source>Advances in Visual Informatics</source>. <publisher-name>Springer Nature</publisher-name>, <publisher-loc>Singapore</publisher-loc> (<year>2024</year>). p. <page-range>648&#x2013;61</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-981-99-7339-2_52</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bhat</surname> <given-names>SS</given-names>
</name>
<name>
<surname>Selvam</surname> <given-names>V</given-names>
</name>
<name>
<surname>Ansari</surname> <given-names>GA</given-names>
</name>
<name>
<surname>Ansari</surname> <given-names>MD</given-names>
</name>
<name>
<surname>Rahman</surname> <given-names>MH</given-names>
</name>
</person-group>. <article-title>Prevalence and early prediction of diabetes using machine learning in North Kashmir: A case study of district bandipora</article-title>. <source>Comput Intell Neurosci</source>. (<year>2022</year>) <volume>2022</volume>:<elocation-id>e2789760</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1155/2022/2789760</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mahesh</surname> <given-names>TR</given-names>
</name>
<name>
<surname>Kumar</surname> <given-names>D</given-names>
</name>
<name>
<surname>Vinoth Kumar</surname> <given-names>V</given-names>
</name>
<name>
<surname>Asghar</surname> <given-names>J</given-names>
</name>
<name>
<surname>Mekcha Bazezew</surname> <given-names>B</given-names>
</name>
<name>
<surname>Natarajan</surname> <given-names>R</given-names>
</name>
<etal/>
</person-group>. <article-title>Blended ensemble learning prediction model for strengthening diagnosis and treatment of chronic diabetes disease</article-title>. <source>Comput Intell Neurosci</source>. (<year>2022</year>) <volume>2022</volume>:<elocation-id>e4451792</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1155/2022/4451792</pub-id>
</citation>
</ref>
<ref id="B14">
<label>14</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ravaut</surname> <given-names>M</given-names>
</name>
<name>
<surname>Harish</surname> <given-names>V</given-names>
</name>
<name>
<surname>Sadeghi</surname> <given-names>H</given-names>
</name>
<name>
<surname>Leung</surname> <given-names>KK</given-names>
</name>
<name>
<surname>Volkovs</surname> <given-names>M</given-names>
</name>
<name>
<surname>Kornas</surname> <given-names>K</given-names>
</name>
<etal/>
</person-group>. <article-title>Development and validation of a machine learning model using administrative health data to predict onset of type 2 diabetes</article-title>. <source>JAMA Network Open</source>. (<year>2021</year>) <volume>4</volume>:<elocation-id>e2111315</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1001/jamanetworkopen.2021.11315</pub-id>
</citation>
</ref>
<ref id="B15">
<label>15</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mansoori</surname> <given-names>A</given-names>
</name>
<name>
<surname>Sahranavard</surname> <given-names>T</given-names>
</name>
<name>
<surname>Hosseini</surname> <given-names>ZS</given-names>
</name>
<name>
<surname>Soflaei</surname> <given-names>SS</given-names>
</name>
<name>
<surname>Emrani</surname> <given-names>N</given-names>
</name>
<name>
<surname>Nazar</surname> <given-names>E</given-names>
</name>
<etal/>
</person-group>. <article-title>Prediction of type 2 diabetes mellitus using hematological factors based on machine learning approaches: a cohort study analysis</article-title>. <source>Sci Rep</source>. (<year>2023</year>) <volume>13</volume>:<fpage>663</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-022-27340-2</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>M</given-names>
</name>
<name>
<surname>Wan</surname> <given-names>J</given-names>
</name>
<name>
<surname>Qin</surname> <given-names>W</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>G</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>X</given-names>
</name>
</person-group>. <article-title>A machine learning-based diagnosis modelling of type 2 diabetes mellitus with environmental metal exposure</article-title>. <source>Comput Methods Programs Biomedicine</source>. (<year>2023</year>) <volume>235</volume>:<elocation-id>107537</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cmpb.2023.107537</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>De Silva</surname> <given-names>K</given-names>
</name>
<name>
<surname>J&#xf6;nsson</surname> <given-names>D</given-names>
</name>
<name>
<surname>Demmer</surname> <given-names>RT</given-names>
</name>
</person-group>. <article-title>A combined strategy of feature selection and machine learning to identify predictors of prediabetes</article-title>. <source>J Am Med Inf Assoc</source>. (<year>2020</year>) <volume>27</volume>:<fpage>396</fpage>&#x2013;<lpage>406</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/jamia/ocz204</pub-id>
</citation>
</ref>
<ref id="B18">
<label>18</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname> <given-names>H</given-names>
</name>
<name>
<surname>Xin</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Li</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>A diabetes prediction model based on Boruta feature selection and ensemble learning</article-title>. <source>BMC Bioinf</source>. (<year>2023</year>) <volume>24</volume>:<fpage>224</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12859-023-05300-5</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Patro</surname> <given-names>KK</given-names>
</name>
<name>
<surname>Allam</surname> <given-names>JP</given-names>
</name>
<name>
<surname>Sanapala</surname> <given-names>U</given-names>
</name>
<name>
<surname>Marpu</surname> <given-names>CK</given-names>
</name>
<name>
<surname>Samee</surname> <given-names>NA</given-names>
</name>
<name>
<surname>Alabdulhafith</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>An effective correlation-based data modeling framework for automatic diabetes prediction using machine and deep learning techniques</article-title>. <source>BMC Bioinf</source>. (<year>2023</year>) <volume>24</volume>:<fpage>372</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12859-023-05488-6</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hassan</surname> <given-names>M</given-names>
</name>
<name>
<surname>Mollick</surname> <given-names>S</given-names>
</name>
<name>
<surname>Yasmin</surname> <given-names>F</given-names>
</name>
</person-group>. <article-title>An unsupervised cluster-based feature grouping model for early diabetes detection</article-title>. <source>Healthcare Analytics</source>. (<year>2022</year>) <volume>2</volume>:<elocation-id>100112</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.health.2022.100112</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Saxena</surname> <given-names>R</given-names>
</name>
<name>
<surname>Sharma</surname> <given-names>SK</given-names>
</name>
<name>
<surname>Gupta</surname> <given-names>M</given-names>
</name>
<name>
<surname>Sampada</surname> <given-names>GC</given-names>
</name>
</person-group>. <article-title>A novel approach for feature selection and classification of diabetes mellitus: machine learning methods</article-title>. <source>Comput Intell Neurosci</source>. (<year>2022</year>) <volume>2022</volume>:<elocation-id>e3820360</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1155/2022/3820360</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mushtaq</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Ramzan</surname> <given-names>MF</given-names>
</name>
<name>
<surname>Ali</surname> <given-names>S</given-names>
</name>
<name>
<surname>Baseer</surname> <given-names>S</given-names>
</name>
<name>
<surname>Samad</surname> <given-names>A</given-names>
</name>
<name>
<surname>Husnain</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Voting classification-based diabetes mellitus prediction using hypertuned machine-learning techniques</article-title>. <source>Mobile Inf Syst</source>. (<year>2022</year>) <volume>2022</volume>:<elocation-id>e6521532</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1155/2022/6521532</pub-id>
</citation>
</ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>M</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>W</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>G</given-names>
</name>
<name>
<surname>Peng</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Predicting cell-type specific disease genes of diabetes with the biological network</article-title>. <source>Comput Biol Med</source>. (<year>2024</year>) <volume>169</volume>:<elocation-id>107849</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compbiomed.2023.107849</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rudin</surname> <given-names>C</given-names>
</name>
</person-group>. <article-title>Stop explaining black box machine learning models for high stakes decisions and use interpretable models instead</article-title>. <source>Nat Mach Intell</source>. (<year>2019</year>) <volume>1</volume>:<page-range>206&#x2013;15</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s42256-019-0048-x</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Smith</surname> <given-names>JW</given-names>
</name>
<name>
<surname>Everhart</surname> <given-names>JE</given-names>
</name>
<name>
<surname>Dickson</surname> <given-names>WC</given-names>
</name>
<name>
<surname>Knowler</surname> <given-names>WC</given-names>
</name>
<name>
<surname>Johannes</surname> <given-names>RS</given-names>
</name>
</person-group>. <article-title>Using the ADAP learning algorithm to forecast the onset of diabetes mellitus</article-title>. <source>Proc Annu Symp Comput Appl Med Care</source>. (<year>1988</year>), <page-range>261&#x2013;5</page-range>.</citation>
</ref>
<ref id="B26">
<label>26</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arbet</surname> <given-names>J</given-names>
</name>
<name>
<surname>Brokamp</surname> <given-names>C</given-names>
</name>
<name>
<surname>Meinzen-Derr</surname> <given-names>J</given-names>
</name>
<name>
<surname>Trinkley</surname> <given-names>KE</given-names>
</name>
<name>
<surname>Spratt</surname> <given-names>HM</given-names>
</name>
</person-group>. <article-title>Lessons and tips for designing a machine learning study using EHR data</article-title>. <source>J Clin Trans Sci</source>. (<year>2021</year>) <volume>5</volume>:<elocation-id>e21</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1017/cts.2020.513</pub-id>
</citation>
</ref>
<ref id="B27">
<label>27</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Maron</surname> <given-names>O</given-names>
</name>
<name>
<surname>Lozano-P&#xe9;rez</surname> <given-names>T</given-names>
</name>
</person-group>. <article-title>A framework for multiple-instance learning</article-title>. In: <source>Advances in neural information processing systems</source>, vol. <volume>10</volume>. (<year>1997</year>) <page-range>570&#x2013;576</page-range>.</citation>
</ref>
<ref id="B28">
<label>28</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Muller</surname> <given-names>LMAJ</given-names>
</name>
<name>
<surname>Gorter</surname> <given-names>KJ</given-names>
</name>
<name>
<surname>Hak</surname> <given-names>E</given-names>
</name>
<name>
<surname>Goudzwaard</surname> <given-names>WL</given-names>
</name>
<name>
<surname>Schellevis</surname> <given-names>FG</given-names>
</name>
<name>
<surname>Hoepelman</surname> <given-names>AIM</given-names>
</name>
<etal/>
</person-group>. <article-title>Increased risk of common infections in patients with type 1 and type 2 diabetes mellitus</article-title>. <source>Clin Infect Dis</source>. (<year>2005</year>) <volume>41</volume>:<page-range>281&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1086/431587</pub-id>
</citation>
</ref>
<ref id="B29">
<label>29</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tonneijck</surname> <given-names>L</given-names>
</name>
<name>
<surname>Muskiet</surname> <given-names>MH</given-names>
</name>
<name>
<surname>Smits</surname> <given-names>MM</given-names>
</name>
<name>
<surname>Van Bommel</surname> <given-names>EJ</given-names>
</name>
<name>
<surname>Heerspink</surname> <given-names>HJ</given-names>
</name>
<name>
<surname>Van Raalte</surname> <given-names>DH</given-names>
</name>
<etal/>
</person-group>. <article-title>Glomerular hyperfiltration in diabetes: mechanisms, clinical significance, and treatment</article-title>. <source>J Am Soc Nephrol</source>. (<year>2017</year>) <volume>28</volume>:<page-range>1023&#x2013;39</page-range>. doi: <pub-id pub-id-type="doi">10.1681/ASN.2016060666</pub-id>
</citation>
</ref>
<ref id="B30">
<label>30</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moriya</surname> <given-names>T</given-names>
</name>
<name>
<surname>Tsuchiya</surname> <given-names>A</given-names>
</name>
<name>
<surname>Okizaki</surname> <given-names>S</given-names>
</name>
<name>
<surname>Hayashi</surname> <given-names>A</given-names>
</name>
<name>
<surname>Tanaka</surname> <given-names>K</given-names>
</name>
<name>
<surname>Shichiri</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Glomerular hyperfiltration and increased glomerular filtration surface are associated with renal function decline in normo- and microalbuminuric type 2 diabetes</article-title>. <source>Kidney Int</source>. (<year>2012</year>) <volume>81</volume>:<page-range>486&#x2013;93</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/ki.2011.404</pub-id>
</citation>
</ref>
<ref id="B31">
<label>31</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Saxena</surname> <given-names>R</given-names>
</name>
<name>
<surname>Voight</surname> <given-names>BF</given-names>
</name>
<name>
<surname>Lyssenko</surname> <given-names>V</given-names>
</name>
<name>
<surname>Burtt</surname> <given-names>NP</given-names>
</name>
<name>
<surname>de Bakker</surname> <given-names>PIW</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>H</given-names>
</name>
<etal/>
</person-group>. <article-title>Genome-wide association analysis identifies loci for type 2 diabetes and triglyceride levels</article-title>. <source>Science</source>. (<year>2007</year>) <volume>316</volume>:<page-range>1331&#x2013;6</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1126/science.1142358</pub-id>
</citation>
</ref>
<ref id="B32">
<label>32</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ye</surname> <given-names>X</given-names>
</name>
<name>
<surname>Kong</surname> <given-names>W</given-names>
</name>
<name>
<surname>Zafar</surname> <given-names>MI</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>L-L</given-names>
</name>
</person-group>. <article-title>Serum triglycerides as a risk factor for cardiovascular diseases in type 2 diabetes mellitus: a systematic review and meta-analysis of prospective studies</article-title>. <source>Cardiovasc Diabetol</source>. (<year>2019</year>) <volume>18</volume>:<fpage>48</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12933-019-0851-z</pub-id>
</citation>
</ref>
<ref id="B33">
<label>33</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rijzewijk</surname> <given-names>LJ</given-names>
</name>
<name>
<surname>Jonker</surname> <given-names>JT</given-names>
</name>
<name>
<surname>van der</surname> <given-names>MRW</given-names>
</name>
<name>
<surname>Lubberink</surname> <given-names>M</given-names>
</name>
<name>
<surname>de</surname> <given-names>JHW</given-names>
</name>
<name>
<surname>Romijn</surname> <given-names>JA</given-names>
</name>
<etal/>
</person-group>. <article-title>Effects of hepatic triglyceride content on myocardial metabolism in type 2 diabetes</article-title>. <source>J&#xa0;Am Coll Cardiol</source>. (<year>2010</year>) <volume>56</volume>:<page-range>225&#x2013;33</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jacc.2010.02.049</pub-id>
</citation>
</ref>
<ref id="B34">
<label>34</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Donnelly</surname> <given-names>LA</given-names>
</name>
<name>
<surname>Dennis</surname> <given-names>JM</given-names>
</name>
<name>
<surname>Coleman</surname> <given-names>RL</given-names>
</name>
<name>
<surname>Sattar</surname> <given-names>N</given-names>
</name>
<name>
<surname>Hattersley</surname> <given-names>AT</given-names>
</name>
<name>
<surname>Holman</surname> <given-names>RR</given-names>
</name>
<etal/>
</person-group>. <article-title>Risk of anemia with metformin use in type 2 diabetes: A MASTERMIND study</article-title>. <source>Diabetes Care</source>. (<year>2020</year>) <volume>43</volume>:<page-range>2493&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.2337/dc20-1104</pub-id>
</citation>
</ref>
<ref id="B35">
<label>35</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Aroda</surname> <given-names>VR</given-names>
</name>
<name>
<surname>Edelstein</surname> <given-names>SL</given-names>
</name>
<name>
<surname>Goldberg</surname> <given-names>RB</given-names>
</name>
<name>
<surname>Knowler</surname> <given-names>WC</given-names>
</name>
<name>
<surname>Marcovina</surname> <given-names>SM</given-names>
</name>
<name>
<surname>Orchard</surname> <given-names>TJ</given-names>
</name>
<etal/>
</person-group>. <article-title>Long-term metformin use and vitamin B12 deficiency in the diabetes prevention program outcomes study</article-title>. <source>J Clin Endocrinol Metab</source>. (<year>2016</year>) <volume>101</volume>:<page-range>1754&#x2013;61</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1210/jc.2015-3754</pub-id>
</citation>
</ref>
<ref id="B36">
<label>36</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mehdi</surname> <given-names>U</given-names>
</name>
<name>
<surname>Toto</surname> <given-names>RD</given-names>
</name>
</person-group>. <article-title>Anemia, diabetes, and chronic kidney disease</article-title>. <source>Diabetes Care</source>. (<year>2009</year>) <volume>32</volume>:<page-range>1320&#x2013;6</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.2337/dc08-0779</pub-id>
</citation>
</ref>
<ref id="B37">
<label>37</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thomas</surname> <given-names>MC</given-names>
</name>
</person-group>. <article-title>Anemia in diabetes: marker or mediator of microvascular disease</article-title>? <source>Nat Rev Nephrol</source>. (<year>2007</year>) <volume>3</volume>:<fpage>20</fpage>&#x2013;<lpage>30</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/ncpneph0378</pub-id>
</citation>
</ref>
<ref id="B38">
<label>38</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Erslev</surname> <given-names>AJ</given-names>
</name>
<name>
<surname>Besarab</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Erythropoietin in the pathogenesis and treatment of the anemia of chronic renal failure</article-title>. <source>Kidney Int</source>. (<year>1997</year>) <volume>51</volume>:<page-range>622&#x2013;30</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/ki.1997.91</pub-id>
</citation>
</ref>
<ref id="B39">
<label>39</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vaziri</surname> <given-names>ND</given-names>
</name>
</person-group>. <article-title>Erythropoietin and transferrin metabolism in nephrotic syndrome</article-title>. <source>Am J Kidney Dis</source>. (<year>2001</year>) <volume>38</volume>:<fpage>1</fpage>&#x2013;<lpage>8</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1053/ajkd.2001.25174</pub-id>
</citation>
</ref>
<ref id="B40">
<label>40</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Howard</surname> <given-names>RL</given-names>
</name>
<name>
<surname>Buddington</surname> <given-names>B</given-names>
</name>
<name>
<surname>Alfrey</surname> <given-names>AC</given-names>
</name>
</person-group>. <article-title>Urinary albumin, transferrin and iron excretion in diabetic patients</article-title>. <source>Kidney Int</source>. (<year>1991</year>) <volume>40</volume>:<page-range>923&#x2013;6</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/ki.1991.295</pub-id>
</citation>
</ref>
<ref id="B41">
<label>41</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Marathias</surname> <given-names>KP</given-names>
</name>
<name>
<surname>Agroyannis</surname> <given-names>B</given-names>
</name>
<name>
<surname>Mavromoustakos</surname> <given-names>T</given-names>
</name>
<name>
<surname>Matsoukas</surname> <given-names>J</given-names>
</name>
<name>
<surname>Vlahakos</surname> <given-names>DV</given-names>
</name>
</person-group>. <article-title>Hematocrit-lowering effect following inactivation of renin-angiotensin system with angiotensin converting enzyme inhibitors and angiotensin receptor blockers</article-title>. <source>Curr Top Med Chem</source>. (<year>2004</year>) <volume>4</volume>:<page-range>483&#x2013;6</page-range>. doi: <pub-id pub-id-type="doi">10.2174/1568026043451311</pub-id>
</citation>
</ref>
<ref id="B42">
<label>42</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vozarova</surname> <given-names>B</given-names>
</name>
<name>
<surname>Weyer</surname> <given-names>C</given-names>
</name>
<name>
<surname>Lindsay</surname> <given-names>RS</given-names>
</name>
<name>
<surname>Pratley</surname> <given-names>RE</given-names>
</name>
<name>
<surname>Bogardus</surname> <given-names>C</given-names>
</name>
<name>
<surname>Tataranni</surname> <given-names>PA</given-names>
</name>
</person-group>. <article-title>High white blood cell count is associated with a worsening of insulin sensitivity and predicts the development of type 2 diabetes</article-title>. <source>Diabetes</source>. (<year>2002</year>) <volume>51</volume>:<page-range>455&#x2013;61</page-range>. doi: <pub-id pub-id-type="doi">10.2337/diabetes.51.2.455</pub-id>
</citation>
</ref>
<ref id="B43">
<label>43</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Twig</surname> <given-names>G</given-names>
</name>
<name>
<surname>Afek</surname> <given-names>A</given-names>
</name>
<name>
<surname>Shamiss</surname> <given-names>A</given-names>
</name>
<name>
<surname>Derazne</surname> <given-names>E</given-names>
</name>
<name>
<surname>Tzur</surname> <given-names>D</given-names>
</name>
<name>
<surname>Gordon</surname> <given-names>B</given-names>
</name>
<etal/>
</person-group>. <article-title>White blood cells count and incidence of type 2 diabetes in young men</article-title>. <source>Diabetes Care</source>. (<year>2013</year>) <volume>36</volume>:<page-range>276&#x2013;82</page-range>. doi: <pub-id pub-id-type="doi">10.2337/dc11-2298</pub-id>
</citation>
</ref>
<ref id="B44">
<label>44</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mangalesh</surname> <given-names>S</given-names>
</name>
<name>
<surname>Dudani</surname> <given-names>S</given-names>
</name>
<name>
<surname>Yadav</surname> <given-names>P</given-names>
</name>
<name>
<surname>Podury</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Evaluation of neutrophil-lymphocyte ratio in diabetes and coronary artery disease: a case control study from India</article-title>. <source>Am Heart J</source>. (<year>2021</year>) <volume>242</volume>:<page-range>156&#x2013;7</page-range>. doi: <pub-id pub-id-type="doi">10.1016/j.ahj.2021.10.030</pub-id>
</citation>
</ref>
<ref id="B45">
<label>45</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reusch</surname> <given-names>JEB</given-names>
</name>
</person-group>. <article-title>Diabetes, microvascular complications, and cardiovascular complications: what is it about glucose</article-title>? <source>J Clin Invest</source>. (<year>2003</year>) <volume>112</volume>:<page-range>986&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1172/JCI19902</pub-id>
</citation>
</ref>
<ref id="B46">
<label>46</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fowler</surname> <given-names>MJ</given-names>
</name>
</person-group>. <article-title>Microvascular and macrovascular complications of diabetes</article-title>. <source>Clin Diabetes</source>. (<year>2011</year>) <volume>29</volume>:<page-range>116&#x2013;22</page-range>. doi: <pub-id pub-id-type="doi">10.2337/diaclin.29.3.116</pub-id>
</citation>
</ref>
<ref id="B47">
<label>47</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Maalouf</surname> <given-names>NM</given-names>
</name>
<name>
<surname>Cameron</surname> <given-names>MA</given-names>
</name>
<name>
<surname>Moe</surname> <given-names>OW</given-names>
</name>
<name>
<surname>Sakhaee</surname> <given-names>K</given-names>
</name>
</person-group>. <article-title>Metabolic basis for low urine pH in type 2 diabetes</article-title>. <source>Clin J Am Soc Nephrol</source>. (<year>2010</year>) <volume>5</volume>:<page-range>1277&#x2013;81</page-range>. doi: <pub-id pub-id-type="doi">10.2215/CJN.08331109</pub-id>
</citation>
</ref>
<ref id="B48">
<label>48</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Eisner</surname> <given-names>BH</given-names>
</name>
<name>
<surname>Porten</surname> <given-names>SP</given-names>
</name>
<name>
<surname>Bechis</surname> <given-names>SK</given-names>
</name>
<name>
<surname>Stoller</surname> <given-names>ML</given-names>
</name>
</person-group>. <article-title>Diabetic kidney stone formers excrete more oxalate and have lower urine pH than nondiabetic stone formers</article-title>. <source>J Urol</source>. (<year>2010</year>) <volume>183</volume>:<page-range>2244&#x2013;8</page-range>. doi: <pub-id pub-id-type="doi">10.1016/j.juro.2010.02.007</pub-id>
</citation>
</ref>
<ref id="B49">
<label>49</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bell</surname> <given-names>DSH</given-names>
</name>
</person-group>. <article-title>Beware the low urine pH&#x2014;the major cause of the increased prevalence of nephrolithiasis in the patient with type 2 diabetes</article-title>. <source>Diabetes Obes Metab</source>. (<year>2012</year>) <volume>14</volume>:<fpage>299</fpage>&#x2013;<lpage>303</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.1463-1326.2011.01519.x</pub-id>
</citation>
</ref>
<ref id="B50">
<label>50</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Akarsu</surname> <given-names>E</given-names>
</name>
<name>
<surname>Buyukhatipoglu</surname> <given-names>H</given-names>
</name>
<name>
<surname>Aktaran</surname> <given-names>S</given-names>
</name>
<name>
<surname>Geyik</surname> <given-names>R</given-names>
</name>
</person-group>. <article-title>The value of urine specific gravity in detecting diabetes insipidus in a patient with uncontrolled diabetes mellitus: urine specific gravity in differential diagnosis</article-title>. <source>J Gen Internal Med</source>. (<year>2006</year>) <volume>21</volume>:<page-range>C1&#x2013;2</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/j.1525-1497.2006.00454.x</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>