<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Nephrol.</journal-id>
<journal-title>Frontiers in Nephrology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Nephrol.</abbrev-journal-title>
<issn pub-type="epub">2813-0626</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fneph.2023.1237804</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Nephrology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Development and deployment of a nationwide predictive model for chronic kidney disease progression in diabetic patients</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Fu</surname>
<given-names>Zhiyan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1528527"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Zhiyu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Clemente</surname>
<given-names>Karen</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Jaisinghani</surname>
<given-names>Mohit</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2342995"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Poon</surname>
<given-names>Ken Mei Ting</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yeo</surname>
<given-names>Anthony Wee Teo</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ang</surname>
<given-names>Gia Lee</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liew</surname>
<given-names>Adrian</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lim</surname>
<given-names>Chee Kong</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Foo</surname>
<given-names>Marjorie Wai Yin</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chow</surname>
<given-names>Wai Leng</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Ta</surname>
<given-names>Wee An</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Integrated Health Information Systems (IHIS)</institution>, <addr-line>Singapore</addr-line>, <country>Singapore</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Mount Elizabeth Novena Hospital</institution>, <addr-line>Singapore</addr-line>, <country>Singapore</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>National Health Group Polyclinics</institution>, <addr-line>Singapore</addr-line>, <country>Singapore</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Department of Renal Medicine, Singapore General Hospital</institution>, <addr-line>Singapore</addr-line>, <country>Singapore</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Epidemiology and Disease Control Division, Ministry of Health</institution>, <addr-line>Singapore</addr-line>, <country>Singapore</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Mengling Feng, National University of Singapore, Singapore</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Mei Liu, University of Florida, United States</p>
<p>Michele Provenzano, University of Bologna, Italy</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Wee An Ta, <email xlink:href="mailto:andy.ta@ihis.com.sg">andy.ta@ihis.com.sg</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>08</day>
<month>01</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>3</volume>
<elocation-id>1237804</elocation-id>
<history>
<date date-type="received">
<day>12</day>
<month>07</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>11</day>
<month>12</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Fu, Wang, Clemente, Jaisinghani, Poon, Yeo, Ang, Liew, Lim, Foo, Chow and Ta</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Fu, Wang, Clemente, Jaisinghani, Poon, Yeo, Ang, Liew, Lim, Foo, Chow and Ta</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Aim</title>
<p>Chronic kidney disease (CKD) is a major complication of diabetes and a significant disease burden on the healthcare system. The aim of this work was to apply a predictive model to identify high-risk patients in the early stages of CKD as a means to provide early intervention to avert or delay kidney function deterioration.</p>
</sec>
<sec>
<title>Materials and methods</title>
<p>Using the data from the National Diabetes Database in Singapore, we applied a machine-learning algorithm to develop a predictive model for CKD progression in diabetic patients and to deploy the model nationwide.</p>
</sec>
<sec>
<title>Results</title>
<p>Our model was rigorously validated. It outperformed existing models and clinician predictions. The area under the receiver operating characteristic curve (AUC) of our model is 0.88, with the 95% confidence interval being 0.87 to 0.89. In recognition of its higher and consistent accuracy and clinical usefulness, our CKD model became the first clinical model deployed nationwide in Singapore and has been incorporated into a national program to engage patients in long-term care plans in battling chronic diseases. The risk score generated by the model stratifies patients into three risk levels, which are embedded into the Diabetes Patient Dashboard for clinicians and care managers who can then allocate healthcare resources accordingly.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>This project provided a successful example of how an artificial intelligence (AI)-based model can be adopted to support clinical decision-making nationwide.</p>
</sec>
</abstract>
<kwd-group>
<kwd>chronic kidney disease</kwd>
<kwd>CKD progression</kwd>
<kwd>diabetes</kwd>
<kwd>kidney function deterioration</kwd>
<kwd>machine learning</kwd>
<kwd>prediction</kwd>
</kwd-group>
<counts>
<fig-count count="5"/>
<table-count count="3"/>
<equation-count count="0"/>
<ref-count count="17"/>
<page-count count="10"/>
<word-count count="5194"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Clinical Research in Nephrology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Chronic kidney disease (CKD) is a major complication arising from diabetes, and around 25%&#x2013;40% of patients with diabetes develop diabetic kidney disease (<xref ref-type="bibr" rid="B1">1</xref>). Diabetes is not only one of the most common diseases but also the leading cause of kidney failure. In Singapore, diabetes accounts for two-thirds of cases of end-stage kidney failure (ESKF) requiring dialysis, with the prevalence being one of the highest in the world according to the United States Renal Data System (USRDS)&#x2019;s data (<ext-link ext-link-type="uri" xlink:href="https://adr.usrds.org/2021/end-stage-renal-disease/11-international-comparisons">https://adr.usrds.org/2021/end-stage-renal-disease/11-international-comparisons</ext-link>). Diabetes leads to kidney damage via two main pathways: chronic hyperglycemia and the activation of the renin&#x2013;angiotensin system. These processes ultimately induce glomerular sclerosis, albuminuria, and kidney impairment (<xref ref-type="bibr" rid="B2">2</xref>). CKD is projected to be the fifth leading cause of death globally by 2040 (<xref ref-type="bibr" rid="B3">3</xref>). Because it is often asymptomatic in the early stages, CKD is usually diagnosed late. The treatment for patients with ESKF requires substantial resources and is still a great challenge to a healthcare system that is under resource limits and budget control. The patients themselves also suffer from a significant loss of quality of life and from the financial burdens affecting them as a result of their condition.</p>
<p>It is critical to identify patients who are at high risk of kidney function deterioration to avert or delay their progression to ESKF. These patients could be treated in the early stages and be engaged early on in their long-term kidney care plans. The Nephrology Evaluation, Management and Optimization (NEMO) program was introduced to optimize renal protection treatment earlier within the primary care setting. Compared with diabetic patients not enrolled in NEMO, the patients in the NEMO program had a 28% lower rate of kidney function deterioration (<ext-link ext-link-type="uri" xlink:href="https://www.healthhub.sg/a-z/medical-and-care-facilities/nemo-programme-for-diabetic-kidney-patients">https://www.healthhub.sg/a-z/medical-and-care-facilities/nemo-programme-for-diabetic-kidney-patients</ext-link>). However, the NEMO program focused only on patients from one polyclinic cluster in Singapore and addressed only one treatment parameter. Consequently, the Holistic Approach in Lowering and Tracking Chronic Kidney Disease (HALT-CKD) program, which is comprised of nephrologists from all public hospitals in Singapore, was launched in 2017 as an enhanced and extended version of the NEMO program. The HALT-CKD national program targets and tracks a much broader cohort, that is, all CKD patients from stage 1 to stage 4 in Singapore. It is also carried out at all polyclinics across Singapore and aims to identify and control evidence-based risk factors of CKD to delay CKD progression.</p>
<p>A prediction method of CKD progression has been studied in some western countries&#x2019; populations to assist in the early intervention of patients. Tangri et&#xa0;al. (<xref ref-type="bibr" rid="B4">4</xref>) developed a risk model of CKD progression to ESKF in the Canadian population and found a list of important indicators&#x2014;age, sex, estimated glomerular filtration rate (eGFR), and levels of urine albumin, serum calcium, serum phosphate, serum bicarbonate, and serum albumin. Later, this model was evaluated and recalibrated for multinational cohorts (<xref ref-type="bibr" rid="B5">5</xref>). The method of Tangri et&#xa0;al. was also applied to predict the probability and the timing of kidney failure that requires kidney replacement therapy (<xref ref-type="bibr" rid="B6">6</xref>). Ramspeck et&#xa0;al. (<xref ref-type="bibr" rid="B7">7</xref>) selected 11 prediction models on kidney failure and validated them on two large cohorts in Europe. The model&#x2019;s performance on the validation cohorts was considerably worse than on the original cohorts due to the narrower patient mix of the validation cohorts. This result indicates the necessity of retraining and even redesigning the CKD prediction model when applying it to populations with different ethnicities.</p>
<p>Low et&#xa0;al. (<xref ref-type="bibr" rid="B8">8</xref>) followed 1,582 patients with type 2 diabetes mellitus, from 2002 to 2014, and developed a logistic regression model for CKD progression specifically for diabetic patients in one Singapore hospital. In recent years, machine-learning algorithms, such as random forest and XGBoost, were also used to create the risk models of CKD progression, which showed a potential for improvement from statistical methods (<xref ref-type="bibr" rid="B9">9</xref>, <xref ref-type="bibr" rid="B10">10</xref>).</p>
<p>However, there are some limitations in the previous studies. First, the study population did not cover the different ethnic groups in Singapore and was not representative of all diabetic patients in the country. Second, there was no clinical validation to confirm the viability of the models and to demonstrate their clinical usefulness in practice. Some studies did not even have validation datasets to check the stability of the model&#x2019;s performance.</p>
<p>In the current study, we developed a predictive model for CKD progression in 5 years for diabetic patients in Singapore. Our study used the nationwide medical data of patients in Singapore by utilizing the two national systems: the National Diabetes Database (NDD) and the Business Research Analytics Insights Network (BRAIN). Our study result has been validated technically and clinically by nephrologists from public hospitals, hence the endorsement from the HALT-CKD program&#x2019;s committee to incorporate it into its action plan.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>Data collection and model development/deployment platform</title>
<p>The patient data used in this study were from the National Diabetes Database (NDD) within the Business Research Analytics Insights Network (BRAIN) platform.</p>
<sec id="s2_1_1">
<label>2.1.1</label>
<title>National Diabetes Database</title>
<p>The NDD consolidated data from multiple repositories across Singapore. It was jointly developed by the Ministry of Health (MOH) and the Integrated Health Information Systems (IHIS) in Singapore to support the national &#x201c;War on Diabetes&#x201d; initiative, which aims to enhance citizens&#x2019; awareness of diabetes and to create a supportive environment to prevent and manage diabetes well (<ext-link ext-link-type="uri" xlink:href="https://www.moh.gov.sg/wodcj">https://www.moh.gov.sg/wodcj</ext-link>). Prior to the creation of the NDD, each hospital had its own registry of diabetic patients. The diabetic patients and their data were sat in silos, as shown in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>. The data may be overlapped and inconsistent due to different data sources and criteria. The NDD standardized the criteria and integrated data from multiple sources and kept it updated with the new information. It aims to facilitate outreach, policy intervention, and better clinical management for diabetic patients.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Formation of the National Diabetes Database.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fneph-03-1237804-g001.tif"/>
</fig>
<p>The NDD has provided high-quality data on diabetic patients in Singapore since 2011. It covers nine major public hospitals in Singapore: Changi General Hospital (CGH), Singapore General Hospital (SGH), KK Women&#x2019;s and Children&#x2019;s Hospital (KKH), Seng Kang Hospital (SKH), Khoo Teck Puat Hospital (KTPH), Tan Tock Seng Hospital (TTSH), National University Hospital (NUH), Ng Teng Fong General Hospital (NTFGH), and Alexandra Hospital (AH). It also covers three polyclinic clusters in Singapore: Singhealth Group Polyclinics (SHP), National Healthcare Group Polyclinics (NHGP), and National University Polyclinics (NUP). The data include demographics, events, laboratory tests, diagnoses, medications, foot screenings, eye screenings, procedures, clinical measurements, finance, referrals, and mortality (see details in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure 1</bold>
</xref>).</p>
</sec>
<sec id="s2_1_2">
<label>2.1.2</label>
<title>Business Research Analytics Insights Network</title>
<p>The NDD and the CKD prediction tools reside in the BRAIN (<xref ref-type="bibr" rid="B11">11</xref>). The CKD prediction model was developed and deployed in the BRAIN platform. The relevant data and predicted scores are stored and aggregated in the NDD. These data are used to produce dashboards for policymakers and clinicians (see examples in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figures 2</bold>
</xref>, <xref ref-type="supplementary-material" rid="SM1">
<bold>3</bold>
</xref>).</p>
</sec>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Study base cohort for the model&#x2019;s development</title>
<p>A retrospective study cohort consists of adult patients diagnosed with diabetes and in the early stages of CKD before 1 January 2013 in Singapore. The study cohort was used to develop a predictive model on CKD progression. <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref> shows the formation of the study cohort with the exclusion criteria: (i) patients not having type 1 diabetes or type 2 diabetes before 1 January 2013, (ii) patients not in CKD stage 1 or stage 2 in 2012, (iii) patients aged &lt; 18 years in 2013, and (iv) patients with missing gender data. The definitions of the different stages of CKD were based on the Kidney Disease Improving Global Outcome Guidelines (KDIGO) with slight modifications (refer to <xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Material</bold></xref> for details). Patients who were at CKD stages 3&#x2013;5 were identified by two estimated glomerular filtration rate (eGFR) readings that were more than 90 days apart to exclude acute kidney injury episodes. The historical data of medication and laboratory tests were extracted to analyze patients&#x2019; medical and clinical conditions<xref ref-type="fn" rid="fn1">
<sup>1</sup>
</xref>.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Definition of the study base cohort for the model&#x2019;s development.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fneph-03-1237804-g002.tif"/>
</fig>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Predictive model variables and feature engineering</title>
<p>The dependent variable of the predictive model was defined as a patient&#x2019;s 5-year risk of CKD progression. That is, whether a patient&#x2019;s kidney damage will deteriorate into CKD stage 3 or above in 5 years. Specifically, the model was trained on the study base cohort to predict on 1 January 2013 their risk of CKD progression in the next 5 years until 1 January 2018.</p>
<p>Independent variables of the model encompass five categories: sociodemographic characteristics, blood pressure, diagnosis, laboratory tests, and medication. Feature engineering was conducted to convert the study data into these variables based on previous studies of CKD progression and input from clinicians and care managers. There were 177 variables generated initially, and feature selection was performed using the backward feature selection method to choose the most relevant features while maximizing the model&#x2019;s performance (<xref ref-type="bibr" rid="B12">12</xref>, <xref ref-type="bibr" rid="B13">13</xref>) during model training. The feature selection process was done by repeatedly removing the variables that were not important according to the feature importance of the XGBoost model or not contributing to the model&#x2019;s performance, until the variables in the final models were all important in prediction, and the removal of any variable would lead to a significant drop in the model&#x2019;s performance. There were 25 variables selected to be included in the model.</p>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Machine-learning model development</title>
<p>We used the Extreme Gradient Boosting algorithm (XGBoost) to build the predictive model and grid search to fine-tune hyper-parameters. The hyper-parameters included the maximum depth of each decision tree (i.e., <italic>max_depth</italic>), the minimum sum of instance weight (hessian) needed in a child (i.e., <italic>min_child_weight</italic>), and the number of decision trees in the final model (i.e., <italic>nrounds</italic>). Our model using XGBoost could handle the missing data directly due to the sparse-aware split finding (<xref ref-type="bibr" rid="B14">14</xref>). The model was developed in R (R is a programming language for statistical computing, created by Ross Ihaka and Robert Gentleman).</p>
<p>The study cohort was randomly split into two datasets by patient ID: a training dataset for the model&#x2019;s development and a test dataset for the model&#x2019;s validation. <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> describes patients&#x2019; distribution in the two datasets. To avoid the overfitting problem and to fine-tune the hyper-parameters, threefold cross-validation was applied to the training dataset. Cross-validation separated the data into several partitions, trained on different combinations of partitions, and evaluated the remaining partitions. The model&#x2019;s performance was measured by three metrics: area under the curve (AUC), positive predictive value (PPV), and sensitivity when PPV and sensitivity have a trade-off relationship influenced by the risk score threshold.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Distribution of patients and CKD progression in the different datasets.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" rowspan="2" align="left">Data</th>
<th valign="top" colspan="2" align="center">Number of patients</th>
<th valign="top" colspan="2" align="center">CKD progression<break/>into stage 3 or<break/>above in 5 years</th>
</tr>
<tr>
<th valign="top" align="center"/>
<th valign="top" align="center">%</th>
<th valign="top" align="center">Number of patients</th>
<th valign="top" align="center">%</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Training</td>
<td valign="top" align="center">14,240</td>
<td valign="top" align="center">75%</td>
<td valign="top" align="center">5,300</td>
<td valign="top" align="center">37.2%</td>
</tr>
<tr>
<td valign="top" align="left">Test</td>
<td valign="top" align="center">4,810</td>
<td valign="top" align="center">25%</td>
<td valign="top" align="center">1,801</td>
<td valign="top" align="center">37.4%</td>
</tr>
<tr>
<td valign="top" align="left">Total</td>
<td valign="top" align="center">19,050</td>
<td valign="top" align="center">100%</td>
<td valign="top" align="center">7,101</td>
<td valign="top" align="center">37.3%</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2_5">
<label>2.5</label>
<title>Technical and clinical validation of the developed model</title>
<p>Technical validation of the model developed was performed using a new cohort with 6-month shift test data. The new cohort consisted of patients filtered using the same exclusion criteria as the base cohort but shifting the ending date to 30 June 2013, instead of 31 December 2012. The patients in the base cohort were removed so that the new cohort had no overlap with the base cohort. The technical cohort contained 5,490 patients and their risk of CKD progression in the next 5 years until 30 June 2018 was predicted using our developed model.</p>
<p>One-year shift data (2013 cohort) were generated for clinical validation by shifting the ending date of the cohort to 31 December 2013 and removing the patients overlapped with the base cohort or the technical validation cohort. One thousand patients were randomly selected from the 2013 cohort, and the same features used by our model were provided to 20 clinicians across three polyclinics groups. The clinicians were asked to assess the patients&#x2019; risk of CKD progression in the next 5 years; we then compared the clinicians&#x2019; assessment with our model&#x2019;s prediction.</p>
<p>The model package and the development script are available at: <ext-link ext-link-type="uri" xlink:href="https://github.com/beverly0005/CKD/tree/main">https://github.com/beverly0005/CKD/tree/main</ext-link>.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results</title>
<p>In the current study, we developed a model for CKD progression in diabetic patients. Our study cohort contained 19,050 adult patients diagnosed with diabetes and CKD stages 1 and 2 before 2013; 37.3% of them deteriorated into CKD stage 3 or above in 5 years. <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> shows the descriptive summary of the study cohort. The training, test, and validation datasets all showed similar characteristics as shown in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Descriptive statistics of the study cohort.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Variable<break/>Study cohort, <italic>N</italic> = 19,059</th>
<th valign="top" align="center">% Missing</th>
<th valign="top" align="center">% Cohort</th>
<th valign="top" align="center"/>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Male (yes, 1; no, 0)</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">48.7%</td>
<td valign="top" align="center"/>
</tr>
<tr>
<td valign="top" align="left">Ethnicity: Chinese (yes, 1; no, 0)</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">69.3%</td>
<td valign="top" align="center"/>
</tr>
<tr>
<td valign="top" align="left">Ethnicity: Indian (yes, 1; no, 0)</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">12.5%</td>
<td valign="top" align="center"/>
</tr>
<tr>
<td valign="top" align="left">Ethnicity: Malay (yes, 1; no, 0)</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">14.4%</td>
<td valign="top" align="center"/>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="center">% missing</td>
<td valign="top" align="center">Mean (SD)</td>
<td valign="top" align="center">Interquantile,<break/>Q1, Q3</td>
</tr>
<tr>
<td valign="top" align="left">Age in 2013 (years)</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">61.6 (11.4)</td>
<td valign="top" align="center">54.0, 69.0</td>
</tr>
<tr>
<td valign="top" align="left">Diabetes duration (years)</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">7.5 (5.9)</td>
<td valign="top" align="center">3.0, 10.0</td>
</tr>
<tr>
<td valign="top" align="left">Latest hemoglobin A<sub>1C</sub> result before 2013 (%)</td>
<td valign="top" align="center">0.2</td>
<td valign="top" align="center">7.9 (1.6)</td>
<td valign="top" align="center">6.8, 8.5</td>
</tr>
<tr>
<td valign="top" align="left">Prior hemoglobin A<sub>1C</sub> result before 2013 (%)</td>
<td valign="top" align="center">5.9</td>
<td valign="top" align="center">7.9 (1.6)</td>
<td valign="top" align="center">6.8, 8.5</td>
</tr>
<tr>
<td valign="top" align="left">Latest albumin urine result before 2013 (mg/g)</td>
<td valign="top" align="center">2.1</td>
<td valign="top" align="center">213.3 (479.7)</td>
<td valign="top" align="center">36.4, 176.6</td>
</tr>
<tr>
<td valign="top" align="left">Latest glucose result before 2013 (mmol/L)</td>
<td valign="top" align="center">3.1</td>
<td valign="top" align="center">8.5 (3.4)</td>
<td valign="top" align="center">6.4, 9.6</td>
</tr>
<tr>
<td valign="top" align="left">Latest hemoglobin result before 2013 (g/dL)</td>
<td valign="top" align="center">48.3</td>
<td valign="top" align="center">13.0 (1.7)</td>
<td valign="top" align="center">11.9, 14.2</td>
</tr>
<tr>
<td valign="top" align="left">Prior hemoglobin result before 2013 (g/dL)</td>
<td valign="top" align="center">77.3</td>
<td valign="top" align="center">12.6 (1.8)</td>
<td valign="top" align="center">11.4, 13.9</td>
</tr>
<tr>
<td valign="top" align="left">Latest urine albumin-to-creatinine ratio before 2013 (mg/mmol)</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">29.8 (83.6)</td>
<td valign="top" align="center">4.7, 21.8</td>
</tr>
<tr>
<td valign="top" align="left">Latest albumin serum result before 2013 (g/L)</td>
<td valign="top" align="center">67.8</td>
<td valign="top" align="center">38.3 (6.0)</td>
<td valign="top" align="center">35.0, 42.0</td>
</tr>
<tr>
<td valign="top" align="left">Prior albumin serum result before (g/L)</td>
<td valign="top" align="center">87.8</td>
<td valign="top" align="center">37.6 (5.9)</td>
<td valign="top" align="center">34.0, 42.0</td>
</tr>
<tr>
<td valign="top" align="left">Latest eGFR before 2013 (mL/min/1.73m<sup>2</sup>)</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">88.2 (16.1)</td>
<td valign="top" align="center">75.0, 100.0</td>
</tr>
<tr>
<td valign="top" align="left">Prior eGFR before 2013 (mL/min/1.73m<sup>2</sup>)</td>
<td valign="top" align="center">50.8</td>
<td valign="top" align="center">98.7 (41.3)</td>
<td valign="top" align="center">76.0, 103.0</td>
</tr>
<tr>
<td valign="top" align="left">Latest alanine aminotransferase result before 2013 (U/L)</td>
<td valign="top" align="center">3.8</td>
<td valign="top" align="center">27.2 (17.6)</td>
<td valign="top" align="center">16.0, 33.0</td>
</tr>
<tr>
<td valign="top" align="left">Latest triglyceride result before 2013 (mmol/L)</td>
<td valign="top" align="center">2.3</td>
<td valign="top" align="center">1.6 (1.1)</td>
<td valign="top" align="center">1.1, 1.9</td>
</tr>
<tr>
<td valign="top" align="left">Latest high-density lipoprotein result before 2013 (mmol/L)</td>
<td valign="top" align="center">1.6</td>
<td valign="top" align="center">1.3 (0.4)</td>
<td valign="top" align="center">1.0, 1.5</td>
</tr>
<tr>
<td valign="top" align="left">Latest low-density lipoprotein result before 2013 (mmol/L)</td>
<td valign="top" align="center">2.1</td>
<td valign="top" align="center">2.5 (0.8)</td>
<td valign="top" align="center">2.0, 2.9</td>
</tr>
<tr>
<td valign="top" align="left">Latest total cholesterol result before 2013 (mmol/L)</td>
<td valign="top" align="center">2.3</td>
<td valign="top" align="center">4.5 (1.0)</td>
<td valign="top" align="center">3.9, 5.0</td>
</tr>
<tr>
<td valign="top" align="left">Number of distinct medicines (ATC codes) for the alimentary tract or metabolism dispensed in 2012</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">3.6 (2.3)</td>
<td valign="top" align="center">2.0, 5.0</td>
</tr>
<tr>
<td valign="top" align="left">Number of distinct medicines (ATC codes) for cardiovascular system dispensed in 2012</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">3.1 (1.6)</td>
<td valign="top" align="center">2.0, 4.0</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>eGFR, estimated glomerular filtration rate.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Following our feature selection process, the original 177 variables were narrowed down to four categories of variables and used for training our CKD model, as elaborated below. The sociodemographic characteristics included age, gender, and race. The diagnosis variable included diabetes duration, which was the number of days from the earliest diabetes diagnosis date to the date of prediction, i.e., 1 January 2013. The variables derived from the laboratory test history included measurements of hemoglobin A<sub>1C</sub>, albumin urine, glucose, hemoglobin, the urine albumin-to-creatinine ratio, albumin serum, eGFR, alanine aminotransferase, triglycerides, high-density lipoprotein, low-density lipoprotein, and total cholesterol. The glomerular filtration rate (GFR) is an important metric that reflects kidney function, but measurement requires a complicated and lengthy procedure. The eGFR is an approximate to glomerular filtration rate. It is calculated using the CKD-EPI equation based on serum creatinine, age, gender, and ethnicity (non-Black <ext-link ext-link-type="uri" xlink:href="http://www.kidney.org/content/ckd-epi-creatinine-euqation-2009">www.kidney.org/content/ckd-epi-creatinine-euqation-2009</ext-link>). The statistics of these laboratory tests are shown in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>. The &#x201c;prior&#x201d; laboratory readings in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> refer to the test result closest to but at least 90 days before the latest laboratory reading before 1 January 2013. This gives an indication of the magnitude of the change in the laboratory results and provides a sense of the progression of the patients&#x2019; clinical conditions and parameters.</p>
<p>The medication data were standardized into the groups from the Anatomical Therapeutic Chemical Classification (ATC) of which the first level contains 14 main anatomical and pharmacological groups (<xref ref-type="bibr" rid="B15">15</xref>). For example, an ATC code with the first level &#x201c;A&#x201d; tells that the drug is related to the alimentary tract or metabolism, while the ATC code with the first level &#x201c;C&#x201d; tells that the drug acts on the cardiovascular system.</p>
<p>
<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref> displays the important predictive factors for the CKD progression in our model. Consistent with the findings in previous studies (<xref ref-type="bibr" rid="B4">4</xref>, <xref ref-type="bibr" rid="B16">16</xref>, <xref ref-type="bibr" rid="B17">17</xref>), the leading factors include eGFR and albumin urine concentration. Interestingly, we found that the inclusion of medication history, specifically drugs for the cardiovascular system and metabolism, and other laboratory tests such as hemoglobin levels improved the model&#x2019;s performance in predicting the 5-year risk of CKD progression for diabetic patients.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Variable importance in the CKD model.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fneph-03-1237804-g003.tif"/>
</fig>
<p>Our developed CKD model outperformed the current clinical practice and the mainstream statistical models for predicting CKD progression. Clinicians use the eGFR to diagnose CKD in current practice. We developed a Cox regression model using only an eGFR to mimic the clinicians&#x2019; assessment. Additionally, we developed another Cox regression model using the features in Tangri&#x2019;s paper (<xref ref-type="bibr" rid="B4">4</xref>), which was widely recognized and had a study design similar to ours. <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref> summarizes the performance of our models and other major models. Our model had the best performance in terms of area under the receiver operating characteristic (ROC) curve (AUC) value. The 95% confidence interval of our model&#x2019;s AUC is 0.87 to 0.89, which is significantly higher than other statistical models.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Performance of the different models.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Model</th>
<th valign="top" align="left">Method</th>
<th valign="top" align="left">Data</th>
<th valign="top" align="left">AUC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" rowspan="2" align="left">Our Model</td>
<td valign="top" rowspan="2" align="left">XGBoost</td>
<td valign="top" align="left">Test data<sup>1</sup>
</td>
<td valign="top" align="center">
<bold>0.88</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">6-month-shift test data<sup>2</sup>
</td>
<td valign="top" align="center">
<bold>0.88</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">Clinician</td>
<td valign="top" align="left">Cox regression</td>
<td valign="top" align="left">Test data</td>
<td valign="top" align="center">0.81</td>
</tr>
<tr>
<td valign="top" align="left">Statistical Model using features in Tangri et&#xa0;al. [2]</td>
<td valign="top" align="left">Cox regression</td>
<td valign="top" align="left">Test data</td>
<td valign="top" align="center">0.84</td>
</tr>
<tr>
<td valign="top" align="left">Tangri et&#xa0;al. [2]</td>
<td valign="top" align="left">Cox regression</td>
<td valign="top" align="left">Based on published paper</td>
<td valign="top" align="center">0.84</td>
</tr>
<tr>
<td valign="top" align="left">Lin et&#xa0;al. [3]</td>
<td valign="top" align="left">Cox regression</td>
<td valign="top" align="left">Based on published paper</td>
<td valign="top" align="center">0.86</td>
</tr>
<tr>
<td valign="top" align="left">Song et&#xa0;al. [4]</td>
<td valign="top" align="left">Gradient boosting</td>
<td valign="top" align="left">Based on published paper</td>
<td valign="top" align="center">0.83</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>
<sup>1</sup>Test data are the cohort data before 1 January 2013, which contained 4,810 patients.</p>
</fn>
<fn>
<p>
<sup>2</sup>Six-month shift-test data are the cohort data before 1 July 2013, which contained 4,590 patients.  The 2 bold 0.88 are the AUC of our model, which are the highest compared to other models.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>In addition, we tested the stability of our model&#x2019;s performance on a new cohort by shifting 6 months. The new cohort consisted of the patients using the same exclusion criteria as in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref> but shifting the ending date to 30 June 2013 instead of 31 December 2012. The new cohort did not overlap with the study cohort discussed above. Our model predicted their risk of CKD progression in the next 5 years until 1 July 2018. As <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref> shows, our model&#x2019;s performance was stable, indicating its potential to be generalized to new patients in different periods.</p>
<p>We have further validated our developed model clinically by comparison with clinicians tasked with predicting CKD progression clinically. We generated a new 2013 diabetes cohort of 1,000 patients as described in Methods and provided their features used by our model to 20 clinicians across all three polyclinics groups in Singapore (i.e., NHGP, NUP, and SHP). These 1,000 patients were not included in the original 2012 cohort that was used to develop the model or in the 6-month shift-test data. We asked the clinicians to assess the patients&#x2019; risk of CKD progression in the next 5 years and then compared the performance of their assessment and our model.</p>
<p>
<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4A</bold>
</xref> displays the receiver operating characteristic (ROC) of our model, in which the red dots represent the combinations of 1 &#x2013; specificity and the sensitivity of the clinicians&#x2019; prediction. The area under the ROC curve (AUC) of the clinical validation data was still 0.88, which demonstrated, again, that our model could be generalized to new patients. <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4B</bold>
</xref> shows the precision&#x2013;recall curve of our model, in which the red dots represent the combinations of sensitivity and the precision of the clinicians&#x2019; predictions. The predictions by clinicians varied across seniority levels. Some clinicians did better than others in their assessments. However, our model outperforms all clinician subgroups because all the dots are under the curves in both <xref ref-type="fig" rid="f4">
<bold>Figures&#xa0;4A, B</bold>
</xref>. For example, our model was able to achieve a 0.74 sensitivity and a 0.81 precision in the clinical validation and it could also flexibly achieve the other balance along the curve in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4B</bold>
</xref>. Compared with clinicians of different seniority levels, our model predicted more accurately.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Performance of clinicians&#x2019; assessment vs. our model&#x2019;s prediction. <bold>(A)</bold> A receiver operating characteristic (ROC) curve. <bold>(B)</bold> A precision&#x2013;recall curve.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fneph-03-1237804-g004.tif"/>
</fig>
</sec>
<sec id="s4" sec-type="discussion">
<label>4</label>
<title>Discussion</title>
<p>CKD is a major complication and concern for diabetic patients, and two-thirds of new kidney failure cases are due to diabetes in Singapore. Identifying high-risk patients and intervening early will help prevent kidney failure, improve patient outcomes, and utilize hospital resources more cost-effectively. Our research contributes to identifying high-risk patients in the early stages of CKD. It is also different from previous studies in the following ways.</p>
<p>First, our project was the first national-level project on kidney diseases in diabetic patients. The study used data from the national cohort that covered the entire diabetic population in Singapore. The diabetic patients&#x2019; data, which were previously scattered among diabetic registries owned by different hospitals and the MOH, were collected and consolidated by the NDD. It contains the most complete datasets of diabetic patients in Singapore and is kept up to date with new data via the BRAIN platform. The national cohort represents the diverse population in Singapore, which includes different ethnicity groups from south and east Asia. As Ta et&#xa0;al. (<xref ref-type="bibr" rid="B11">11</xref>) introduced, the BRAIN platform runs on the existing infrastructure and draws the most current data in a timely manner from the source system. It has six core elements: (i) secure access channels, which enable the BRAIN platform to connect to multiple data sources; (ii) a national health identification service, which enables the BRAIN to uniquely identify people and their records received from different health domains; (iii) an enterprise terminology service, which enables the BRAIN platform to harmonize data across multiple source systems; (iv) a de-identification service, which supports the de-identification of patient data before presenting them to researchers or prediction tools, and the re-identification of patient data when needed; (v) user groups, which include data analysts, policymakers, and stakeholders; and (vi) tools for model and dashboard developments. Tools such as Python, R, Stata, and Tableau are available on the BRAIN platform. The BRAIN platform and NDD database are readily available to stimulate more research on diabetes in the future.</p>
<p>Second, our research identified new indicative factors that improve the prediction accuracy. In addition to the eGFR and the concentration of albumin urine, which have been proven to be predictive in previously developed models, our model found that medication history (e.g., drugs for the cardiovascular system and metabolism) and other laboratory tests (e.g., hemoglobin) may also play an important role in predicting CKD progression in diabetic patients. Some features that we used have correlations with each other, for example, urine albumin and the urine albumin-to-creatinine ratio. We kept both features because they can provide added value to predict CKD progression. Removing one will decrease the model&#x2019;s performance on the validation set.</p>
<p>Third, our research developed a machine-learning algorithm that outperforms the current clinical practice and the mainstream statistical models in predicting CKD progression, as indicated in the results above. The technical validation of a 6-month shifted cohort also confirmed that our developed model is stable by showing the same good performance (with an AUC of 0.88) as the original base cohort. No calibration was needed when applying to different validation data, as our model was well trained using a representative national cohort in Singapore.</p>
<p>Fourth, our model performed better than clinicians in predicting CKD progression. More importantly, our model also reduced the variation of prediction among clinicians and provided a higher and more consistent accuracy.</p>
<p>Fifth, our model was the first clinical model endorsed by the MOH and the HALT-CKD program in Singapore to be deployed nationwide by incorporating it into the HALT-CKD program, which was established to prevent and slow down the deterioration of CKD in Singapore, with many nephrologists from public hospitals in its committee. <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref> shows how our model supports the HALT-CKD program to achieve more targeted early intervention. The cohorts of diabetic patients and their data were consolidated in the NDD. The ETL jobs have been scheduled to automatically update the NDD data daily, transform the data into the model inputs, and trigger the CKD model to generate risk scores for CKD1/2 patients once a month. The predicted score generated by the model can be flexibly banded into low, high, and very high risk to allow for the allocation of healthcare resources accordingly. The score is displayed on a diabetic patient&#x2019;s dashboard together with other relevant data.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>A workflow demonstrating how the HALT-CKD program is supported by the CKD predictive model. The cohorts of diabetic patients and their data were consolidated in the NDD (National Diabetes Database). The ETL jobs have been scheduled to automatically update the NDD data daily, transform the data into the model inputs, and trigger the CKD model to generate a risk score for the CKD stage 1 and 2 patients monthly. The risk score was banded into three levels and displayed on a diabetes patient dashboard. The users, such as clinicians, can view and print the dashboard results from the electronic health system and implement a more targeted early intervention for CKD stage 1 and 2 patients.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fneph-03-1237804-g005.tif"/>
</fig>
<p>The two dashboards, as demonstrated in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figures 2</bold>
</xref>, <xref ref-type="supplementary-material" rid="SM1">
<bold>3</bold>
</xref>, were derived and integrated into the electronic health system. The Diabetes National Dashboard is a national-level dashboard that provides a snapshot of the diabetic profile. It provides cohort insights and supports decision-making for health policies and budget planning. The other dashboard is an individual-level dashboard, which incorporates the CKD model and consolidates key information for each patient, for example, laboratory tests and appointment information. It not only saves clinicians, who normally suffer from heavy clinical workloads, from carrying out the risk assessment of CKD progression but also provides improved accuracy when identifying the patients at high risk who may otherwise be missed given the time constraints of the consultation.</p>
<p>When patients come to visit clinicians, especially those in primary care who are the first line of care, the CKD stage 1/2 patients with a high risk of CKD progression will be flagged in the dashboard. The clinicians can view and print the dashboard results from the electronic health system, use it as a tool for patient coaching, and implement more targeted early intervention for them. For example, in the HALT-CKD program, there are two phases of intervention. In phase 1, patients with a high and very high risk will have more intensive drug therapy intervention, such as more frequent drug titration. In phase 2, which is more resource intensive, patients with a very high risk will be selected for lifestyle modification. Considering the 711,800 diabetic patients aged from 20 to 79 years old in Singapore in 2021<xref ref-type="fn" rid="fn2">
<sup>2</sup>
</xref>, approximately 474,500 diabetic patients, i.e., two-thirds of diabetic patients, will end up with kidney failure. By integrating the prediction model into the clinical workflow, these patients could benefit from early CKD intervention. The enhanced performance of the model enables the identification of more patients at a high risk of kidney deterioration. In general, the individual-level dashboard can fit into the current clinical workflow and enhance the clinicians&#x2019; ability to care for patients.</p>
<p>However, our study has a number of limitations, thus leaving room for future study. First, patients&#x2019; data from private health institutes are not available although they only constitute a small portion of the healthcare data of Singapore. This affects data completeness and might have influenced the model&#x2019;s performance. Second, the impact of lifestyle factors, such as physical exercise and diet, has not been considered in the model. The model&#x2019;s performance may be further enhanced by involving lifestyle factors. Third, our study is targeted at diabetic patients in Singapore, who mainly consist of Chinese, Malay, and Indian people. The model&#x2019;s performance may decline, and the model should be retrained if applied to populations with different ethnicities.</p>
<p>In conclusion, our risk model of CKD progression for diabetic patients demonstrated its strength in prediction performance. It outperformed other predictive models and clinicians&#x2019; judgements in clinical validation. It has been endorsed by the MOH and the HALT-CKD to be deployed nationwide in Singapore to support clinicians in the assessment and medical treatment of patients in the early stages of CKD. Feedback will be collected from clinicians and the program committee to improve the predictive model further in the future. This project provided a successful example to demonstrate how an artificial intelligence (AI)-based model can be adopted to support clinical decision-making nationwide.</p>
</sec>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material</bold>
</xref>. Further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s6" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>The studies involving humans data were approved by the National Health Group Singapore. The studies were conducted in accordance with the local legislation and institutional requirements. Written informed consent for participation was not required from the participants or the participants&#x2019; legal guardians/next of kin in accordance with the national legislation and institutional requirements.</p>
</sec>
<sec id="s7" sec-type="author-contributions">
<title>Author contributions</title>
<p>ZF and WT contributed to the study design and main concept; ZF and ZW performed data analysis, developed the model, and drafted the paper. KC, MJ, and KP deployed the model. AY, GA, AL, CL, MF, and WC contributed to the data acquisition, designed the application of the model, and integrated the model into the national program. AL and CL revised the manuscript for important intellectual content. All authors read and approved the final draft before submission. WT is the guarantor of this work.</p>
</sec>
</body>
<back>
<sec id="s8" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article.</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>We appreciate the support and guidance received from the Ministry of Health (Singapore), Holistic Approach in Lowering and Tracking Chronic Kidney Disease committee, and all the participant polyclinics in Singapore: Singhealth Group Polyclinics (SHP), National Healthcare Group Polyclinics (NHGP) and National University Polyclinics (NUP).</p>
</ack>
<sec id="s9" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s10" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s11" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fneph.2023.1237804/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fneph.2023.1237804/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet_1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document"/>
</sec>
<fn-group>
<fn id="fn1">
<label>1</label>
<p>The paper used 1-year medication data before 1 January 2013, and all past data of laboratory tests before 1 January 2013 from 1 January 2005.</p>
</fn>
<fn id="fn2">
<label>2</label>
<p>Statistics is available at: <ext-link ext-link-type="uri" xlink:href="https://www.diabetesatlas.org/data/en/country/179/sg.html">https://www.diabetesatlas.org/data/en/country/179/sg.html</ext-link>
</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Remuzzi</surname> <given-names>G</given-names>
</name>
<name>
<surname>Schieppati</surname> <given-names>A</given-names>
</name>
<name>
<surname>Ruggenenti</surname> <given-names>P</given-names>
</name>
</person-group>. <article-title>Nephropathy in patients with type 2 diabetes</article-title>. <source>New Engl J Med</source> (<year>2002</year>) <volume>346</volume>(<issue>15</issue>):<page-range>1145&#x2013;51</page-range>. doi: <pub-id pub-id-type="doi">10.1056/NEJMcp011773</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fong</surname> <given-names>JMN</given-names>
</name>
<name>
<surname>Tsang</surname> <given-names>LPM</given-names>
</name>
<name>
<surname>Kwek</surname> <given-names>JL</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>W</given-names>
</name>
</person-group>. <article-title>Diabetic kidney disease in primary care</article-title>. <source>Singapore Med J</source> (<year>2020</year>) <volume>61</volume>(<issue>8</issue>):<fpage>399</fpage>. doi: <pub-id pub-id-type="doi">10.11622/smedj.2020127</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Luyckx</surname> <given-names>VA</given-names>
</name>
<name>
<surname>Al-Aly</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Bello</surname> <given-names>AK</given-names>
</name>
<name>
<surname>Bellorin-Font</surname> <given-names>E</given-names>
</name>
<name>
<surname>Carlini</surname> <given-names>RG</given-names>
</name>
<name>
<surname>Fabian</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>Sustainable development goals relevant to kidney health: an update on progress</article-title>. <source>Nat Rev Nephrology.</source> (<year>2021</year>) <volume>17</volume>(<issue>1</issue>):<fpage>15</fpage>&#x2013;<lpage>32</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41581-020-00363-6</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tangri</surname> <given-names>N</given-names>
</name>
<name>
<surname>Stevens</surname> <given-names>LA</given-names>
</name>
<name>
<surname>Griffith</surname> <given-names>J</given-names>
</name>
<name>
<surname>Tighiouart</surname> <given-names>H</given-names>
</name>
<name>
<surname>Djurdjev</surname> <given-names>O</given-names>
</name>
<name>
<surname>Naimark</surname> <given-names>D</given-names>
</name>
<etal/>
</person-group>. <article-title>A predictive model for progression of chronic kidney disease to kidney failure</article-title>. <source>Jama.</source> (<year>2011</year>) <volume>305</volume>(<issue>15</issue>):<page-range>1553&#x2013;9</page-range>. doi: <pub-id pub-id-type="doi">10.1001/jama.2011.451</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tangri</surname> <given-names>N</given-names>
</name>
<name>
<surname>Grams</surname> <given-names>ME</given-names>
</name>
<name>
<surname>Levey</surname> <given-names>AS</given-names>
</name>
<name>
<surname>Coresh</surname> <given-names>J</given-names>
</name>
<name>
<surname>Appel</surname> <given-names>LJ</given-names>
</name>
<name>
<surname>Astor</surname> <given-names>BC</given-names>
</name>
<etal/>
</person-group>. <article-title>Multinational assessment of accuracy of equations for predicting risk of kidney failure: a meta-analysis</article-title>. <source>Jama.</source> (<year>2016</year>) <volume>315</volume>(<issue>2</issue>):<page-range>164&#x2013;74</page-range>. doi: <pub-id pub-id-type="doi">10.1001/jama.2015.18202</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Grams</surname> <given-names>ME</given-names>
</name>
<name>
<surname>Sang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Ballew</surname> <given-names>SH</given-names>
</name>
<name>
<surname>Carrero</surname> <given-names>JJ</given-names>
</name>
<name>
<surname>Djurdjev</surname> <given-names>O</given-names>
</name>
<name>
<surname>Heerspink</surname> <given-names>HJ</given-names>
</name>
<etal/>
</person-group>. <article-title>Predicting timing of clinical outcomes in patients with chronic kidney disease and severely decreased glomerular filtration rate</article-title>. <source>Kidney Int</source> (<year>2018</year>) <volume>93</volume>(<issue>6</issue>):<page-range>1442&#x2013;51</page-range>. doi: <pub-id pub-id-type="doi">10.1016/j.kint.2018.01.009</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ramspek</surname> <given-names>CL</given-names>
</name>
<name>
<surname>Evans</surname> <given-names>M</given-names>
</name>
<name>
<surname>Wanner</surname> <given-names>C</given-names>
</name>
<name>
<surname>Drechsler</surname> <given-names>C</given-names>
</name>
<name>
<surname>Chesnaye</surname> <given-names>NC</given-names>
</name>
<name>
<surname>Szymczak</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>Kidney failure prediction models: a comprehensive external validation study in patients with advanced CKD</article-title>. <source>J Am Soc Nephrology: JASN.</source> (<year>2021</year>) <volume>32</volume>(<issue>5</issue>):<fpage>1174</fpage>. doi: <pub-id pub-id-type="doi">10.1681/ASN.2020071077</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Low</surname> <given-names>S</given-names>
</name>
<name>
<surname>Lim</surname> <given-names>SC</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>S</given-names>
</name>
<name>
<surname>Yeoh</surname> <given-names>LY</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>YL</given-names>
</name>
<etal/>
</person-group>. <article-title>Development and validation of a predictive model for chronic kidney disease progression in type 2 diabetes mellitus based on a 13-year study in Singapore</article-title>. <source>Diabetes Res Clin practice.</source> (<year>2017</year>) <volume>123</volume>:<fpage>49</fpage>&#x2013;<lpage>54</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.diabres.2016.11.008</pub-id>
</citation>
</ref>
<ref id="B9">
<label>9</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname> <given-names>X</given-names>
</name>
<name>
<surname>Waitman</surname> <given-names>LR</given-names>
</name>
<name>
<surname>Alan</surname> <given-names>S</given-names>
</name>
<name>
<surname>Robbins</surname> <given-names>DC</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Longitudinal risk prediction of chronic kidney disease in diabetic patients using a temporal-enhanced gradient boosting machine: retrospective cohort study</article-title>. <source>JMIR Med informatics.</source> (<year>2020</year>) <volume>8</volume>(<issue>1</issue>):<elocation-id>e15510</elocation-id>. doi: <pub-id pub-id-type="doi">10.2196/15510</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chan</surname> <given-names>L</given-names>
</name>
<name>
<surname>Nadkarni</surname> <given-names>GN</given-names>
</name>
<name>
<surname>Fleming</surname> <given-names>F</given-names>
</name>
<name>
<surname>McCullough</surname> <given-names>JR</given-names>
</name>
<name>
<surname>Connolly</surname> <given-names>P</given-names>
</name>
<name>
<surname>Mosoyan</surname> <given-names>G</given-names>
</name>
<etal/>
</person-group>. <article-title>Derivation and validation of a machine learning risk score using biomarker and electronic patient data to predict progression of diabetic kidney disease</article-title>. <source>Diabetologia.</source> (<year>2021</year>) <volume>64</volume>(<issue>7</issue>):<page-range>1504&#x2013;15</page-range>. doi: <pub-id pub-id-type="doi">10.1007/s00125-021-05444-0</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11</label>
<citation citation-type="confproc">
<person-group person-group-type="editor">
<name>
<surname>Ta</surname> <given-names>WA</given-names>
</name>
<name>
<surname>Goh</surname> <given-names>HL</given-names>
</name>
<name>
<surname>Tan</surname> <given-names>CS</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Aung</surname> <given-names>KCY</given-names>
</name>
<name>
<surname>Teoh</surname> <given-names>ZW</given-names>
</name>
<etal/>
</person-group> eds. (<year>2018</year>). <article-title>Development and implementation of nationwide predictive model for admission prevention: System architecture &amp; machine learning</article-title>, in: <conf-name>2018 IEEE EMBS International Conference on Biomedical &amp; Health Informatics (BHI)</conf-name>. <publisher-loc>New Jersey, USA: IEEE</publisher-loc>.</citation>
</ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fu</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Applying supervised machine learning in bioinformatics analysis</article-title>. <source>Horizons Comput Sci Res</source> (<year>2015</year>) <volume>12</volume>:<fpage>1</fpage>.</citation>
</ref>
<ref id="B13">
<label>13</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Goh</surname> <given-names>HL</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>TY</given-names>
</name>
<name>
<surname>Zou</surname> <given-names>L</given-names>
</name>
<name>
<surname>Ngoh</surname> <given-names>CPY</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Koh</surname> <given-names>JZ</given-names>
</name>
<etal/>
</person-group>. <source>AI Prognostication Tool for Severe Community-Acquired Pneumonia and Covid-19 Respiratory Infections</source>. <publisher-loc>San Diego, CA</publisher-loc>: <publisher-name>KDD</publisher-name> (<year>2020</year>).</citation>
</ref>
<ref id="B14">
<label>14</label>
<citation citation-type="confproc">
<person-group person-group-type="editor">
<name>
<surname>Chen</surname> <given-names>T</given-names>
</name>
<name>
<surname>Guestrin</surname> <given-names>C</given-names>
</name>
</person-group> eds. (<year>2016</year>). <article-title>Xgboost: A scalable tree boosting system</article-title>, in: <conf-name>Proceedings of the 22nd acm sigkdd international conference on knowledge discovery and data mining</conf-name>. <publisher-loc>New York, USA: Association for Computing Machinery</publisher-loc>.</citation>
</ref>
<ref id="B15">
<label>15</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Ong</surname> <given-names>CLJ</given-names>
</name>
<name>
<surname>Fu</surname> <given-names>Z</given-names>
</name>
</person-group>. <article-title>AI models to assist vancomycin dosage titration</article-title>. <source>Front Pharmacol</source> (<year>2022</year>) <volume>13</volume>. doi: <pub-id pub-id-type="doi">10.1055/s-0041-1742095</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Low</surname> <given-names>S</given-names>
</name>
<name>
<surname>Tai</surname> <given-names>ES</given-names>
</name>
<name>
<surname>Yeoh</surname> <given-names>LY</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>YL</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>JJ</given-names>
</name>
<name>
<surname>Tan</surname> <given-names>KHX</given-names>
</name>
<etal/>
</person-group>. <article-title>Onset and progression of kidney disease in type 2 diabetes among multi-ethnic Asian population</article-title>. <source>J Diabetes its Complications.</source> (<year>2016</year>) <volume>30</volume>(<issue>7</issue>):<page-range>1248&#x2013;54</page-range>. doi: <pub-id pub-id-type="doi">10.1016/j.jdiacomp.2016.05.020</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname> <given-names>C-C</given-names>
</name>
<name>
<surname>Li</surname> <given-names>C-I</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>C-S</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>W-Y</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>C-H</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>S-Y</given-names>
</name>
<etal/>
</person-group>. <article-title>Development and validation of a risk prediction model for end-stage renal disease in patients with type 2 diabetes</article-title>. <source>Sci Rep</source> (<year>2017</year>) <volume>7</volume>(<issue>1</issue>):<fpage>1</fpage>&#x2013;<lpage>13</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41598-017-09243-9</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>