<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Immunol.</journal-id>
<journal-title>Frontiers in Immunology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Immunol.</abbrev-journal-title>
<issn pub-type="epub">1664-3224</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fimmu.2023.1224631</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Immunology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Machine learning in predicting <italic>T</italic>-score in the Oxford classification system of IgA nephropathy</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Xu</surname>
<given-names>Lin-Lin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2315864"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Di</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2320068"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Weng</surname>
<given-names>Hao-Yi</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1901796"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Li-Zhong</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Ruo-Yan</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Gang</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Shi</surname>
<given-names>Su-Fang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1267844"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Li-Jun</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhong</surname>
<given-names>Xu-Hui</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/181077"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Hong</surname>
<given-names>Shen-Da</given-names>
</name>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1192471"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Duan</surname>
<given-names>Li-Xin</given-names>
</name>
<xref ref-type="aff" rid="aff7">
<sup>7</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lv</surname>
<given-names>Ji-Cheng</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1465544"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Zhou</surname>
<given-names>Xu-Jie</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/304607"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Hong</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/228552"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Renal Division, Peking University First Hospital, Kidney Genetics Center, Peking University Institute of Nephrology, Key Laboratory of Renal Disease, Ministry of Health of China, Key Laboratory of Chronic Kidney Disease Prevention and Treatment, Peking University, Ministry of Education</institution>, <addr-line>Beijing</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Hunan Provincial Key Lab on Bioinformatics, School of Computer Science and Engineering, Central South University</institution>, <addr-line>Changsha</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>WeGene, Shenzhen Zaozhidao Technology</institution>, <addr-line>Shenzhen</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Shenzhen WeGene Clinical Laboratory</institution>, <addr-line>Shenzhen</addr-line>, <country>China</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Department of Pediatrics, Peking University First Hospital</institution>, <addr-line>Beijing</addr-line>, <country>China</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Institute of Medical Technology, Health Science Center of Peking University</institution>, <addr-line>Beijing</addr-line>, <country>China</country>
</aff>
<aff id="aff7">
<sup>7</sup>
<institution>The Sichuan Provincial Key Laboratory for Human Disease Gene Study, Research Unit for Blindness Prevention of Chinese Academy of Medical Sciences (2019RU026), Sichuan Academy of Medical Sciences and Sichuan Provincial People&#x2019;s Hospital, University of Electronic Science and Technology of China</institution>, <addr-line>Chengdu</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Huji Xu, Tsinghua University, China</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Youhua Xu, Macau University of Science and Technology, Macao SAR, China; Jinxia Zhao, Peking University Third Hospital, China</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Xu-Jie Zhou, <email xlink:href="mailto:zhouxujie@bjmu.edu.cn">zhouxujie@bjmu.edu.cn</email>
</p>
</fn>
<fn fn-type="equal" id="fn003">
<p>&#x2020;These authors have contributed equally to this work and share first authorship</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>04</day>
<month>08</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>14</volume>
<elocation-id>1224631</elocation-id>
<history>
<date date-type="received">
<day>18</day>
<month>05</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>21</day>
<month>07</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Xu, Zhang, Weng, Wang, Chen, Chen, Shi, Liu, Zhong, Hong, Duan, Lv, Zhou and Zhang</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Xu, Zhang, Weng, Wang, Chen, Chen, Shi, Liu, Zhong, Hong, Duan, Lv, Zhou and Zhang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Background</title>
<p>Immunoglobulin A nephropathy (IgAN) is one of the leading causes of end-stage kidney disease (ESKD). Many studies have shown the significance of pathological manifestations in predicting the outcome of patients with IgAN, especially <italic>T</italic>-score of Oxford classification. Evaluating prognosis may be hampered in patients without renal biopsy.</p>
</sec>
<sec>
<title>Methods</title>
<p>A baseline dataset of 690 patients with IgAN and an independent follow-up dataset of 1,168 patients were used as training and testing sets to develop the pathology <italic>T</italic>-score prediction (<italic>T</italic>
<sub>pre</sub>) model based on the stacking algorithm, respectively. The 5-year ESKD prediction models using clinical variables (base model), clinical variables and real pathological <italic>T</italic>-score (base model plus <italic>T</italic>
<sub>bio</sub>), and clinical variables and <italic>T</italic>
<sub>pre</sub> (base model plus <italic>T</italic>
<sub>pre</sub>) were developed separately in 1,168 patients with regular follow-up to evaluate whether <italic>T</italic>
<sub>pre</sub> could assist in predicting ESKD. In addition, an external validation set consisting of 355 patients was used to evaluate the performance of the 5-year ESKD prediction model using <italic>T</italic>
<sub>pre</sub>.</p>
</sec>
<sec>
<title>Results</title>
<p>The features selected by AUCRF for the <italic>T</italic>
<sub>pre</sub> model included age, systolic arterial pressure, diastolic arterial pressure, proteinuria, eGFR, serum IgA, and uric acid. The AUC of the <italic>T</italic>
<sub>pre</sub> was 0.82 (95% CI: 0.80&#x2013;0.85) in an independent testing set. For the 5-year ESKD prediction model, the AUC of the base model was 0.86 (95% CI: 0.75&#x2013;0.97). When the <italic>T</italic>
<sub>bio</sub> was added to the base model, there was an increase in AUC [from 0.86 (95% CI: 0.75&#x2013;0.97) to 0.92 (95% CI: 0.85&#x2013;0.98); <italic>P</italic> = 0.03]. There was no difference in AUC between the base model plus <italic>T</italic>
<sub>pre</sub> and the base model plus <italic>T</italic>
<sub>bio</sub> [0.90 (95% CI: 0.82&#x2013;0.99) <italic>vs</italic>. 0.92 (95% CI: 0.85&#x2013;0.98), <italic>P</italic> = 0.52]. The AUC of the 5-year ESKD prediction model using <italic>T</italic>
<sub>pre</sub> was 0.93 (95% CI: 0.87&#x2013;0.99) in the external validation set.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>A pathology <italic>T</italic>-score prediction (<italic>T</italic>
<sub>pre</sub>) model using routine clinical characteristics was constructed, which could predict the pathological severity and assist clinicians to predict the prognosis of IgAN patients lacking kidney pathology scores.</p>
</sec>
</abstract>
<kwd-group>
<kwd>IgA nephropathy</kwd>
<kwd>machine learning</kwd>
<kwd>Oxford classification system</kwd>
<kwd>prediction model</kwd>
<kwd>end-stage kidney disease</kwd>
</kwd-group>
<counts>
<fig-count count="4"/>
<table-count count="5"/>
<equation-count count="0"/>
<ref-count count="43"/>
<page-count count="11"/>
<word-count count="6382"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Autoimmune and Autoinflammatory Disorders: Autoinflammatory Disorders</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Immunoglobulin A (IgA) nephropathy (IgAN) is one of the most common forms of glomerulonephritis worldwide. The clinical manifestations are heterogeneous, ranging from asymptomatic proteinuria or microscopic hematuria to rapid deterioration in kidney function (<xref ref-type="bibr" rid="B1">1</xref>). It was reported that approximately 20%&#x2013;30% of patients with IgAN would progress to kidney failure within 20 years (<xref ref-type="bibr" rid="B2">2</xref>). Therefore, early identification of high-risk patients with IgAN prone to ESKD is beneficial for early intervention in delaying disease progression. Great endeavors have been taken by many researchers to search for the risk factors for developing ESKD in patients with IgAN. Generally accepted risk factors affecting the progression of IgAN included decreased glomerular filtration rate (GFR), 24-h proteinuria &gt;1 g/day, hypertension, and renal pathological manifestations (<xref ref-type="bibr" rid="B3">3</xref>&#x2013;<xref ref-type="bibr" rid="B9">9</xref>). These risk factors have been used to build various scoring models for predicting the prognosis of IgAN based on traditional statistical methods (<xref ref-type="bibr" rid="B4">4</xref>, <xref ref-type="bibr" rid="B10">10</xref>&#x2013;<xref ref-type="bibr" rid="B14">14</xref>). However, these scoring models are constructed by the small sample sizes and different pathological scoring criteria, which may affect the accuracy and generalization of these scoring models. Moreover, the interactions between the characteristics and their effect on ESKD, the non-linear relationship among predictors, and the effects of therapeutic regimens make the interpretation of the data more complicated.</p>
<p>Machine learning, as a branch discipline of artificial intelligence, has obvious advantages in processing high-dimensional and sparse data. Machine learning algorithms can learn the relationship between input features and target outcomes as well as the relationship between features through a large amount of training data. Several studies have successfully constructed ESKD prediction models for patients with IgAN through machine learning algorithms (<xref ref-type="bibr" rid="B15">15</xref>&#x2013;<xref ref-type="bibr" rid="B20">20</xref>). By comparing the performance of traditional statistical methods and different machine learning algorithms in predicting ESKD or halving of estimated glomerular filtration rate from baseline, Chen et&#xa0;al. showed that the XGBoost algorithm performed best (<xref ref-type="bibr" rid="B16">16</xref>). XGBoost, as a machine learning algorithm, assembles the weak prediction models to construct a prediction model (<xref ref-type="bibr" rid="B16">16</xref>, <xref ref-type="bibr" rid="B21">21</xref>). Several studies have tried to construct event prediction models for a specific clinical outcome based on the XGBoost algorithm (<xref ref-type="bibr" rid="B22">22</xref>, <xref ref-type="bibr" rid="B23">23</xref>). However, no matter whether it was a traditional prediction formula or a machine learning-based predictive model in IgAN, pathology scores showed consistently significant weighting among many parameters (<xref ref-type="bibr" rid="B15">15</xref>, <xref ref-type="bibr" rid="B16">16</xref>, <xref ref-type="bibr" rid="B19">19</xref>, <xref ref-type="bibr" rid="B24">24</xref>). In 2009, the Oxford classification, an international consensus, was proposed to classify IgA nephropathy based on histopathological features to predict its prognosis and guide clinical treatment. The revised Oxford classification in 2017 divided IgAN into five categories, namely, &#x201c;(1) mesangial hypercellularity (M); (2) endocapillary hypercellularity (E); (3) segmental glomerulosclerosis (S); (4) tubular atrophy/interstitial fibrosis (T); (5) cellular/fibrocellular crescents (C)&#x201d; (<xref ref-type="bibr" rid="B25">25</xref>), which were shown to be the independent predictors in predicting renal outcome (<xref ref-type="bibr" rid="B24">24</xref>, <xref ref-type="bibr" rid="B26">26</xref>). Since 2009, over 20 validation studies have tried to prove the predictive value of the MEST scores in some retrospective cohorts of patients with IgAN, which provided consistent evidence that the mesangial hypercellularity (M), segmental glomerulosclerosis (S), and tubular atrophy/interstitial fibrosis (T) each reliably provided prognostic value by univariate analysis (<xref ref-type="bibr" rid="B26">26</xref>), but T lesion was suggested to be the strongest predictor of renal survival. Hernan et&#xa0;al. summarized the results of these studies and found that M was of independent prognostic value in 5 out of 19, E in 4 out of 19, S in 7 out of 19, and T in 13 out of 19 (<xref ref-type="bibr" rid="B26">26</xref>). The <italic>C</italic>-score was adopted in the revised classification system in 2017, and three of the five prognostic studies on IgA nephropathy showed that <italic>C</italic>-score was associated with poor prognosis (<xref ref-type="bibr" rid="B26">26</xref>&#x2013;<xref ref-type="bibr" rid="B28">28</xref>). In the constructed IgAN prognosis prediction models, it was observed that the T lesions showed greater weight in predicting prognosis compared with many other clinical and pathological parameters (<xref ref-type="bibr" rid="B14">14</xref>, <xref ref-type="bibr" rid="B16">16</xref>). For example, in the prognosis prediction model constructed by Chen et&#xa0;al., there were three indexes that can be integrated to predict ESKD, namely, <italic>T</italic>, global sclerosis, and urine protein, among which the <italic>T</italic>-score ranked first in the weight of importance (<xref ref-type="bibr" rid="B16">16</xref>). However, the <italic>T</italic>-score is derived from the kidney biopsy, an invasive manipulation, sometimes refused by patients and cannot be repeated in clinical routine for detecting disease progression. Hence, it is of great significance to explore whether pathological T lesions can be predicted by the patient&#x2019;s clinical variables at the same time.</p>
<p>The purposes of our study are 1) to construct a pathology <italic>T</italic>-score (<italic>T</italic>
<sub>pre</sub>) prediction model based on the patient&#x2019;s clinical variables at the same time which may be able to predict whether there is a pathological T lesion and 2) to evaluate whether the predicted T can be used to assist in predicting ESKD.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>Study participants</title>
<p>This study had two independent datasets. Dataset 1, a baseline dataset without follow-up data, comprised 690 patients with IgAN. These patients received the kidney biopsy in our center but returned to local for follow-up. Dataset 2, a follow-up dataset (PKU-IgAN cohort), included 1,808 patients with IgAN who were registered and with long-term follow-up in the Peking University First Hospital IgAN database from 1997 to 2020 (<xref ref-type="bibr" rid="B29">29</xref>). All patients with IgAN were diagnosed based on the histologic and immunofluorescence study of the renal biopsy, and those with &lt;8 glomeruli per biopsy section were excluded (<xref ref-type="bibr" rid="B29">29</xref>). After excluding 243 patients without blood lipid data, 28 patients presented at younger than 16 years of age, and 14 patients presented acute kidney failure, 1,523 patients in dataset 2 were finally enrolled in this study, consisting of 1,168 patients with Oxford MEST-C scores and 355 patients lacking Oxford MEST-C scores.</p>
<p>Finally, a total of 690 patients in dataset 1 and 1,168 patients with Oxford MEST-C scores in dataset 2 were enrolled in our study as the modeling group, and 355 patients without Oxford MEST-C scores in dataset 2 were enrolled in this study as the external validation group (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>).</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>The flowchart of this study. WSVM, weighted support vector machine; WRF, weighted random forest; WLR, weighted logistic regression; AKI, acute kidney injury.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-14-1224631-g001.tif"/>
</fig>
<p>All clinical characteristics were collected at the time of the renal biopsy. The estimated glomerular filtration rate (eGFR) was calculated using the Chronic Kidney Disease Epidemiology Collaboration (CKD-EPI) formula (<xref ref-type="bibr" rid="B30">30</xref>). Renal biopsies were categorized according to established criteria for the Oxford MEST-C scoring system (<xref ref-type="bibr" rid="B24">24</xref>, <xref ref-type="bibr" rid="B26">26</xref>, <xref ref-type="bibr" rid="B31">31</xref>). Mean arterial pressure (MAP, mm Hg) was defined as diastolic pressure plus a third of the pulse pressure. The end-stage kidney disease (ESKD) was defined as eGFR &lt;15 ml/min/1.73 m<sup>2</sup>, dialysis, or kidney transplantation. Our study was approved by the Ethics Committee of Peking University First Hospital (IRB number 2020Y197). Written informed consent was provided by all participants.</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Pathology <italic>T</italic>-score prediction model</title>
<p>The pathology <italic>T</italic>-score prediction (<italic>T</italic>
<sub>pre</sub>) model, constructed by the stacking algorithm, was used to predict whether IgAN patients would have T lesions (yes or no). The stacking algorithm is an integrated machine learning algorithm that can summarize several models and predict new observations. It utilizes the prediction of a collection of models as input for training a second-level model. This second-level model aims to find the best combination of the prediction of first-level models. Stacking can shield the capabilities of a range of well-performing models so that a better output prediction model can be achieved (<xref ref-type="bibr" rid="B32">32</xref>). In our study, we combined three machine learning algorithms, namely, support vector machine (SVM), random forest (RF), and logistic regression as first-level models, and then logistic regression as the second-level model to output the final probability of the binary <italic>T</italic>-score (with or without tubular atrophy/interstitial fibrosis, <italic>T</italic>
<sub>pre</sub>).</p>
<p>The input variables used in this model were chosen by AUCRF (<xref ref-type="bibr" rid="B33">33</xref>), a method using the random forest to find the optimal set for prediction. Variables entered into the AUCRF included age, sex, body mass index, systolic arterial pressure, diastolic arterial pressure, mean arterial pressure, hypertension, eGFR, proteinuria, microhematuria, history of gross hematuria, serum IgA, serum uric acid, serum triglycerides, total cholesterol, high-density lipoprotein, and low-density lipoprotein.</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Five-year ESKD prediction model</title>
<p>Several studies have demonstrated the value of tubular atrophy/interstitial fibrosis (T) in predicting ESKD in patients with IgAN (<xref ref-type="bibr" rid="B16">16</xref>, <xref ref-type="bibr" rid="B19">19</xref>, <xref ref-type="bibr" rid="B24">24</xref>, <xref ref-type="bibr" rid="B34">34</xref>, <xref ref-type="bibr" rid="B35">35</xref>). To evaluate whether the predicted <italic>T</italic>-score could help predict ESKD and how effective it was, we constructed a 5-year ESKD prediction model based on the XGBoost algorithm. To illustrate the significance of tubular atrophy/interstitial fibrosis in predicting ESKD, we first constructed a 5-year ESKD prediction model with only clinical variables as input variables (base model). Then, the 5-year ESKD prediction model using clinical variables and the real pathological T lesions score (<italic>T</italic>
<sub>bio</sub>, T0 was assigned 0, T1 and T2 were assigned 1) was also developed (base model plus <italic>T</italic>
<sub>bio</sub>) to evaluate the additive value of atrophy/interstitial fibrosis (T) in predicting ESKD. Finally, to evaluate whether the value of <italic>T</italic>
<sub>pre</sub> in predicting ESKD of patients with IgAN was consistent with real pathological T lesions (<italic>T</italic>
<sub>bio</sub>) when the base model plus <italic>T</italic>
<sub>bio</sub> was trained in the training set, the <italic>T</italic>
<sub>bio</sub> of the testing set was replaced by the corresponding <italic>T</italic>
<sub>pre</sub> predicted by the pathology <italic>T</italic>-score prediction model and then the testing set was used to evaluate the model performance (the base model plus <italic>T</italic>
<sub>pre</sub>). For the base model plus <italic>T</italic>
<sub>pre</sub>, the purpose of training the model using real pathological <italic>T</italic>-score (<italic>T</italic>
<sub>bio</sub>) was for the model to learn the true value of T for predicting ESKD.</p>
<p>XGBoost is a kind of ensemble of the decision tree, whose advantages include higher-order interactions and complex non-linear relationships between the model features and the outcome (<xref ref-type="bibr" rid="B21">21</xref>). It has been shown to achieve impressive performance in predicting renal failure risk and provide explanations for variables by ranking their importance (<xref ref-type="bibr" rid="B16">16</xref>, <xref ref-type="bibr" rid="B34">34</xref>). We also applied other machine learning algorithms to our data set for evaluating whether the predicted T could be used in ESKD prediction models based on different algorithms, including RF, penalized regression, artificial neural network (ANN), and SVM.</p>
<p>Characteristics selected by the Cox proportional hazards model were collected at the time of the renal biopsy at enrollment [age, sex, systolic arterial pressure, diastolic arterial pressure, proteinuria, eGFR, serum IgA, serum uric acid, serum triglycerides, total cholesterol, low-density lipoprotein, and history of previous use of renin&#x2013;angiotensin system (RAS) inhibitors and immunosuppressants as well as pathological T lesions], whereas the binary outcome (ESKD within 5 years after diagnostic kidney biopsy, yes or no) represented the output data. For these variables, we imputed missing values to the means for continuous characteristics and the mode for categorical characteristics. Because of missing information on serum triglycerides, total cholesterol, and low-density lipoprotein in some cases, 243 patients without blood lipid data were excluded to avoid inaccuracy due to missing value filling (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>).</p>
<p>To confirm that the <italic>T</italic>
<sub>pre</sub> can be used in the ESKD prediction model at multiple levels, we also constructed a lifetime ESKD prediction model based on XGBoost. The process and approach were the same as building the 5-year ESKD prediction model. The primary outcome was time-to-event ESKD. The survival time for the kidney without ESKD event was calculated from the kidney biopsy to the last follow-up.</p>
<p>The XGBoost was allowed to generate boosting trees at most 110 times, and the maximum depth of each tree was constrained to 5. To avoid overfitting, we further set the L2 regularization term on weights as 1 and stop training if the performance did not improve by more than 15 rounds. At last, the optimal prediction model parameters and architectures were selected by the five-fold cross-validation.</p>
<p>The patients of dataset 2 without Oxford MEST-C scores combined with the corresponding <italic>T</italic>
<sub>pre</sub> were used as an additional external validation set to evaluate the performance of the ESKD prediction model using <italic>T</italic>
<sub>pre</sub>.</p>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Statistical analysis</title>
<p>The sociodemographic and clinical variables were calculated and expressed as the mean &#xb1; standard deviation for variables with approximately symmetrical distributions and as median (interquartile range 25th&#x2013;75th percentile) for variables with skewed distribution. All categorical variables are expressed as frequencies and percentages. Univariate analyses based on the Cox proportional hazards model (<xref ref-type="bibr" rid="B36">36</xref>) were conducted to evaluate the association between the baseline clinical characteristics and ESKD event. Clinical characteristics associated with ESKD event in univariate analysis (<italic>P</italic> &lt; 0.05) or if they were clinically relevant were used as input features of the 5-year ESKD prediction model.</p>
<p>For predicting 5-year ESKD status (yes or no) and <italic>T</italic>-score (0 or 1), the performance of the models was assessed by calculating the accuracy, sensitivity, specificity, and area under the receiver operating characteristic (ROC) curve (AUC). For predicting lifetime ESKD risk, we quantify the performance of the model by concordance statistic (<italic>C</italic>-statistic), which is a general concept of the area under the curve (AUC) for time-to-event survival data (<xref ref-type="bibr" rid="B37">37</xref>). The <italic>C</italic>-statistic compares the rank of predicting probability and the rank of the survival time in the real world. The calibration ability of the models was assessed by the Hosmer&#x2013;Lemeshow test and calibration scatter plot, in which <italic>P</italic>-value &gt;0.05 indicated no very significant difference between the predicted probability predicted by the model and the true outcome frequencies during a certain time period. SPSS version 26.0 software and R 3.6.3 were used for the statistical analysis. All <italic>P</italic>-values were two-tailed, and <italic>P &lt;</italic>0.05 was considered statistically significant.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results</title>
<sec id="s3_1">
<label>3.1</label>
<title>Characteristics of the study participants</title>
<p>The clinical characteristics of 690 patients with IgAN in dataset 1 are shown in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>. The mean age of these patients was 32.38 &#xb1; 11.32 years at the time of renal biopsy. The male-to-female ratio was 1.2:1. The mean arterial pressure was 94.44 &#xb1; 14.02&#xa0;mm Hg. The median value of eGFR was 84.66 (range, 63.32&#x2013;107.50) ml/min per 1.73 m<sup>2</sup>, and daily proteinuria was 1.38 (range, 0.66&#x2013;2.89) g/day.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Baseline characteristics of patients with IgAN enrolled in this study to construct the pathology <italic>T</italic>-score prediction model at the time of kidney biopsy.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="left">Characteristics</th>
<th valign="top" align="center">Training set</th>
<th valign="top" align="center">Testing set</th>
<th valign="top" align="center">
<italic>P</italic>-value</th>
</tr>
<tr>
<th valign="top" align="center">(dataset 1)</th>
<th valign="top" align="center">(dataset 2 with MEST-C scores)</th>
<th valign="top" align="center"/>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Patients (<italic>n</italic>)</td>
<td valign="top" align="center">690</td>
<td valign="top" align="center">1,168</td>
<td valign="top" align="center"/>
</tr>
<tr>
<td valign="top" align="left">Age at biopsy, years</td>
<td valign="top" align="center">32.38 &#xb1; 11.32</td>
<td valign="top" align="center">35.10 &#xb1; 11.73</td>
<td valign="top" align="center">1.00 &#xd7; 10<sup>&#x2212;6</sup>
</td>
</tr>
<tr>
<td valign="top" align="left">Sex (male/female)</td>
<td valign="top" align="center">370/320</td>
<td valign="top" align="center">583/585</td>
<td valign="top" align="center">0.12</td>
</tr>
<tr>
<td valign="top" align="left">Systolic blood pressure, mm Hg</td>
<td valign="top" align="center">124.77 &#xb1; 18.28</td>
<td valign="top" align="center">123.67 &#xb1; 15.09</td>
<td valign="top" align="center">0.18</td>
</tr>
<tr>
<td valign="top" align="left">Diastolic blood pressure, mm Hg</td>
<td valign="top" align="center">79.28 &#xb1; 13.11</td>
<td valign="top" align="center">78.54 &#xb1; 11.00</td>
<td valign="top" align="center">0.22</td>
</tr>
<tr>
<td valign="top" align="left">Mean arterial pressure, mm Hg</td>
<td valign="top" align="center">94.44 &#xb1; 14.02</td>
<td valign="top" align="center">93.59 &#xb1; 11.42</td>
<td valign="top" align="center">0.17</td>
</tr>
<tr>
<td valign="top" align="left">eGFR, ml/min per 1.73 m<sup>2</sup>
</td>
<td valign="top" align="center">84.66 (63.32&#x2013;107.50)</td>
<td valign="top" align="center">85.91 (60.94&#x2013;107.23)</td>
<td valign="top" align="center">0.69</td>
</tr>
<tr>
<td valign="top" align="left">Proteinuria, g/day</td>
<td valign="top" align="center">1.38 (0.66&#x2013;2.89)</td>
<td valign="top" align="center">1.27 (0.66&#x2013;2.45)</td>
<td valign="top" align="center">0.10</td>
</tr>
<tr>
<td valign="top" align="left">Serum IgA level, g/l</td>
<td valign="top" align="center">3.13 &#xb1; 1.21</td>
<td valign="top" align="center">3.29 &#xb1; 1.20</td>
<td valign="top" align="center">0.01</td>
</tr>
<tr>
<td valign="top" align="left">Uric acid, &#x3bc;mol/l</td>
<td valign="top" align="center">347.10 &#xb1; 114.95</td>
<td valign="top" align="center">367.63 &#xb1; 101.86</td>
<td valign="top" align="center">1.52 &#xd7; 10<sup>&#x2212;4</sup>
</td>
</tr>
<tr>
<td valign="top" align="left">Triglycerides, mmol/l</td>
<td valign="top" align="center">1.61 (1.10&#x2013;2.38)</td>
<td valign="top" align="center">1.62 (1.07&#x2013;2.42)</td>
<td valign="top" align="center">0.64</td>
</tr>
<tr>
<td valign="top" align="left">Total cholesterol, mmol/l</td>
<td valign="top" align="center">4.70 (3.99&#x2013;5.61)</td>
<td valign="top" align="center">4.77 (4.02&#x2013;5.67)</td>
<td valign="top" align="center">0.23</td>
</tr>
<tr>
<td valign="top" align="left">Low-density lipoprotein, mmol/l</td>
<td valign="top" align="center">2.71 (2.12&#x2013;3.33)</td>
<td valign="top" align="center">2.75 (2.23&#x2013;3.38)</td>
<td valign="top" align="center">0.19</td>
</tr>
<tr>
<td valign="top" align="left">Renal biopsy, <italic>n</italic>/<italic>n</italic> (%)</td>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
</tr>
<tr>
<td valign="top" align="left">Mesangial (M) 1</td>
<td valign="top" align="center">560/690 (81.16%)</td>
<td valign="top" align="center">461/1,168 (39.47%)</td>
<td valign="top" align="center">3.37 &#xd7; 10<sup>&#x2212;68</sup>
</td>
</tr>
<tr>
<td valign="top" align="left">Endocapillary (E) 1</td>
<td valign="top" align="center">128/690 (18.55%)</td>
<td valign="top" align="center">400/1,168 (34.25%)</td>
<td valign="top" align="center">4.23 &#xd7; 10<sup>&#x2212;13</sup>
</td>
</tr>
<tr>
<td valign="top" align="left">Glomerular sclerosis (S) 1</td>
<td valign="top" align="center">225/690 (32.61%)</td>
<td valign="top" align="center">733/1,168 (62.76%)</td>
<td valign="top" align="center">3.33 &#xd7; 10<sup>&#x2212;36</sup>
</td>
</tr>
<tr>
<td valign="top" align="left">Tubulointerstitial damage (T1+T2)</td>
<td valign="top" align="center">182/690 (26.38%)</td>
<td valign="top" align="center">392/1,168 (33.56%)</td>
<td valign="top" align="center">1.00 &#xd7; 10<sup>&#x2212;3</sup>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Data are expressed as mean &#xb1; SD, median (interquartile range), absolute, and percent frequency.</p>
</fn>
<fn>
<p>IgAN, immunoglobulin A nephropathy; eGFR, estimated glomerular filtration rate.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>For the 1,168 follow-up patients with Oxford MEST-C scores in dataset 2, the mean age was 35.10 &#xb1; 11.73 years at the time of renal biopsy. The male-to-female ratio was 1:1. The mean arterial pressure was 93.59 &#xb1; 11.42&#xa0;mm Hg. The eGFR was 85.91 (range, 60.94&#x2013;107.23) ml/min per 1.73 m<sup>2</sup>, and daily proteinuria was 1.27 (range, 0.66&#x2013;2.45) g/day (<xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>). For the variables used to train the pathology <italic>T</italic>-score prediction (<italic>T</italic>
<sub>pre</sub>) model, there were no statistically significant differences in clinical parameters between dataset 1 and dataset 2 except for age (32.38 &#xb1; 11.32 <italic>vs</italic>. 35.10 &#xb1; 11.73, <italic>P</italic> = 1.00 &#xd7; 10<sup>&#x2212;6</sup>), serum IgA level (3.13 &#xb1; 1.21 <italic>vs</italic>. 3.29 &#xb1; 1.20, <italic>P</italic> = 0.01), and serum uric acid level (347.10 &#xb1; 114.95 <italic>vs</italic>. 367.63 &#xb1; 101.86, <italic>P</italic> = 1.52 &#xd7; 10<sup>&#x2212;4</sup>). Among these, 158 patients (13.53%) had reached the event of ESKD during the median 67.5-month follow-up. The unadjusted hazard ratios (HRs) between the different variables and ESKD are reported in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>. The risk of ESKD significantly increased for every 10.0&#xa0;mm Hg increase in the MAP [HR: 1.34, 95% confidence interval (CI): 1.18&#x2013;1.53, <italic>P</italic> = 1.10 &#xd7; 10<sup>&#x2212;5</sup>] and increased for every 1.0 g/day in the daily proteinuria (HR: 1.10, 95% CI: 1.05&#x2013;1.15, <italic>P</italic> = 1.60 &#xd7; 10<sup>&#x2212;5</sup>). For each ml/min per 1.73 m<sup>2</sup> decrease in eGFR, the risk of ESKD increased by 4% (HR: 0.96, 95% CI: 0.96&#x2013;0.97, <italic>P</italic> = 1.24 &#xd7; 10<sup>&#x2212;27</sup>). For each mg/dl increase in uric acid, the risk of ESKD increased by 38% (HR: 1.38, 95% CI: 1.29&#x2013;1.49, <italic>P</italic> = 1.47 &#xd7; 10<sup>&#x2212;19</sup>). Moreover, there was the strongest association between the risk of ESKD and the presence of tubulointerstitial lesions (HR: 3.34, 95% CI: 2.73&#x2013;4.07, <italic>P</italic> = 1.72 &#xd7; 10<sup>&#x2212;32</sup>).</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Risk estimated by Cox proportional hazard model for ESKD in patients of dataset 2 with Oxford MEST-C scores.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Risk factor</th>
<th valign="top" align="center">Non-ESKD (<italic>n</italic> = 1,010)</th>
<th valign="top" align="center">ESKD (<italic>n</italic> = 158)</th>
<th valign="top" align="center">
<italic>P</italic>-value</th>
<th valign="top" align="center">HR (95% CI)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Age, years</td>
<td valign="top" align="center">35.21 &#xb1; 11.84</td>
<td valign="top" align="center">34.41 &#xb1; 11.02</td>
<td valign="top" align="center">0.55</td>
<td valign="top" align="center">1.00 (0.98&#x2013;1.01)</td>
</tr>
<tr>
<td valign="top" align="left">Male (%)</td>
<td valign="top" align="center">482 (47.72%)</td>
<td valign="top" align="center">101 (63.92%)</td>
<td valign="top" align="center">1.96 &#xd7; 10<sup>&#x2212;4</sup>
</td>
<td valign="top" align="center">1.85 (1.34&#x2013;2.57)</td>
</tr>
<tr>
<td valign="top" align="left">Systolic arterial pressure, mm Hg</td>
<td valign="top" align="center">123.00 &#xb1; 14.70</td>
<td valign="top" align="center">128.02 &#xb1; 16.77</td>
<td valign="top" align="center">3.00 &#xd7; 10<sup>&#x2212;6</sup>
</td>
<td valign="top" align="center">1.02 (1.01&#x2013;1.03)</td>
</tr>
<tr>
<td valign="top" align="left">Diastolic arterial pressure, mm Hg</td>
<td valign="top" align="center">78.13 &#xb1; 10.66</td>
<td valign="top" align="center">81.18 &#xb1; 12.71</td>
<td valign="top" align="center">2.88 &#xd7; 10<sup>&#x2212;4</sup>
</td>
<td valign="top" align="center">1.03 (1.01&#x2013;1.04)</td>
</tr>
<tr>
<td valign="top" align="left">Mean arterial pressure, mm Hg</td>
<td valign="top" align="center">93.09 &#xb1; 11.06</td>
<td valign="top" align="center">96.80 &#xb1; 13.10</td>
<td valign="top" align="center">1.10 &#xd7; 10<sup>&#x2212;5</sup>
</td>
<td valign="top" align="center">1.03 (1.02&#x2013;1.04)</td>
</tr>
<tr>
<td valign="top" align="left">Proteinuria, g/day</td>
<td valign="top" align="center">1.17 (0.61&#x2013;2.28)</td>
<td valign="top" align="center">1.99 (1.15&#x2013;3.56)</td>
<td valign="top" align="center">1.60 &#xd7; 10<sup>&#x2212;5</sup>
</td>
<td valign="top" align="center">1.10 (1.05&#x2013;1.15)</td>
</tr>
<tr>
<td valign="top" align="left">eGFR, ml/min per 1.73 m<sup>2</sup>
</td>
<td valign="top" align="center">89.13 (66.05&#x2013;110.14)</td>
<td valign="top" align="center">53.69 (37.47&#x2013;85.39)</td>
<td valign="top" align="center">1.24 &#xd7; 10<sup>&#x2212;27</sup>
</td>
<td valign="top" align="center">0.96 (0.96&#x2013;0.97)</td>
</tr>
<tr>
<td valign="top" align="left">Serum IgA level, g/l</td>
<td valign="top" align="center">3.30 &#xb1; 1.22</td>
<td valign="top" align="center">3.17 &#xb1; 0.99</td>
<td valign="top" align="center">0.27</td>
<td valign="top" align="center">0.93 (0.81&#x2013;1.06)</td>
</tr>
<tr>
<td valign="top" align="left">Uric acid, &#x3bc;mol/l</td>
<td valign="top" align="center">358.08 &#xb1; 97.12</td>
<td valign="top" align="center">429.22 &#xb1; 110.27</td>
<td valign="top" align="center">1.47 &#xd7; 10<sup>&#x2212;19</sup>
</td>
<td valign="top" align="center">1.01 (1.00&#x2013;1.01)</td>
</tr>
<tr>
<td valign="top" align="left">Triglycerides, mmol/l</td>
<td valign="top" align="center">1.59 (1.06&#x2013;2.37)</td>
<td valign="top" align="center">1.85 (1.13&#x2013;2.69)</td>
<td valign="top" align="center">0.01</td>
<td valign="top" align="center">1.12 (1.03&#x2013;1.23)</td>
</tr>
<tr>
<td valign="top" align="left">Total cholesterol, mmol/l</td>
<td valign="top" align="center">4.77 (4.03&#x2013;5.66)</td>
<td valign="top" align="center">4.76 (3.99&#x2013;5.84)</td>
<td valign="top" align="center">0.72</td>
<td valign="top" align="center">1.02 (0.93&#x2013;1.11)</td>
</tr>
<tr>
<td valign="top" align="left">Low-density lipoprotein, mmol/l</td>
<td valign="top" align="center">2.75 (2.24&#x2013;3.36)</td>
<td valign="top" align="center">2.75 (2.19&#x2013;3.59)</td>
<td valign="top" align="center">0.77</td>
<td valign="top" align="center">1.02 (0.90&#x2013;1.16)</td>
</tr>
<tr>
<th valign="top" colspan="5" align="left">Renal biopsy</th>
</tr>
<tr>
<td valign="top" align="left">M0/M1</td>
<td valign="top" align="center">636/374 (62.97%/37.03%)</td>
<td valign="top" align="center">71/87 (44.94%/55.06%)</td>
<td valign="top" align="center">2.00 &#xd7; 10<sup>&#x2212;6</sup>
</td>
<td valign="top" align="center">2.14 (1.56&#x2013;2.93)</td>
</tr>
<tr>
<td valign="top" align="left">E0/E1</td>
<td valign="top" align="center">668/342 (66.14%/33.86%)</td>
<td valign="top" align="center">100/58 (63.29%/36.71%)</td>
<td valign="top" align="center">0.36</td>
<td valign="top" align="center">1.16 (0.84&#x2013;1.61)</td>
</tr>
<tr>
<td valign="top" align="left">S0/S1</td>
<td valign="top" align="center">402/608 (39.80%/60.20%)</td>
<td valign="top" align="center">33/125 (20.89%/79.11%)</td>
<td valign="top" align="center">2.10 &#xd7; 10<sup>&#x2212;5</sup>
</td>
<td valign="top" align="center">2.31 (1.57&#x2013;3.39)</td>
</tr>
<tr>
<td valign="top" align="left">T0/T1+T2</td>
<td valign="top" align="center">723/287 (71.58%/28.42%)</td>
<td valign="top" align="center">53/105 (33.54%/66.46%)</td>
<td valign="top" align="center">1.72 &#xd7; 10<sup>&#x2212;32</sup>
</td>
<td valign="top" align="center">3.34 (2.73&#x2013;4.07)</td>
</tr>
<tr>
<td valign="top" align="left">C0/C1+C2</td>
<td valign="top" align="center">421/589 (41.68%/58.32%)</td>
<td valign="top" align="center">54/104 (34.18%/65.82%)</td>
<td valign="top" align="center">1.00 &#xd7; 10<sup>&#x2212;3</sup>
</td>
<td valign="top" align="center">1.47 (1.17&#x2013;1.86)</td>
</tr>
<tr>
<th valign="top" colspan="5" align="left">Therapy</th>
</tr>
<tr>
<td valign="top" align="left">Renin&#x2013;angiotensin system blocks</td>
<td valign="top" align="center">960 (95.05%)</td>
<td valign="top" align="center">151 (95.57%)</td>
<td valign="top" align="center">0.14</td>
<td valign="top" align="center">0.57 (0.26&#x2013;1.21)</td>
</tr>
<tr>
<td valign="top" align="left">Corticosteroids/cytotoxic drugs</td>
<td valign="top" align="center">474 (46.93%)</td>
<td valign="top" align="center">107 (67.72%)</td>
<td valign="top" align="center">3.60 &#xd7; 10<sup>&#x2212;5</sup>
</td>
<td valign="top" align="center">2.02 (1.45&#x2013;2.82)</td>
</tr>
<tr>
<td valign="top" align="left">Follow-up, months</td>
<td valign="top" align="center">67.50 (37.75&#x2013;105.25)</td>
<td valign="top" align="center">67.50 (38.00&#x2013;97.25)</td>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Data are expressed as mean &#xb1; SD, median (interquartile range), absolute, and percent frequency.</p>
</fn>
<fn>
<p>ESKD, end-stage kidney disease; CI, confidence interval; HR, hazard ratio; eGFR, estimated glomerular filtration rate.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Performance of the pathology <italic>T</italic>-score prediction model</title>
<p>Feature reductions were conducted using the AUCRF algorithm, which was used to select the optimal random forest model with the least number of predictive variables to predict the presence or absence of T lesions. Clinical variables with a probability of selection higher than 0.7 were selected in repeated cross-validation of the optimal random forest model (optimal AUC = 0.82). Finally, the features selected by AUCRF for the T prediction model included age, systolic arterial pressure, diastolic arterial pressure, proteinuria, eGFR, serum IgA, and uric acid (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>). The 690 IgAN patients with Oxford MEST-C scores in dataset 1 as the training set were taken to develop a pathology <italic>T</italic>-score prediction model. The 1,168 IgAN patients with Oxford MEST-C scores in dataset 2 as the testing set were used only for reporting the performance of the model and were not used for development or fine-tuning.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Variables selected by AUCRF for the pathology <italic>T</italic>-score prediction model. The importance scores of the clinical variables with a probability of selection higher than 0.7 in repeated cross-validation of the optimal random forest model to predict the presence or absence of T lesions.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-14-1224631-g002.tif"/>
</fig>
<p>If a predictive model has an AUC of higher than 0.75, it will be considered to have a good discriminating ability. The pathology T prediction model achieved a discrimination of 0.82 (95% CI: 0.80&#x2013;0.85) [area under the receiver operating characteristic (ROC) curve (AUC)] in the testing set (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3A</bold>
</xref>). The ROC curve had 0.74 sensitivity and 0.77 specificity, which indicated that it had better clinical utility.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Receiver operating characteristic curves of the prediction models. The receiver operating characteristic curves for <bold>(A)</bold> the pathology <italic>T</italic>-score prediction (<italic>T</italic>
<sub>pre</sub>) model and <bold>(B)</bold> the 5-year ESKD prediction model. The base model was the 5-year ESKD prediction model based on the XGBoost algorithm with only clinical variables as input variables. The base model + <italic>T</italic>
<sub>bio</sub> was the 5-year ESKD prediction model based on XGBoost using clinical variables and the real pathological T lesions score (<italic>T</italic>
<sub>bio</sub>, T0 was assigned 0, and T1 and T2 were assigned 1). The base model + <italic>T</italic>
<sub>pre</sub> was when the base model plus <italic>T</italic>
<sub>bio</sub> was trained using clinical variables and <italic>T</italic>
<sub>bio</sub>, and the <italic>T</italic>
<sub>bio</sub> of the testing set was replaced by the corresponding <italic>T</italic>
<sub>pre</sub> predicted by the pathology <italic>T</italic>-score prediction model. The clinical variables used for the 5-year ESKD prediction model included age, sex, systolic arterial pressure, diastolic arterial pressure, proteinuria, eGFR, serum IgA, uric acid, triglycerides, total cholesterol, low-density lipoprotein, and history of previous use of renin&#x2013;angiotensin system (RAS) inhibitors and immunosuppressants. AUC, area under the curve.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-14-1224631-g003.tif"/>
</fig>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Performance of the 5-year ESKD prediction model</title>
<p>The unadjusted Cox regression analysis suggested that sex, systolic arterial pressure, diastolic arterial pressure, proteinuria, eGFR, uric acid, triglycerides, and tubular atrophy/interstitial fibrosis (T) were risk factors for developing ESKD (<xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>). A study supported elevated serum IgA as a causal factor in IgA nephropathy through Mendelian randomization (<xref ref-type="bibr" rid="B38">38</xref>). Some studies have suggested the association between the poor prognosis of renal disease and dyslipidemia. Higher triglycerides and cholesterol levels have been proven to be independent risk factors for the progression of kidney disease (<xref ref-type="bibr" rid="B39">39</xref>). Hence, clinical variables (age, sex, systolic arterial pressure, diastolic arterial pressure, proteinuria, eGFR, serum IgA, uric acid, triglycerides, total cholesterol, low-density lipoprotein, history of previous use of RAS inhibitors and immunosuppressants) and the pathology T lesions (<italic>T</italic>
<sub>bio</sub>, T0 was assigned 0, T1 and T2 were assigned 1) were used as the input variables of the 5-year ESKD prediction model.</p>
<p>To make the predictive model achieve a good performance, the 1,168 follow-up IgAN patients with Oxford MEST-C scores in dataset 2 were randomly divided into training and testing sets at a ratio of 8:2. The training set included 936 patients and the testing set included 232 patients. The training set was used to perform five-fold cross-validation to select the optimal prediction model. The testing set was used to assess the performance.</p>
<p>The performance value of the 5-year ESKD prediction model using only the above clinical variables as input variables (base model) was 0.86 (95% CI: 0.75&#x2013;0.97) in the test set (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3B</bold>
</xref>). To test whether the <italic>T</italic>
<sub>bio</sub> could improve the predictive performance of the 5-year ESKD prediction model, we added <italic>T</italic>
<sub>bio</sub> to the base model. An increase in AUC [from 0.86 (95% CI: 0.75&#x2013;0.97) to 0.92 (95% CI: 0.85&#x2013;0.98); <italic>P</italic> = 0.03] showed a better discriminating ability, which indicated that the T was important for judging the prognosis of patients with IgAN (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3B</bold>
</xref>). To test whether <italic>T</italic>
<sub>pre</sub> had a similar effect on judging the prognosis of IgAN patients, after training the 5-year ESKD prediction model with the training set, we replaced the <italic>T</italic>
<sub>bio</sub> in the testing set with the corresponding <italic>T</italic>
<sub>pre</sub> to see the discrimination effect. The AUC was 0.90 (95% CI: 0.82&#x2013;0.99) in the testing set (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3B</bold>
</xref>). The performance of the base model plus <italic>T</italic>
<sub>pre</sub> did not differ from that of the base model plus <italic>T</italic>
<sub>bio</sub> [AUC for the base model plus <italic>T</italic>
<sub>pre</sub> 0.90 (95% CI: 0.82&#x2013;0.99) <italic>vs</italic>. AUC for the base model plus <italic>T</italic>
<sub>bio</sub> 0.92 (95% CI: 0.85&#x2013;0.98), <italic>P</italic> = 0.52, <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>], which showed that the value of the <italic>T</italic>
<sub>pre</sub> in predicting the ESKD of patients was comparable to that of <italic>T</italic>
<sub>bio</sub>. The calibration of the three prediction models is shown in <xref ref-type="fig" rid="f4">
<bold>Figures&#xa0;4A&#x2013;C</bold>
</xref>. The <italic>P</italic>-values for the Hosmer&#x2013;Lemeshow test of the base model, the base model plus <italic>T</italic>
<sub>bio</sub>, and the base model plus <italic>T</italic>
<sub>pre</sub> were 0.42, 0.79, and 0.92, respectively, which indicated that these models had a good calibration. These results suggested the importance of T in predicting ESKD, and <italic>T</italic>
<sub>pre</sub> can be used to assist clinicians in assessing the prognosis of patients without pathology reports.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Performance comparison for the prediction on 5-year ESKD status with different predictors in the testing subset.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Model</th>
<th valign="top" align="center">Accuracy</th>
<th valign="top" align="center">Sensitivity</th>
<th valign="top" align="center">Specificity</th>
<th valign="top" align="center">AUC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Clinical variables</td>
<td valign="middle" align="center">0.85</td>
<td valign="middle" align="center">0.79</td>
<td valign="middle" align="center">0.86</td>
<td valign="middle" align="center">0.86</td>
</tr>
<tr>
<td valign="top" align="left">Clinical variables plus <italic>T</italic>
<sub>bio</sub>
</td>
<td valign="middle" align="center">0.83</td>
<td valign="middle" align="center">0.93</td>
<td valign="middle" align="center">0.78</td>
<td valign="middle" align="center">0.92</td>
</tr>
<tr>
<td valign="top" align="left">Clinical variables plus <italic>T</italic>
<sub>pre</sub>
</td>
<td valign="middle" align="center">0.86</td>
<td valign="middle" align="center">0.86</td>
<td valign="middle" align="center">0.86</td>
<td valign="middle" align="center">0.90</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The clinical variables include age, sex, systolic arterial pressure, diastolic arterial pressure, proteinuria, eGFR, serum IgA, uric acid, triglycerides, total cholesterol, low-density lipoprotein, and history of previous use of renin&#x2013;angiotensin system (RAS) inhibitors and immunosuppressants.</p>
</fn>
<fn>
<p>T<sub>bio</sub>, the real pathological T-score quantified as either 0 (absent) or 1 (T1 or T2); T<sub>pre</sub>, the pathological T-score predicted by the baseline pathology T-score prediction (T<sub>pre</sub>) model.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Calibration plots of the 5-year ESKD prediction models. The calibration plots for <bold>(A)</bold> the base model, <bold>(B)</bold> the base model plus <italic>T</italic>
<sub>bio</sub>, and <bold>(C)</bold> the base model plus <italic>T</italic>
<sub>pre</sub>. The <italic>P</italic>-values for the Hosmer&#x2013;Lemeshow test of the base model, the base model plus <italic>T</italic>
<sub>bio</sub>, and the base model plus <italic>T</italic>
<sub>pre</sub> were 0.42, 0.79, and 0.92, respectively, which indicated that these models had a good calibration.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-14-1224631-g004.tif"/>
</fig>
<p>
<xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref> shows the performance of the 5-year ESKD prediction model based on different machine learning algorithms in the testing set using <italic>T</italic>
<sub>pre</sub>. All models have good prediction performance, which indicated that <italic>T</italic>
<sub>pre</sub> could be used in ESKD predictive models built on different algorithms.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Performance of the 5-year ESKD prediction model using <italic>T</italic>
<sub>pre</sub> based on different machine learning algorithms in the testing set.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Model</th>
<th valign="top" align="center">Accuracy</th>
<th valign="top" align="center">Sensitivity</th>
<th valign="top" align="center">Specificity</th>
<th valign="top" align="center">AUC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">XGBoost</td>
<td valign="top" align="center">0.86</td>
<td valign="top" align="center">0.86</td>
<td valign="top" align="center">0.86</td>
<td valign="top" align="center">0.90</td>
</tr>
<tr>
<td valign="top" align="left">Random forest</td>
<td valign="top" align="center">0.82</td>
<td valign="top" align="center">0.79</td>
<td valign="top" align="center">0.87</td>
<td valign="top" align="center">0.89</td>
</tr>
<tr>
<td valign="top" align="left">Penalized regression</td>
<td valign="top" align="center">0.80</td>
<td valign="top" align="center">0.93</td>
<td valign="top" align="center">0.80</td>
<td valign="top" align="center">0.88</td>
</tr>
<tr>
<td valign="top" align="left">Artificial neural network</td>
<td valign="top" align="center">0.78</td>
<td valign="top" align="center">0.86</td>
<td valign="top" align="center">0.77</td>
<td valign="top" align="center">0.86</td>
</tr>
<tr>
<td valign="top" align="left">Support vector machine</td>
<td valign="top" align="center">0.71</td>
<td valign="top" align="center">0.86</td>
<td valign="top" align="center">0.62</td>
<td valign="top" align="center">0.77</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The model was trained using clinical variables and the T<sub>bio</sub>, and the T<sub>bio</sub> was replaced with the corresponding T<sub>pre</sub> predicted by the pathology T-score prediction model in the test subset.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>For the lifetime ESKD prediction model based on XGBoost using only clinical variables (base model), the <italic>C</italic>-statistic was 0.82 (95% CI: 0.80&#x2013;0.84) in the testing set. The discriminating ability of the base model plus <italic>T</italic>
<sub>pre</sub> was also comparable to the base model plus <italic>T</italic>
<sub>bio</sub> [<italic>C</italic>-statistic: 0.85 (95% CI: 0.83&#x2013;0.86) <italic>vs</italic>. 0.85 (95% CI: 0.83&#x2013;0.86), <italic>P</italic> = 0.11] in the testing set.</p>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>External validation of the ESKD prediction model using <italic>T</italic>
<sub>pre</sub>
</title>
<p>The 355 patients without MEST-C scores in dataset 2 were included as the external validation population for evaluating the performance of the 5-year ESKD prediction model. Because patients did not have MEST-C scores, the <italic>T</italic>
<sub>pre</sub> predicted by the pathology <italic>T</italic>-score prediction model was used in the 5-year ESKD prediction model. The AUC of the 5-year ESKD prediction model using <italic>T</italic>
<sub>pre</sub> based on XGBoost was 0.93 (95% CI: 0.87&#x2013;0.99). We listed the AUC of the applied other machine learning algorithms in <xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref>.</p>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>Performance of the 5-year ESKD prediction model using <italic>T</italic>
<sub>pre</sub> based on different machine learning algorithms in the external validation set.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Model</th>
<th valign="middle" align="center">Accuracy</th>
<th valign="middle" align="center">Sensitivity</th>
<th valign="middle" align="center">Specificity</th>
<th valign="middle" align="center">AUC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">XGBoost</td>
<td valign="middle" align="center">0.82</td>
<td valign="middle" align="center">1.00</td>
<td valign="middle" align="center">0.81</td>
<td valign="middle" align="center">0.93</td>
</tr>
<tr>
<td valign="top" align="left">Logistic regression</td>
<td valign="middle" align="center">0.72</td>
<td valign="middle" align="center">1.00</td>
<td valign="middle" align="center">0.71</td>
<td valign="middle" align="center">0.90</td>
</tr>
<tr>
<td valign="top" align="left">Artificial neural network</td>
<td valign="middle" align="center">0.50</td>
<td valign="middle" align="center">1.00</td>
<td valign="middle" align="center">0.48</td>
<td valign="middle" align="center">0.79</td>
</tr>
<tr>
<td valign="top" align="left">Support vector machine</td>
<td valign="middle" align="center">0.87</td>
<td valign="middle" align="center">0.67</td>
<td valign="middle" align="center">0.87</td>
<td valign="middle" align="center">0.74</td>
</tr>
<tr>
<td valign="top" align="left">Random forest</td>
<td valign="middle" align="center">0.86</td>
<td valign="middle" align="center">0.83</td>
<td valign="middle" align="center">0.86</td>
<td valign="middle" align="center">0.92</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The characteristics used in the basic model include age, gender, SBP, DBP, eGFR, IgA, UTP, UA, TG, TCHO, LDL, history of corticosteroids/cytotoxic drugs, and renin&#x2013;angiotensin system blockers.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>In the lifetime ESKD prediction model using <italic>T</italic>
<sub>pre</sub>, the <italic>C</italic>-statistic was 0.92 (95% CI: 0.90&#x2013;0.94). We have shown here that both models have a good performance in the external validation set, indicating the reliability of <italic>T</italic>
<sub>pre</sub> for assisting in evaluating the prognosis of IgAN.</p>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<label>4</label>
<title>Discussion</title>
<p>We developed a pathology <italic>T</italic>-score prediction (<italic>T</italic>
<sub>pre</sub>) model that can predict whether the patient with IgAN may have tubulointerstitial lesions at this time based on clinical variables when the patient did not undergo a renal biopsy or did not want to repeat the renal biopsy for progression assessment. We further constructed the 5-year/lifetime ESKD prediction model based on the XGBoost algorithm to confirm the importance of T in predicting ESKD, and <italic>T</italic>
<sub>pre</sub> can replace the real pathological T lesions for assisting clinicians in evaluating the prognosis of IgAN patients without pathology reports. In addition, the ESKD prediction model built based on different machine learning algorithms had good discriminating ability by using clinical variables and <italic>T</italic>
<sub>pre</sub>, which indicated the reliability and universality of <italic>T</italic>
<sub>pre</sub> for assisting in evaluating the prognosis of IgAN.</p>
<p>For developing the pathology <italic>T</italic>-score (<italic>T</italic>
<sub>pre</sub>) prediction model, we first used the AUCRF algorithm to select the clinical variables that may be associated with the tubulointerstitial lesions. Feature selection before training the predictive model can prevent dimensional disaster, reduce training time, prevent overfitting, enhance model generalization ability, and enhance the understanding of features and feature values, which also determines the upper limit of the effect of a machine learning task. The AUCRF is based on the RF algorithm, which is used for feature reduction based on optimizing the area under the ROC curve (AUC) of the random forest (<xref ref-type="bibr" rid="B33">33</xref>). It was found that age, systolic arterial pressure, diastolic arterial pressure, proteinuria, eGFR, serum IgA, and uric acid may be the clinical characteristics associated with tubular atrophy/interstitial fibrosis. Mechanism studies are needed to explore the inherent causality of these correlations and predictive capability. There have been reports indicating the association between reduced initial eGFR, higher initial MAP, proteinuria, and tubular atrophy/interstitial fibrosis (<xref ref-type="bibr" rid="B31">31</xref>). Next, we used the stacking algorithm to construct the pathology <italic>T</italic>-score prediction (<italic>T</italic>
<sub>pre</sub>) model based on the clinical characteristics selected by the AUCRF. A single learner has over- or underfitting problems, and to obtain a learner with excellent generalization performance, we can train multiple individual learners to form a strong learner through a certain combination strategy. This method of integrating multiple individual learners is called ensemble learning. Stacking is one of the methods of ensemble learning. The advantage of integration is that different models can learn different features of the data, and the results after fusion tend to perform better (<xref ref-type="bibr" rid="B40">40</xref>). As our results showed, when we used an independent dataset as the testing set, the AUC of the pathological <italic>T</italic>-score prediction (<italic>T</italic>
<sub>pre</sub>) model reached 0.82, which indicates the good discriminating ability of this <italic>T</italic>
<sub>pre</sub> prediction model.</p>
<p>A host of studies have indicated that pathological T lesions play an important role in predicting prognosis (<xref ref-type="bibr" rid="B14">14</xref>, <xref ref-type="bibr" rid="B35">35</xref>, <xref ref-type="bibr" rid="B41">41</xref>). At the same time, most current ESKD prediction models based on different methods or algorithms all include pathology <italic>T</italic>-score (<xref ref-type="bibr" rid="B14">14</xref>, <xref ref-type="bibr" rid="B16">16</xref>, <xref ref-type="bibr" rid="B19">19</xref>). Nevertheless, a renal puncture is invasive, which may cause a series of complications and has a host of contraindications, such as severe hypertension, coagulation disorders, solitary kidney, and so on (<xref ref-type="bibr" rid="B42">42</xref>). Furthermore, the number of patients at high risk of renal puncture may increase in the near future because of the aging of the population and the increased use of anticoagulant medication (<xref ref-type="bibr" rid="B43">43</xref>). For the patients who lack the report of kidney biopsy or do not want to undergo repeat renal puncture for disease progression assessment and evaluation of the effect of drug therapy, the clinician could not assess the prognosis of these patients with IgAN by using the established ESKD prediction model. The pathology <italic>T</italic>-score prediction (<italic>T</italic>
<sub>pre</sub>) model we developed may solve this problem. We also constructed a 5-year/lifetime ESKD prediction model based on XGBoost to assess whether the value of <italic>T</italic>
<sub>pre</sub> in predicting ESKD of patients with IgAN was consistent with real pathological <italic>T</italic>-score. The performance of the base model plus <italic>T</italic>
<sub>pre</sub> was similar to the base model plus <italic>T</italic>
<sub>bio</sub>, which showed that the <italic>T</italic>
<sub>pre</sub> can replace the real pathological <italic>T</italic>-score for prognostic prediction.</p>
<p>As far as we know, this study is the first to construct a pathology <italic>T</italic>-score prediction model in IgA nephropathy. At the same time, it is also the first study to use a machine learning algorithm to identify clinical variables that may influence the development of tubular atrophy/interstitial fibrosis, which may be useful for assessing the prognosis and targeted medication guidance. However, there is a limitation in our study. The model has been developed and tested in a single-center cohort of patients with IgAN; therefore, multicenter prospective cohort and ethnic-based cohort studies are necessary, which will further confirm the reliability of the pathology <italic>T</italic>-score prediction model, expand the scope of application of the model, and provide possibilities for clinical application.</p>
<p>In conclusion, our pathology <italic>T</italic>-score prediction (<italic>T</italic>
<sub>pre</sub>) model is a reliable tool for predicting the presence or absence of pathological T lesions. At the same time, it can also be used to assist clinicians in predicting the prognosis of patients with IgAN. A prospective multicenter cohort study is necessary to explore the potential value and robustness of this T prediction tool in the management of IgA nephropathy.</p>
</sec>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The data  presented in the study are deposited in the GitHub repository (<uri xlink:href="https://github.com/zhangd17-web/IGAN_MI">https://github.com/zhangd17-web/IGAN_MI</uri>).</p>
</sec>
<sec id="s6" sec-type="author-contributions">
<title>Author contributions</title>
<p>Research idea and study design: HZ, X-JZ, LW, DZ, and LX. Data acquisition: LX, SS, X-JZ, and HZ. Data analysis/interpretation: LX, DZ, LW, HW, and X-JZ. Statistical analysis: LX and DZ. Supervision or mentorship: X-JZ, HZ, HW, LW, RC, GC, LL, SS, XZ, SH, LD, and JL. Each author contributed important intellectual content during manuscript drafting or revision and agrees to be personally accountable for the individual&#x2019;s own contributions and to ensure that questions pertaining to the accuracy or integrity of any portion of the work, even one in which the author was not directly involved, are appropriately investigated and resolved, with documentation in the literature if appropriate.</p>
</sec>
</body>
<back>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>This work was supported by the National Science Foundation of China (82022010, 82131430172, 81970613, 82070733), Beijing Natural Science Foundation (Z190023), Academy of Medical Sciences&#x2014;Newton Advanced Fellowship (NAFR13\1033), King&#x2019;s College London&#x2014;Peking University Health Science Center Joint Institute for Medical Research (BMU2021KCL004), Fok Ying Tung Education Foundation (171030), Chinese Academy of Medical Sciences (CAMS) Innovation Fund for Medical Sciences (2019-I2M-5-046, 2020-JKCS-009), and National High Level Hospital Clinical Research Funding (Interdisciplinary Clinical Research Project of Peking University First Hospital, 2022CR41, 2022CR40). The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript.</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>We thank all the patients and researchers who participated in the study.</p>
</ack>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>Authors DZ, HW, LW, RC and GC were employed by company WeGene. </p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
<p>The reviewer JZ declared a shared parent affiliation with the authors LX, SS, LL, X-HZ, JL, X-JZ, and HZ to the handling editor at the time of review.</p>
</sec>
<sec id="s9" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lai</surname> <given-names>KN</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>SC</given-names>
</name>
<name>
<surname>Schena</surname> <given-names>FP</given-names>
</name>
<name>
<surname>Novak</surname> <given-names>J</given-names>
</name>
<name>
<surname>Tomino</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Fogo</surname> <given-names>AB</given-names>
</name>
<etal/>
</person-group>. <article-title>IgA nephropathy</article-title>. <source>Nat Rev Dis Primers</source> (<year>2016</year>) <volume>2</volume>:<fpage>16001</fpage>. doi: <pub-id pub-id-type="doi">10.1038/nrdp.2016.1</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Magistroni</surname> <given-names>R</given-names>
</name>
<name>
<surname>D'Agati</surname> <given-names>VD</given-names>
</name>
<name>
<surname>Appel</surname> <given-names>GB</given-names>
</name>
<name>
<surname>Kiryluk</surname> <given-names>K</given-names>
</name>
</person-group>. <article-title>New developments in the genetics, pathogenesis, and therapy of IgA nephropathy</article-title>. <source>Kidney Int</source> (<year>2015</year>) <volume>88</volume>(<issue>5</issue>):<page-range>974&#x2013;89</page-range>. doi: <pub-id pub-id-type="doi">10.1038/ki.2015.252</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Le</surname> <given-names>W</given-names>
</name>
<name>
<surname>Liang</surname> <given-names>S</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Deng</surname> <given-names>K</given-names>
</name>
<name>
<surname>Bao</surname> <given-names>H</given-names>
</name>
<name>
<surname>Zeng</surname> <given-names>C</given-names>
</name>
<etal/>
</person-group>. <article-title>Long-term renal survival and related risk factors in patients with IgA nephropathy: results from a cohort of 1155 cases in a Chinese adult population</article-title>. <source>Nephrol Dial Transplant</source> (<year>2012</year>) <volume>27</volume>(<issue>4</issue>):<page-range>1479&#x2013;85</page-range>. doi: <pub-id pub-id-type="doi">10.1093/ndt/gfr527</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goto</surname> <given-names>M</given-names>
</name>
<name>
<surname>Wakai</surname> <given-names>K</given-names>
</name>
<name>
<surname>Kawamura</surname> <given-names>T</given-names>
</name>
<name>
<surname>Ando</surname> <given-names>M</given-names>
</name>
<name>
<surname>Endoh</surname> <given-names>M</given-names>
</name>
<name>
<surname>Tomino</surname> <given-names>Y</given-names>
</name>
</person-group>. <article-title>A scoring system to predict renal outcome in IgA nephropathy: a nationwide 10-year prospective cohort study</article-title>. <source>Nephrol Dial Transplant</source> (<year>2009</year>) <volume>24</volume>(<issue>10</issue>):<page-range>3068&#x2013;74</page-range>. doi: <pub-id pub-id-type="doi">10.1093/ndt/gfp273</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Beukhof</surname> <given-names>JR</given-names>
</name>
<name>
<surname>Kardaun</surname> <given-names>O</given-names>
</name>
<name>
<surname>Schaafsma</surname> <given-names>W</given-names>
</name>
<name>
<surname>Poortema</surname> <given-names>K</given-names>
</name>
<name>
<surname>Donker</surname> <given-names>AJ</given-names>
</name>
<name>
<surname>Hoedemaeker</surname> <given-names>PJ</given-names>
</name>
<etal/>
</person-group>. <article-title>Toward individual prognosis of IgA nephropathy</article-title>. <source>Kidney Int</source> (<year>1986</year>) <volume>29</volume>(<issue>2</issue>):<page-range>549&#x2013;56</page-range>. doi: <pub-id pub-id-type="doi">10.1038/ki.1986.33</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rekola</surname> <given-names>S</given-names>
</name>
<name>
<surname>Bergstrand</surname> <given-names>A</given-names>
</name>
<name>
<surname>Bucht</surname> <given-names>H</given-names>
</name>
</person-group>. <article-title>Development of hypertension in IgA nephropathy as a marker of a poor prognosis</article-title>. <source>Am J Nephrol</source> (<year>1990</year>) <volume>10</volume>(<issue>4</issue>):<page-range>290&#x2013;5</page-range>. doi: <pub-id pub-id-type="doi">10.1159/000168122</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Radford</surname> <given-names>MG</given-names>
<suffix>Jr.</suffix>
</name>
<name>
<surname>Donadio</surname> <given-names>JV</given-names>
<suffix>Jr.</suffix>
</name>
<name>
<surname>Bergstralh</surname> <given-names>EJ</given-names>
</name>
<name>
<surname>Grande</surname> <given-names>JP</given-names>
</name>
</person-group>. <article-title>Predicting renal outcome in IgA nephropathy</article-title>. <source>J Am Soc Nephrol JASN</source> (<year>1997</year>) <volume>8</volume>(<issue>2</issue>):<fpage>199</fpage>&#x2013;<lpage>207</lpage>. doi: <pub-id pub-id-type="doi">10.1681/ASN.V82199</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>D</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>J</given-names>
</name>
<name>
<surname>Duan</surname> <given-names>S</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>P</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>L</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>L</given-names>
</name>
<etal/>
</person-group>. <article-title>Clinicopathological features to predict progression of igA nephropathy with mild proteinuria</article-title>. <source>Kidney Blood Pressure Res</source> (<year>2018</year>) <volume>43</volume>(<issue>2</issue>):<page-range>318&#x2013;28</page-range>. doi: <pub-id pub-id-type="doi">10.1159/000487901</pub-id>
</citation>
</ref>
<ref id="B9">
<label>9</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barbour</surname> <given-names>SJ</given-names>
</name>
<name>
<surname>Espino-Hernandez</surname> <given-names>G</given-names>
</name>
<name>
<surname>Reich</surname> <given-names>HN</given-names>
</name>
<name>
<surname>Coppo</surname> <given-names>R</given-names>
</name>
<name>
<surname>Roberts</surname> <given-names>IS</given-names>
</name>
<name>
<surname>Feehally</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>The MEST score provides earlier risk prediction in lgA nephropathy</article-title>. <source>Kidney Int</source> (<year>2016</year>) <volume>89</volume>(<issue>1</issue>):<page-range>167&#x2013;75</page-range>. doi: <pub-id pub-id-type="doi">10.1038/ki.2015.322</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wakai</surname> <given-names>K</given-names>
</name>
<name>
<surname>Kawamura</surname> <given-names>T</given-names>
</name>
<name>
<surname>Endoh</surname> <given-names>M</given-names>
</name>
<name>
<surname>Kojima</surname> <given-names>M</given-names>
</name>
<name>
<surname>Tomino</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Tamakoshi</surname> <given-names>A</given-names>
</name>
<etal/>
</person-group>. <article-title>A scoring system to predict renal outcome in IgA nephropathy: from a nationwide prospective study</article-title>. <source>Nephrol Dial Transplant</source> (<year>2006</year>) <volume>21</volume>(<issue>10</issue>):<page-range>2800&#x2013;8</page-range>. doi: <pub-id pub-id-type="doi">10.1093/ndt/gfl342</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Okonogi</surname> <given-names>H</given-names>
</name>
<name>
<surname>Utsunomiya</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Miyazaki</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Koike</surname> <given-names>K</given-names>
</name>
<name>
<surname>HIrano</surname> <given-names>K</given-names>
</name>
<name>
<surname>Tsuboi</surname> <given-names>N</given-names>
</name>
<etal/>
</person-group>. <article-title>A predictive clinical grading system for immunoglobulin A nephropathy by combining proteinuria and estimated glomerular filtration rate</article-title>. <source>Nephron Clin Pract</source> (<year>2011</year>) <volume>118</volume>(<issue>3</issue>):<page-range>c292&#x2013;300</page-range>. doi: <pub-id pub-id-type="doi">10.1159/000322613</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xie</surname> <given-names>J</given-names>
</name>
<name>
<surname>Kiryluk</surname> <given-names>K</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>W</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>S</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>P</given-names>
</name>
<etal/>
</person-group>. <article-title>Predicting progression of IgA nephropathy: new clinical progression risk score</article-title>. <source>PloS One</source> (<year>2012</year>) <volume>7</volume>(<issue>6</issue>):<elocation-id>e38904</elocation-id>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0038904</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tanaka</surname> <given-names>S</given-names>
</name>
<name>
<surname>Ninomiya</surname> <given-names>T</given-names>
</name>
<name>
<surname>Katafuchi</surname> <given-names>R</given-names>
</name>
<name>
<surname>Masutani</surname> <given-names>K</given-names>
</name>
<name>
<surname>Tsuchimoto</surname> <given-names>A</given-names>
</name>
<name>
<surname>Noguchi</surname> <given-names>H</given-names>
</name>
<etal/>
</person-group>. <article-title>Development and validation of a prediction rule using the Oxford classification in IgA nephropathy</article-title>. <source>Clin J Am Soc Nephrol CJASN</source> (<year>2013</year>) <volume>8</volume>(<issue>12</issue>):<page-range>2082&#x2013;90</page-range>. doi: <pub-id pub-id-type="doi">10.2215/CJN.03480413</pub-id>
</citation>
</ref>
<ref id="B14">
<label>14</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barbour</surname> <given-names>SJ</given-names>
</name>
<name>
<surname>Coppo</surname> <given-names>R</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>ZH</given-names>
</name>
<name>
<surname>Suzuki</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Matsuzaki</surname> <given-names>K</given-names>
</name>
<etal/>
</person-group>. <article-title>Evaluating a new international risk-prediction tool in igA nephropathy</article-title>. <source>JAMA Internal Med</source> (<year>2019</year>) <volume>179</volume>(<issue>7</issue>):<page-range>942&#x2013;52</page-range>. doi: <pub-id pub-id-type="doi">10.1001/jamainternmed.2019.0600</pub-id>
</citation>
</ref>
<ref id="B15">
<label>15</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>D</given-names>
</name>
<name>
<surname>Tan</surname> <given-names>X</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>F</given-names>
</name>
<etal/>
</person-group>. <article-title>Prediction of ESRD in igA nephropathy patients from an asian cohort: A random forest model</article-title>. <source>Kidney Blood Pressure Res</source> (<year>2018</year>) <volume>43</volume>(<issue>6</issue>):<page-range>1852&#x2013;64</page-range>. doi: <pub-id pub-id-type="doi">10.1159/000495818</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>T</given-names>
</name>
<name>
<surname>Li</surname> <given-names>X</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Xia</surname> <given-names>E</given-names>
</name>
<name>
<surname>Qin</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Liang</surname> <given-names>S</given-names>
</name>
<etal/>
</person-group>. <article-title>Prediction and risk stratification of kidney outcomes in igA nephropathy</article-title>. <source>Am J Kidney Dis</source> (<year>2019</year>) <volume>74</volume>(<issue>3</issue>):<page-range>300&#x2013;9</page-range>. doi: <pub-id pub-id-type="doi">10.1053/j.ajkd.2019.02.016</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Han</surname> <given-names>X</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>X</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>X</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>Y</given-names>
</name>
<etal/>
</person-group>. <article-title>Random forest can accurately predict the development of end-stage renal disease in immunoglobulin a nephropathy patients</article-title>. <source>Ann Trans Med</source> (<year>2019</year>) <volume>7</volume>(<issue>11</issue>):<fpage>234</fpage>. doi: <pub-id pub-id-type="doi">10.21037/atm.2018.12.11</pub-id>
</citation>
</ref>
<ref id="B18">
<label>18</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Konieczny</surname> <given-names>A</given-names>
</name>
<name>
<surname>Stojanowski</surname> <given-names>J</given-names>
</name>
<name>
<surname>Krajewska</surname> <given-names>M</given-names>
</name>
<name>
<surname>Kusztal</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Machine learning in prediction of igA nephropathy outcome: A comparative approach</article-title>. <source>J Pers Med</source> (<year>2021</year>) <volume>11</volume>(<issue>4</issue>):<fpage>312</fpage>. doi: <pub-id pub-id-type="doi">10.3390/jpm11040312</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schena</surname> <given-names>FP</given-names>
</name>
<name>
<surname>Anelli</surname> <given-names>VW</given-names>
</name>
<name>
<surname>Trotta</surname> <given-names>J</given-names>
</name>
<name>
<surname>Di Noia</surname> <given-names>T</given-names>
</name>
<name>
<surname>Manno</surname> <given-names>C</given-names>
</name>
<name>
<surname>Tripepi</surname> <given-names>G</given-names>
</name>
<etal/>
</person-group>. <article-title>Development and testing of an artificial intelligence tool for predicting end-stage kidney disease in patients with immunoglobulin A nephropathy</article-title>. <source>Kidney Int</source> (<year>2021</year>) <volume>99</volume>(<issue>5</issue>):<page-range>1179&#x2013;88</page-range>. doi: <pub-id pub-id-type="doi">10.1016/j.kint.2020.07.046</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Diciolla</surname> <given-names>M</given-names>
</name>
<name>
<surname>Binetti</surname> <given-names>G</given-names>
</name>
<name>
<surname>Di Noia</surname> <given-names>T</given-names>
</name>
<name>
<surname>Pesce</surname> <given-names>F</given-names>
</name>
<name>
<surname>Schena</surname> <given-names>FP</given-names>
</name>
<name>
<surname>V&#xe5;gane</surname> <given-names>AM</given-names>
</name>
<etal/>
</person-group>. <article-title>Patient classification and outcome prediction in IgA nephropathy</article-title>. <source>Comput Biol Med</source> (<year>2015</year>) <volume>66</volume>:<page-range>278&#x2013;86</page-range>. doi: <pub-id pub-id-type="doi">10.1016/j.compbiomed.2015.09.003</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>T</given-names>
</name>
<name>
<surname>Guestrin</surname> <given-names>C</given-names>
</name>
</person-group>. <article-title>XGBoost: A Scalable Tree Boosting System</article-title>. In: <source>Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</source>. <publisher-loc>San Francisco, California, USA</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name> (<year>2016</year>). p. <page-range>785&#x2013;94</page-range>.</citation>
</ref>
<ref id="B22">
<label>22</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khera</surname> <given-names>R</given-names>
</name>
<name>
<surname>Haimovich</surname> <given-names>J</given-names>
</name>
<name>
<surname>Hurley</surname> <given-names>NC</given-names>
</name>
<name>
<surname>McNamara</surname> <given-names>R</given-names>
</name>
<name>
<surname>Spertus</surname> <given-names>JA</given-names>
</name>
<name>
<surname>Desai</surname> <given-names>N</given-names>
</name>
<etal/>
</person-group>. <article-title>Use of machine learning models to predict death after acute myocardial infarction</article-title>. <source>JAMA Cardiol</source> (<year>2021</year>) <volume>6</volume>(<issue>6</issue>):<page-range>633&#x2013;41</page-range>. doi: <pub-id pub-id-type="doi">10.1001/jamacardio.2021.0122</pub-id>
</citation>
</ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pfaff</surname> <given-names>ER</given-names>
</name>
<name>
<surname>Girvin</surname> <given-names>AT</given-names>
</name>
<name>
<surname>Bennett</surname> <given-names>TD</given-names>
</name>
<name>
<surname>Bhatia</surname> <given-names>A</given-names>
</name>
<name>
<surname>Brooks</surname> <given-names>IM</given-names>
</name>
<name>
<surname>Deer</surname> <given-names>RR</given-names>
</name>
<etal/>
</person-group>. <article-title>Identifying who has long COVID in the USA: a machine learning approach using N3C data</article-title>. <source>Lancet Digital Health</source> (<year>2022</year>) <volume>4</volume>(<issue>7</issue>):<page-range>e532&#x2013;e41</page-range>. doi: <pub-id pub-id-type="doi">10.1016/S2589-7500(22)00048-6</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cattran</surname> <given-names>DC</given-names>
</name>
<name>
<surname>Coppo</surname> <given-names>R</given-names>
</name>
<name>
<surname>Cook</surname> <given-names>HT</given-names>
</name>
<name>
<surname>Feehally</surname> <given-names>J</given-names>
</name>
<name>
<surname>Roberts</surname> <given-names>IS</given-names>
</name>
<name>
<surname>Troyanov</surname> <given-names>S</given-names>
</name>
<etal/>
</person-group>. <article-title>The Oxford classification of IgA nephropathy: rationale, clinicopathological correlations, and classification</article-title>. <source>Kidney Int</source> (<year>2009</year>) <volume>76</volume>(<issue>5</issue>):<page-range>534&#x2013;45</page-range>. doi: <pub-id pub-id-type="doi">10.1038/ki.2009.243</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rui</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Zhai</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>C</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>L</given-names>
</name>
</person-group>. <article-title>The predictive value of Oxford MEST-C classification to immunosuppressive therapy of IgA nephropathy</article-title>. <source>Int Urol Nephrol</source> (<year>2022</year>) <volume>54</volume>(<issue>4</issue>):<page-range>959&#x2013;67</page-range>. doi: <pub-id pub-id-type="doi">10.1007/s11255-021-02974-9</pub-id>
</citation>
</ref>
<ref id="B26">
<label>26</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Trimarchi</surname> <given-names>H</given-names>
</name>
<name>
<surname>Barratt</surname> <given-names>J</given-names>
</name>
<name>
<surname>Cattran</surname> <given-names>DC</given-names>
</name>
<name>
<surname>Cook</surname> <given-names>HT</given-names>
</name>
<name>
<surname>Coppo</surname> <given-names>R</given-names>
</name>
<name>
<surname>Haas</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>Oxford Classification of IgA nephropathy 2016: an update from the IgA Nephropathy Classification Working Group</article-title>. <source>Kidney Int</source> (<year>2017</year>) <volume>91</volume>(<issue>5</issue>):<page-range>1014&#x2013;21</page-range>. doi: <pub-id pub-id-type="doi">10.1016/j.kint.2017.02.003</pub-id>
</citation>
</ref>
<ref id="B27">
<label>27</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schimpf</surname> <given-names>JI</given-names>
</name>
<name>
<surname>Klein</surname> <given-names>T</given-names>
</name>
<name>
<surname>Fitzner</surname> <given-names>C</given-names>
</name>
<name>
<surname>Eitner</surname> <given-names>F</given-names>
</name>
<name>
<surname>Porubsky</surname> <given-names>S</given-names>
</name>
<name>
<surname>Hilgers</surname> <given-names>RD</given-names>
</name>
<etal/>
</person-group>. <article-title>Renal outcomes of STOP-IgAN trial patients in relation to baseline histology (MEST-C scores)</article-title>. <source>BMC Nephrol</source> (<year>2018</year>) <volume>19</volume>(<issue>1</issue>):<fpage>328</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s12882-018-1128-6</pub-id>
</citation>
</ref>
<ref id="B28">
<label>28</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>W</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>C</given-names>
</name>
<name>
<surname>Xing</surname> <given-names>L</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>M</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>R</given-names>
</name>
<etal/>
</person-group>. <article-title>Epidemiology and risk factors for progression in Chinese patients with IgA nephropathy</article-title>. <source>Medicina clinica</source> (<year>2021</year>) <volume>157</volume>(<issue>6</issue>):<page-range>267&#x2013;73</page-range>. doi: <pub-id pub-id-type="doi">10.1016/j.medcli.2020.05.064</pub-id>
</citation>
</ref>
<ref id="B29">
<label>29</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>L</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Er</surname> <given-names>L</given-names>
</name>
<name>
<surname>Barbour</surname> <given-names>SJ</given-names>
</name>
<etal/>
</person-group>. <article-title>External validation of international risk-prediction models of igA nephropathy in an asian-caucasian cohort</article-title>. <source>Kidney Int Rep</source> (<year>2020</year>) <volume>5</volume>(<issue>10</issue>):<page-range>1753&#x2013;63</page-range>. doi: <pub-id pub-id-type="doi">10.1016/j.ekir.2020.07.036</pub-id>
</citation>
</ref>
<ref id="B30">
<label>30</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Levey</surname> <given-names>AS</given-names>
</name>
<name>
<surname>Stevens</surname> <given-names>LA</given-names>
</name>
<name>
<surname>Schmid</surname> <given-names>CH</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>YL</given-names>
</name>
<name>
<surname>Castro</surname> <given-names>AF</given-names>
<suffix>3rd</suffix>
</name>
<name>
<surname>Feldman</surname> <given-names>HI</given-names>
</name>
<etal/>
</person-group>. <article-title>A new equation to estimate glomerular filtration rate</article-title>. <source>Ann Internal Med</source> (<year>2009</year>) <volume>150</volume>(<issue>9</issue>):<page-range>604&#x2013;12</page-range>. doi: <pub-id pub-id-type="doi">10.7326/0003-4819-150-9-200905050-00006</pub-id>
</citation>
</ref>
<ref id="B31">
<label>31</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Roberts</surname> <given-names>IS</given-names>
</name>
<name>
<surname>Cook</surname> <given-names>HT</given-names>
</name>
<name>
<surname>Troyanov</surname> <given-names>S</given-names>
</name>
<name>
<surname>Alpers</surname> <given-names>CE</given-names>
</name>
<name>
<surname>Amore</surname> <given-names>A</given-names>
</name>
<name>
<surname>Barratt</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>The Oxford classification of IgA nephropathy: pathology definitions, correlations, and reproducibility</article-title>. <source>Kidney Int</source> (<year>2009</year>) <volume>76</volume>(<issue>5</issue>):<page-range>546&#x2013;56</page-range>. doi: <pub-id pub-id-type="doi">10.1038/ki.2009.168</pub-id>
</citation>
</ref>
<ref id="B32">
<label>32</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wolpert</surname> <given-names>DH</given-names>
</name>
</person-group>. <article-title>Stacked generalization</article-title>. <source>Neural Networks</source> (<year>1992</year>) <volume>5</volume>(<issue>2</issue>):<page-range>241&#x2013;59</page-range>. doi: <pub-id pub-id-type="doi">10.1016/S0893-6080(05)80023-1</pub-id>
</citation>
</ref>
<ref id="B33">
<label>33</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Calle</surname> <given-names>ML</given-names>
</name>
<name>
<surname>Urrea</surname> <given-names>V</given-names>
</name>
<name>
<surname>Boulesteix</surname> <given-names>AL</given-names>
</name>
<name>
<surname>Malats</surname> <given-names>N</given-names>
</name>
</person-group>. <article-title>AUC-RF: a new strategy for genomic profiling with random forest</article-title>. <source>Hum heredity</source> (<year>2011</year>) <volume>72</volume>(<issue>2</issue>):<page-range>121&#x2013;32</page-range>. doi: <pub-id pub-id-type="doi">10.1159/000330778</pub-id>
</citation>
</ref>
<ref id="B34">
<label>34</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>T</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>T</given-names>
</name>
<name>
<surname>Li</surname> <given-names>X</given-names>
</name>
<name>
<surname>Zeng</surname> <given-names>C</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Z</given-names>
</name>
<etal/>
</person-group>. <article-title>An interpretable machine learning survival model for predicting long-term kidney outcomes in igA nephropathy</article-title>. <source>AMIA Annu Symposium Proc AMIA Symposium</source> (<year>2020</year>) <volume>2020</volume>:<page-range>737&#x2013;46</page-range>.</citation>
</ref>
<ref id="B35">
<label>35</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lv</surname> <given-names>J</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>S</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>D</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Troyanov</surname> <given-names>S</given-names>
</name>
<name>
<surname>Cattran</surname> <given-names>DC</given-names>
</name>
<etal/>
</person-group>. <article-title>Evaluation of the Oxford Classification of IgA nephropathy: a systematic review and meta-analysis</article-title>. <source>Am J Kidney Dis</source> (<year>2013</year>) <volume>62</volume>(<issue>5</issue>):<page-range>891&#x2013;9</page-range>. doi: <pub-id pub-id-type="doi">10.1053/j.ajkd.2013.04.021</pub-id>
</citation>
</ref>
<ref id="B36">
<label>36</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Prentice</surname> <given-names>RL</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Regression models and multivariate life tables</article-title>. <source>J Am Stat Assoc</source> (<year>2021</year>) <volume>116</volume>(<issue>535</issue>):<page-range>1330&#x2013;45</page-range>. doi: <pub-id pub-id-type="doi">10.1080/01621459.2020.1713792</pub-id>
</citation>
</ref>
<ref id="B37">
<label>37</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Park</surname> <given-names>SH</given-names>
</name>
<name>
<surname>Hahm</surname> <given-names>MH</given-names>
</name>
<name>
<surname>Bae</surname> <given-names>BK</given-names>
</name>
<name>
<surname>Chong</surname> <given-names>GO</given-names>
</name>
<name>
<surname>Jeong</surname> <given-names>SY</given-names>
</name>
<name>
<surname>Na</surname> <given-names>S</given-names>
</name>
<etal/>
</person-group>. <article-title>Magnetic resonance imaging features of tumor and lymph node to predict clinical outcome in node-positive cervical cancer: a retrospective analysis</article-title>. <source>Radiat Oncol (London England)</source> (<year>2020</year>) <volume>15</volume>(<issue>1</issue>):<fpage>86</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s13014-020-01502-w</pub-id>
</citation>
</ref>
<ref id="B38">
<label>38</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>L</given-names>
</name>
<name>
<surname>Khan</surname> <given-names>A</given-names>
</name>
<name>
<surname>Sanchez-Rodriguez</surname> <given-names>E</given-names>
</name>
<name>
<surname>Zanoni</surname> <given-names>F</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Steers</surname> <given-names>N</given-names>
</name>
<etal/>
</person-group>. <article-title>Genetic regulation of serum IgA levels and susceptibility to common immune, infectious, kidney, and cardio-metabolic traits</article-title>. <source>Nat Commun</source> (<year>2022</year>) <volume>13</volume>(<issue>1</issue>):<fpage>6859</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41467-022-34456-6</pub-id>
</citation>
</ref>
<ref id="B39">
<label>39</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Trevisan</surname> <given-names>R</given-names>
</name>
<name>
<surname>Dodesini</surname> <given-names>AR</given-names>
</name>
<name>
<surname>Lepore</surname> <given-names>G</given-names>
</name>
</person-group>. <article-title>Lipids and renal disease</article-title>. <source>J Am Soc Nephrol JASN</source> (<year>2006</year>) <volume>17</volume>(<supplement>4 Suppl 2</supplement>):<page-range>S145&#x2013;7</page-range>. doi: <pub-id pub-id-type="doi">10.1681/ASN.2005121320</pub-id>
</citation>
</ref>
<ref id="B40">
<label>40</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Naimi</surname> <given-names>AI</given-names>
</name>
<name>
<surname>Balzer</surname> <given-names>LB</given-names>
</name>
</person-group>. <article-title>Stacked generalization: an introduction to super learning</article-title>. <source>Eur J Epidemiol</source> (<year>2018</year>) <volume>33</volume>(<issue>5</issue>):<page-range>459&#x2013;64</page-range>. doi: <pub-id pub-id-type="doi">10.1007/s10654-018-0390-z</pub-id>
</citation>
</ref>
<ref id="B41">
<label>41</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Myllym&#xe4;ki</surname> <given-names>JM</given-names>
</name>
<name>
<surname>Honkanen</surname> <given-names>TT</given-names>
</name>
<name>
<surname>Syrj&#xe4;nen</surname> <given-names>JT</given-names>
</name>
<name>
<surname>Helin</surname> <given-names>HJ</given-names>
</name>
<name>
<surname>Rantala</surname> <given-names>IS</given-names>
</name>
<name>
<surname>Pasternack</surname> <given-names>AI</given-names>
</name>
<etal/>
</person-group>. <article-title>Severity of tubulointerstitial inflammation and prognosis in immunoglobulin A nephropathy</article-title>. <source>Kidney Int</source> (<year>2007</year>) <volume>71</volume>(<issue>4</issue>):<page-range>343&#x2013;8</page-range>. doi: <pub-id pub-id-type="doi">10.1038/sj.ki.5002046</pub-id>
</citation>
</ref>
<ref id="B42">
<label>42</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hergesell</surname> <given-names>O</given-names>
</name>
<name>
<surname>Felten</surname> <given-names>H</given-names>
</name>
<name>
<surname>Andrassy</surname> <given-names>K</given-names>
</name>
<name>
<surname>K&#xfc;hn</surname> <given-names>K</given-names>
</name>
<name>
<surname>Ritz</surname> <given-names>E</given-names>
</name>
</person-group>. <article-title>Safety of ultrasound-guided percutaneous renal biopsy-retrospective analysis of 1090 consecutive cases</article-title>. <source>Nephrol Dial Transplant</source> (<year>1998</year>) <volume>13</volume>(<issue>4</issue>):<page-range>975&#x2013;7</page-range>. doi: <pub-id pub-id-type="doi">10.1093/ndt/13.4.975</pub-id>
</citation>
</ref>
<ref id="B43">
<label>43</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stiles</surname> <given-names>KP</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>CM</given-names>
</name>
<name>
<surname>Chung</surname> <given-names>EM</given-names>
</name>
<name>
<surname>Lyon</surname> <given-names>RD</given-names>
</name>
<name>
<surname>Lane</surname> <given-names>JD</given-names>
</name>
<name>
<surname>Abbott</surname> <given-names>KC</given-names>
</name>
</person-group>. <article-title>Renal biopsy in high-risk patients with medical diseases of the kidney</article-title>. <source>Am J Kidney Dis</source> (<year>2000</year>) <volume>36</volume>(<issue>2</issue>):<page-range>419&#x2013;33</page-range>. doi: <pub-id pub-id-type="doi">10.1053/ajkd.2000.8998</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>