<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1252159</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2023.1252159</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>An ensemble learning approach for diabetes prediction using boosting techniques</article-title>
<alt-title alt-title-type="left-running-head">Ganie et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2023.1252159">10.3389/fgene.2023.1252159</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Ganie</surname>
<given-names>Shahid Mohammad</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2053020/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Pramanik</surname>
<given-names>Pijush Kanti Dutta</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2012762/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Bashir Malik</surname>
<given-names>Majid</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Mallik</surname>
<given-names>Saurav</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/635395/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Qin</surname>
<given-names>Hong</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/928677/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>AI Research Centre</institution>, <institution>School of Business</institution>, <institution>Woxsen University</institution>, <addr-line>Hyderabad</addr-line>, <country>India</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>School of Computer Applications and Technology</institution>, <institution>Galgotias University</institution>, <addr-line>Greater Noida</addr-line>, <country>India</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Computer Science</institution>, <institution>Baba Ghulam Shah Badshah University</institution>, <addr-line>Rajauri</addr-line>, <country>India</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Department of Environmental Health</institution>, <institution>School of Public Health</institution>, <institution>Harvard University</institution>, <addr-line>Boston</addr-line>, <addr-line>MA</addr-line>, <country>United States</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>College of Engineering and Computer Science</institution>, <institution>University of Tennessee at Chattanooga</institution>, <addr-line>Chattanooga</addr-line>, <addr-line>TN</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/531759/overview">Quan Zou</ext-link>, University of Electronic Science and Technology of China, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2373425/overview">Fanying Tang</ext-link>, Novartis, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2374789/overview">Anqi Zou</ext-link>, Boston University, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Pijush Kanti Dutta Pramanik, <email>pijushjld@yahoo.co.in</email>; Saurav Mallik, <email>sauravmtech2@gmail.com</email>; Hong Qin, <email>hong-qin@utc.edu</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>26</day>
<month>10</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>14</volume>
<elocation-id>1252159</elocation-id>
<history>
<date date-type="received">
<day>03</day>
<month>07</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>16</day>
<month>10</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Ganie, Pramanik, Bashir Malik, Mallik and Qin.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Ganie, Pramanik, Bashir Malik, Mallik and Qin</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>
<bold>Introduction:</bold> Diabetes is considered one of the leading healthcare concerns affecting millions worldwide. Taking appropriate action at the earliest stages of the disease depends on early diabetes prediction and identification. To support healthcare providers for better diagnosis and prognosis of diseases, machine learning has been explored in the healthcare industry in recent years.</p>
<p>
<bold>Methods:</bold> To predict diabetes, this research has conducted experiments on five boosting algorithms on the Pima diabetes dataset. The dataset was obtained from the University of California, Irvine (UCI) machine learning repository, which contains several important clinical features. Exploratory data analysis was used to identify the characteristics of the dataset. Moreover, upsampling, normalisation, feature selection, and hyperparameter tuning were employed for predictive analytics.</p>
<p>
<bold>Results:</bold> The results were analysed using various statistical/machine learning metrics and k-fold cross-validation techniques. Gradient boosting achieved the greatest accuracy rate of 92.85% among all the classifiers. Precision, recall, f1-score, and receiver operating characteristic (ROC) curves were used to further validate the model.</p>
<p>
<bold>Discussion:</bold> The suggested model outperformed the current studies in terms of prediction accuracy, demonstrating its applicability to other diseases with similar predicate indications.</p>
</abstract>
<kwd-group>
<kwd>diabetes prediction</kwd>
<kwd>ensemble learning</kwd>
<kwd>XGBoost</kwd>
<kwd>CatBoost</kwd>
<kwd>LightGBM</kwd>
<kwd>AdaBoost</kwd>
<kwd>gradient boost</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Genomics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Diabetes mellitus is a severe and chronic disease characterised by metabolic disorders in which the pancreas either fails to produce insulin, or the body cannot effectively utilise the insulin produced (<xref ref-type="bibr" rid="B25">Sneha and Gangil, 2019</xref>). Lack of awareness about the symptoms and complications of diabetes is prevalent due to limited healthcare resources in many parts of the world (<xref ref-type="bibr" rid="B26">Webber, 2013</xref>). There are approximately 40 different types of diabetes, with some common types being Type 1 (insulin-dependent), Type 2 (insulin-independent), gestational diabetes, and pre-diabetes (<xref ref-type="bibr" rid="B15">Kharroubi and Darwish, 2015</xref>).</p>
<p>According to statistical reports from various healthcare organisations, it is estimated that globally, 463 million adults, which accounts for 9.3% of the population aged between 20 and 79 years, are affected by this chronic disease (<xref ref-type="bibr" rid="B3">Diabetes Federation International and IDF, 2019</xref>). This highlights the widespread prevalence and significance of diabetes as a global health issue.</p>
<p>Projections suggest that the prevalence of diabetes will continue to increase significantly, with an estimated 578 million individuals affected by 2030. According to the Diabetes Atlas 2019 by the International Diabetes Federation (IDF), approximately 50% or 231 million people living with diabetes remain undiagnosed and unaware of their condition due to limited healthcare resources (<xref ref-type="bibr" rid="B3">Diabetes Federation International and IDF, 2019</xref>).</p>
<p>In 2019 alone, diabetes was responsible for 4.2 million deaths worldwide. This chronic disease can have detrimental effects on various organs in the human body, including the brain, nerves, heart, kidneys, eyes, and skin. Recognising the symptoms and signs of diabetes is crucial for early detection and management. Some common early symptoms observed in individuals with diabetes or those at risk include excessive thirst, fatigue, unexplained weight gain, dizziness, skin discoloration, sexual dysfunction, fungal infections, high blood sugar levels, and frequent urination (<xref ref-type="bibr" rid="B25">Sneha and Gangil, 2019</xref>). These symptoms serve as important indicators for seeking medical attention and further evaluation.</p>
<p>Indeed, given the significant impact and global burden of diabetes, there is an urgent need to leverage computational intelligence techniques for improved prediction and prevention of this disease. By utilising advanced machine learning and artificial intelligence algorithms, we can develop models that can effectively identify individuals at risk of developing diabetes. These models can analyse large-scale datasets, extract meaningful patterns, and generate accurate predictions.</p>
<p>The application of computational intelligence techniques in diabetes prediction can have several benefits. Firstly, it can enable early disease detection, allowing for timely intervention and management strategies. This early detection can aid in preventing or delaying diabetes-related problems, improving overall health outcomes for people.</p>
<p>Furthermore, by accurately predicting diabetes, healthcare professionals can implement preventive measures and provide personalised care plans for high-risk individuals. This can involve lifestyle modifications, dietary interventions, exercise regimens, and medication management to effectively manage and control blood sugar levels.</p>
<p>Overall, applying computational intelligence techniques to diabetes prediction can significantly enhance medical results, lessen the condition&#x2019;s toll, and encourage proactive and preventative healthcare practices for those at risk.</p>
<p>However, healthcare data are growing drastically, and the traditional machine learning approaches have been found inadequate to handle such voluminous data for accurate disease predictions. Ensemble learning techniques offer better performance in this regard.</p>
<p>This work aims to create a model that accurately predicts diabetes using ensemble learning approaches. Our work&#x2019;s contribution is as follows:<list list-type="simple">
<list-item>
<p>&#x2022; Performing exploratory data analysis to improve the dataset&#x2019;s quality assessment.</p>
</list-item>
<list-item>
<p>&#x2022; Performing data augmentation and processing using upsampling and data normalisation, respectively.</p>
</list-item>
<list-item>
<p>&#x2022; Using a k-fold cross-validation procedure to confirm the results.</p>
</list-item>
<list-item>
<p>&#x2022; Building the model by employing boosting algorithms in conjunction with an ensemble learning strategy.</p>
</list-item>
<list-item>
<p>&#x2022; Increasing prediction accuracy through hyperparameter tuning.</p>
</list-item>
<list-item>
<p>&#x2022; Determining the contribution of the features towards diabetes.</p>
</list-item>
<list-item>
<p>&#x2022; Comparing the proposed model&#x2019;s performance assessment to other research studies of a similar nature.</p>
</list-item>
</list>
</p>
<p>The rest of the paper is organised as follows. Related work is discussed in <xref ref-type="sec" rid="s2">Section 2</xref>. The adopted methodology and the dataset are presented in <xref ref-type="sec" rid="s3">Section 3</xref>. Then, the experimental details and results are described and analysed in <xref ref-type="sec" rid="s4">Section 4</xref>. Next, the comparative analysis with existing similar works is presented in <xref ref-type="sec" rid="s5">Section 5</xref>. Lastly, the conclusion and the future direction of the research are provided in <xref ref-type="sec" rid="s6">Section 6</xref>.</p>
</sec>
<sec id="s2">
<title>2 Related work</title>
<p>In recent years, copious work has been done on the prediction of diabetes using machine learning and ensemble learning tools and techniques (<xref ref-type="bibr" rid="B7">Ganie et al., 2022a</xref>; <xref ref-type="bibr" rid="B5">Ganie and Malik, 2022a</xref>). Different datasets, algorithms, and methodologies used by the researchers to carry out this research work have been discussed. The developed models have yielded better results and can be used to support healthcare providers in data-driven decision-making. This section reviews some of the key relevant papers on applying ensemble learning approaches to forecast diabetes.</p>
<p>
<xref ref-type="bibr" rid="B17">Li et al. (2020)</xref> developed a model to predict diabetes using ensemble learning techniques to enhance disease prediction using the Pima diabetes dataset. They achieved the highest results with extreme gradient boosting (XGBoost), with an accuracy rate of 80.20%. The authors proposed the improved feature combination classifier using the XGBoost model, which can be explored to better predict diseases in the healthcare industry. <xref ref-type="bibr" rid="B19">Mahabub (2019)</xref> tested different ensemble learning techniques, such as AdaBoost, gradient boost, XGBoost, random forest, etc., to predict diabetes, considering several clinical parameters such as pregnancy, skin thickness, glucose, insulin, blood pressure, diabetes pedigree function, body mass index (BMI), age, and class variable (outcome). They achieved the highest accuracy rate of 84.42% with the multilayer perceptron algorithm. <xref ref-type="bibr" rid="B20">Mushtaq et al. (2022)</xref> proposed an optimised model using a voting classification based on the ensemble method to predict diabetes using the Pima diabetes dataset. This research work used a two-stage model selection process to develop the model. The voting classifier reached the best accuracy rate of 81.50% among all the classifiers. Furthermore, Tomek and synthetic minority oversampling technique (SMOTE) techniques were used for data balancing to remove the biases from the dataset. The authors suggested that the research be continued to estimate the likelihood that nondiabetic patients will develop this condition in the future.</p>
<p>
<xref ref-type="bibr" rid="B2">Beschi Raja et al. (2019)</xref> employed different boosting algorithms for the diabetes prediction model development. The gradient boosting algorithm attained the highest accuracy rate of 89.70% among all the classifiers. Other statistical measurements have also been evaluated to validate the proposed model. <xref ref-type="bibr" rid="B14">Khan et al. (2021)</xref> developed a model for diabetes prediction using boosting method. The authors explored different classifiers such as gradient boosting, hybrid k-nearest neighbour (kNN), j48, deep learning, naive Bayes, and artificial neural network (ANN) for predictive analytics. Among all the classifiers, the gradient boosting algorithm attained the best results. In addition, the results were validated using the k-fold cross-validation method. The authors suggested that this model can be used as a prognosis tool in the healthcare industry for early disease prediction. <xref ref-type="bibr" rid="B16">Lai et al. (2019)</xref> developed a complete framework for the predictive analysis of diabetes. The gradient boosting machine techniques were used with hyperparameter tuning, particularly for class balancing, which minimised the loss of prediction probabilities regarding classification.</p>
<p>
<xref ref-type="bibr" rid="B24">Singh et al. (2021)</xref> introduced an ensemble approach based framework called eDiaPredict to forecast the diabetes status of patients. The proposed methodology incorporates XGBoost, random forest, support vector machine (SVM), neural network, and decision tree. The efficacy of eDiaPredict is demonstrated through its implementation on the PIMA Indian diabetes dataset, resulting in an attained accuracy, precision, and sensitivity of 95%, 88%, and 90.32%, respectively, with the combination of XGBoost and random forest. <xref ref-type="bibr" rid="B11">Hasan et al. (2020)</xref> presented a framework for predicting diabetes using kNN, decision trees, random forest, AdaBoost, Naive Bayes, XGBoost, and multilayer perceptron. They employed a weighted ensemble of the machine learning models to improve the prediction accuracy, and experimented on the PIMA Indian diabetes dataset. The proposed ensemble model achieved a significantly higher AUC and specificity of 0.950 and 0.934, respectively. However, it exhibited lower accuracy, precision and sensitivity of 88.84%, 84.32%, and 78%, respectively.</p>
</sec>
<sec id="s3">
<title>3 Research methodology</title>
<p>
<xref ref-type="fig" rid="F1">Figure 1</xref> illustrates the procedural flow of the proposed framework employed in this experimental study. It outlines the sequential steps undertaken to enhance the prediction accuracy of diabetes using an ensemble learning technique based on boosting methods. The Pima Indians diabetes dataset, obtained from the Kaggle community, was utilised for this study. Initially, the necessary Python library packages were installed in Jupyter Notebook. Exploratory data analysis was conducted to enhance the dataset&#x2019;s quality assessment. During this phase, missing values were identified and replaced through data imputation. The Interquartile Range method was applied to detect outliers in the dataset (<xref ref-type="bibr" rid="B9">Ganie et al., 2023</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Proposed methodology for research work.</p>
</caption>
<graphic xlink:href="fgene-14-1252159-g001.tif"/>
</fig>
<p>Other necessary libraries were run to check the dataset for any corrupted data. Upsampling and normalising were also carried out before the five boosting methods under consideration were created. The dataset was split with a ratio of 80:20, where 80% of the data was employed for training the boosting algorithms, and 20% was used to test and validate their efficacy. Hyperparameter tuning was applied during the model-building process for better results.</p>
<sec id="s3-1">
<title>3.1 Boosting algorithms adopted</title>
<p>Ensemble learning has been utilised in several real-life problems (<xref ref-type="bibr" rid="B9">Ganie et al., 2023</xref>). In healthcare, ensemble learning has gained significant popularity due to its effectiveness in predicting, detecting, diagnosing, and prognosing various diseases. In this particular experiment focusing on diabetes prediction, we examined the following five boosting algorithms based on ensemble learning:<list list-type="simple">
<list-item>
<p>&#x2022; <bold>XGBoost:</bold> XGBoost operates by integrating diverse types of decision trees, also known as weak learners, to independently compute similarity scores (<xref ref-type="bibr" rid="B22">Santhanam et al., 2016</xref>). By incorporating gradient descent and regularisation techniques, XGBoost effectively addresses the issue of overfitting that can arise during the training phase. Modifying the gradient descent and regularisation procedure, it aids in overcoming the issue of overfitting during the training phase.</p>
</list-item>
<list-item>
<p>&#x2022; <bold>CatBoost:</bold> The CatBoost short form of categorical boosting is faster than other boosting algorithms, as it does not require the exploration of data preprocessing (<xref ref-type="bibr" rid="B10">Hancock and Khoshgoftaar, 2020</xref>). It is used to deal with high cardinality categorical variables. In the case of low cardinality variables, one-hot encoding technique is used for conversion.</p>
</list-item>
<list-item>
<p>&#x2022; <bold>LightGBM:</bold> Light gradient boosting machine (LightGBM) is an extension of a gradient boosting algorithm capable of handling large datasets with less memory utilisation during the model evaluation process (<xref ref-type="bibr" rid="B18">Machado et al., 2019</xref>). Gradient-based one-sided sampling method is used for splitting the data samples, reducing the number of features in sparse datasets during training.</p>
</list-item>
<list-item>
<p>&#x2022; <bold>AdaBoost:</bold> AdaBoost, also known as adaptive boosting, operates by dynamically adjusting weak learners&#x2019; weights without prior knowledge (<xref ref-type="bibr" rid="B23">Sevinc, 2022</xref>). During the training process, the weakness of each base learner is evaluated based on the estimator&#x2019;s error rate. The AdaBoost algorithm commonly employs decision tree stumps to address classification and regression problems.</p>
</list-item>
<list-item>
<p>&#x2022; <bold>Gradient boosting:</bold> The gradient boosting (GB) method trains weak learners in a sequential manner, with each estimator being added one by one by adjusting their weights (<xref ref-type="bibr" rid="B1">Aziz et al., 2020</xref>). This algorithm&#x2019;s main goal is to forecast residual errors from earlier estimators and reduce the difference between anticipated and actual values. This iterative process allows for continuous improvement in the overall predictive performance.</p>
</list-item>
</list>
</p>
</sec>
<sec id="s3-2">
<title>3.2 Attribute information</title>
<p>The dataset consists of 768 instances and nine attributes. The first eight attributes are independent variables, also known as predicates, while the last attribute is the dependent or target variable. <xref ref-type="table" rid="T1">Table 1</xref> provides detailed information about the attributes, including their descriptions, measurements, and range values.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Attributes information of the dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Attribute</th>
<th align="left">Description</th>
<th align="left">Measurement</th>
<th align="left">Value range</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Pregnancy (PR)</td>
<td align="left">Participant number of times pregnant</td>
<td align="left">Numeric</td>
<td align="left">0&#x2013;17</td>
</tr>
<tr>
<td align="left">Glucose (GL)</td>
<td align="left">Plasma glucose concentration of the participant</td>
<td align="left">mg/dL</td>
<td align="left">0&#x2013;199</td>
</tr>
<tr>
<td align="left">Blood pressure (BP)</td>
<td align="left">Diastolic blood pressure of the participant</td>
<td align="left">mmHg</td>
<td align="left">0&#x2013;122</td>
</tr>
<tr>
<td align="left">Skin thickness (ST)</td>
<td align="left">Triceps skin fold thickness of the participant</td>
<td align="left">mm</td>
<td align="left">0&#x2013;99</td>
</tr>
<tr>
<td align="left">Insulin (IN)</td>
<td align="left">Participant&#x2019;s insulin level (2-h serum)</td>
<td align="left">(mu U/mL)</td>
<td align="left">0&#x2013;846</td>
</tr>
<tr>
<td align="left">Body mass index (BMI)</td>
<td align="left">Body fat based on the height and weight of the participant</td>
<td align="left">kg/m<sup>2</sup>
</td>
<td align="left">0&#x2013;67</td>
</tr>
<tr>
<td align="left">Diabetes pedigree function (DPF)</td>
<td align="left">Likelihood of diabetes based on the family history of the participant</td>
<td align="left">
<italic>p</italic>-value</td>
<td align="left">0.07&#x2013;2.42</td>
</tr>
<tr>
<td align="left">Age (AG)</td>
<td align="left">Age of the participant</td>
<td align="left">Years</td>
<td align="left">21&#x2013;81</td>
</tr>
<tr>
<td align="left">Diabetes (DB)</td>
<td align="left">Class attribute</td>
<td align="left">0 &#x3d; no diabetes, 1 &#x3d; diabetes</td>
<td align="left">0 or 1</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-3">
<title>3.3 Dataset description</title>
<p>Descriptive statistics are crucial in revealing the characteristics of data samples, summarising information to facilitate human interpretation. <xref ref-type="table" rid="T2">Table 2</xref> presents attribute information along with their corresponding measures, including the record count, minimum (min) value, maximum (max) value, mean, and standard deviation (std). For example, the Pregnancy (PR) attribute has a record count of 786, a mean value of 3.84, a standard deviation of 3.36, and the maximum and minimum PR values are 17 and 0, respectively. Similar statistical measurements have been computed for the remaining attributes as well. These metrics provide valuable insights into the data distribution and properties.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Attributes information of the dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Attribute</th>
<th align="left">Count</th>
<th align="left">Mean</th>
<th align="left">Std</th>
<th align="left">Min</th>
<th align="left">Max</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">PR</td>
<td rowspan="9" align="left">768</td>
<td align="left">3.84</td>
<td align="left">3.36</td>
<td align="left">0</td>
<td align="left">17</td>
</tr>
<tr>
<td align="left">GL</td>
<td align="left">120.89</td>
<td align="left">31.97</td>
<td align="left">0</td>
<td align="left">199</td>
</tr>
<tr>
<td align="left">BP</td>
<td align="left">69.10</td>
<td align="left">19.35</td>
<td align="left">0</td>
<td align="left">122</td>
</tr>
<tr>
<td align="left">ST</td>
<td align="left">20.53</td>
<td align="left">15.95</td>
<td align="left">0</td>
<td align="left">99</td>
</tr>
<tr>
<td align="left">IN</td>
<td align="left">79.79</td>
<td align="left">115.24</td>
<td align="left">0</td>
<td align="left">846</td>
</tr>
<tr>
<td align="left">BMI</td>
<td align="left">31.99</td>
<td align="left">7.88</td>
<td align="left">0</td>
<td align="left">67.10</td>
</tr>
<tr>
<td align="left">DPF</td>
<td align="left">0.47</td>
<td align="left">0.33</td>
<td align="left">0.78</td>
<td align="left">2.42</td>
</tr>
<tr>
<td align="left">AG</td>
<td align="left">33.24</td>
<td align="left">11.76</td>
<td align="left">21</td>
<td align="left">81</td>
</tr>
<tr>
<td align="left">DB</td>
<td align="left">0.34</td>
<td align="left">0.47</td>
<td align="left">0</td>
<td align="left">1</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-4">
<title>3.4 Histogram of attributes</title>
<p>A histogram is a useful tool for visualising and understanding the distribution of data samples in a dataset. It provides insights into whether the data follows a uniform, normal, left-skewed, or right-skewed distribution. In <xref ref-type="fig" rid="F2">Figure 2</xref>, normally distributed histograms are presented, depicting the grouping of all attributes within their respective range values. This visualisation helps to better understand the data distribution and identify any patterns or anomalies present in the dataset. The X-axis describes the input attributes, and the Y-axis presents the value of that attributes. The distribution of attributes for nondiabetics and diabetics is shown in <xref ref-type="fig" rid="F3">Figure 3</xref>. In the figure, 0 (blue color) and 1 (orange color) represent nondiabetic and diabetic patients, respectively. It can be seen that in most of the attribute combinations, the tendency of being diabetic increases when their respective range values increase. For example, in the age vs. glucose level plot, we understand that patients more than 30 years with glucose levels more than 125 are more likely to be diabetic patients.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Histogram of attributes.</p>
</caption>
<graphic xlink:href="fgene-14-1252159-g002.tif"/>
</fig>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Distribution of attributes for nondiabetics and diabetics.</p>
</caption>
<graphic xlink:href="fgene-14-1252159-g003.tif"/>
</fig>
</sec>
<sec id="s3-5">
<title>3.5 Boxplot for each attribute</title>
<p>
<xref ref-type="fig" rid="F4">Figure 4</xref> depicts the boxplot of all the considered attributes of the dataset. It provides a good indication of how the dispersion of values is spread out. The Interquartile Range (IQR) method, based on the probability density function, has been employed to display boxplots for the characteristics to manage outliers in the dataset. This approach aids in visually representing data distribution, particularly focusing on the median, quartiles, and any potential outliers. By incorporating the IQR method, the boxplots provide valuable insights into the central tendency and variability of each attribute, while effectively addressing and visualising the presence of outliers.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Boxplot of attributes.</p>
</caption>
<graphic xlink:href="fgene-14-1252159-g004.tif"/>
</fig>
</sec>
<sec id="s3-6">
<title>3.6 Correlation coefficient analysis</title>
<p>The dataset&#x2019;s attribute associations are examined and visualised using the correlation coefficient analysis (CCA) approach (<xref ref-type="bibr" rid="B12">Hussain and Naaz, 2021</xref>). A high correlation between the independent qualities set and the dependent attribute is desired to judge a good dataset (<xref ref-type="bibr" rid="B9">Ganie et al., 2023</xref>). The CCA plot of every variable used to predict disease is shown in <xref ref-type="fig" rid="F5">Figure 5</xref>. The intensity and direction of the correlations between the qualities are shown by the x-axis and y-axis, which describe the range of associations and range from &#x2b;1 to &#x2212;1. The interdependencies between the variables in the dataset are better understood and identified thanks to this study.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Correlation coefficient analysis.</p>
</caption>
<graphic xlink:href="fgene-14-1252159-g005.tif"/>
</fig>
</sec>
</sec>
<sec id="s4">
<title>4 Experiment, results, and discussion</title>
<p>The experimental minutiae and findings generated by the application of boosting algorithms for diabetes prediction are presented and discussed in this part. The outcomes obtained after using the suggested framework are methodically presented and examined. The evaluation is carried out thoroughly, considering several measures for the evaluated boosting algorithms, including accuracy, recall, precision, F1-score, micro-weighted score, average weighted score, and the receiver operating characteristic (ROC) curve. These measures give us important information about how well the boosting algorithms perform and how well they forecast diabetes.</p>
<sec id="s4-1">
<title>4.1 System specification</title>
<p>The research work was conducted using an HP Z60 workstation with the following hardware specifications: Intel XEON 2.4&#xa0;GHz CPU (12 core), 8&#xa0;GB RAM, 1&#xa0;TB hard disk, and running on Windows 10 Pro 64-bit operating system.</p>
<p>The tools utilised for implementation included Python as the programming language, the web-based computing platform Jupyter Notebook, and the graphical user interface-based Anaconda Navigator.</p>
</sec>
<sec id="s4-2">
<title>4.2 Data preprocessing</title>
<p>Data preparation is essential in creating a strong and reliable system before applying machine learning techniques to the model (<xref ref-type="bibr" rid="B13">Jazayeri et al., 2020</xref>). In this work, various strategies were used to manage various data preparation issues.</p>
<p>Firstly, missing values were located and dealt with using the data imputation method. All of the missing values were found using the isnull() function, and they were then filled using the mean and mode imputation method and the SimpleImputer() method. With this method, the mean, median, or mode of the relevant column was used to fill in the gaps left by the missing data.</p>
<p>To handle outliers, the IQR method was applied. The distribution of each data sample was altered using the Z-score to make the mean equal to 0. This process helped in identifying and replacing outliers in the dataset.</p>
<p>Furthermore, data cleaning methods were employed to address duplication, inconsistency, and corrupted data. These techniques ensured the integrity and reliability of the dataset by removing or resolving any duplicate records, inconsistent values, or corrupted data points.</p>
<p>By implementing these data preprocessing techniques, the dataset was prepared and optimised for subsequent machine learning methods, enhancing the quality and reliability of the analysis.</p>
</sec>
<sec id="s4-3">
<title>4.3 Data upsampling</title>
<p>If the dataset is not balanced, machine learning and deep learning algorithms produce subpar outcomes (<xref ref-type="bibr" rid="B9">Ganie et al., 2023</xref>). In this work, the dataset was highly biased toward the negative class, i.e., &#x201c;0-non-diabetic&#x201d; over the positive class &#x201c;1-diabetic.&#x201d; Initially, out of 786 instances, 500 records were negative class, whereas only 268 instances were held for positive class. After splitting, we had 614 records in the training dataset, in which 396 was for non-diabetic and 218 for diabetic. To balance the training set, the SMOTE was used, as shown in <xref ref-type="fig" rid="F6">Figure 6</xref>.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Upsampling technique for class balancing in training dataset. <bold>(A)</bold> Before SMOTE, <bold>(B)</bold> After SMOTE.</p>
</caption>
<graphic xlink:href="fgene-14-1252159-g006.tif"/>
</fig>
</sec>
<sec id="s4-4">
<title>4.4 Data normalisation</title>
<p>Normalisation is a part of the feature scaling process that fits the data samples into a specific range. The nature of the dataset can determine the range of values. Mostly, the values fit between the range of 0&#x2013;1. In our study, we used min-max scaling to bring the attribute values between 0 and 1. The mathematical expression used to perform data min-max scaling is given in Eq. <xref ref-type="disp-formula" rid="e1">1</xref>.<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi mathvariant="italic">min</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi mathvariant="italic">min</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <italic>x</italic> is the attribute value, and <italic>x</italic>
<sub>
<italic>min</italic>
</sub> and <italic>x</italic>
<sub>
<italic>max</italic>
</sub> denote the minimum and maximum values of <italic>x</italic>, respectively.</p>
</sec>
<sec id="s4-5">
<title>4.5 K-fold cross validation</title>
<p>K-fold cross validation is typically used to remove the biasness in the dataset. In this method, the dataset is partitioned into <italic>k</italic> approximately equal-sized subsets, also known as &#x201c;folds&#x201d;. In this experiment, applied K-fold cross validation on the training dataset and got the best result using the value of <italic>k</italic> as 10. The results in the following sections are based on this value.</p>
</sec>
<sec id="s4-6">
<title>4.6 Hyperparameter tuning</title>
<p>Hyperparameter tuning is important because it controls the training algorithm&#x2019;s behavior and significantly impacts the model&#x2019;s performance evaluation. Grid search and random search methods were used for hyperparameter tuning, as presented in <xref ref-type="table" rid="T3">Table 3</xref>. The listed values for each parameter for the respective algorithms were found to be the best performers in our experiment.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Hyperparameter tuning of boosting algorithms.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Boosting algorithm</th>
<th align="left">Hyperparameters</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">XGBoost</td>
<td align="left">learning_rate &#x3d; 0.01, n_estimators &#x3d; 1000, max_depth &#x3d; 4, min_child_weight &#x3d; 8, subsample &#x3d; 0.6, reg_alpha &#x3d; 0.005, seed &#x3d; 27</td>
</tr>
<tr>
<td align="left">CatBoost</td>
<td align="left">learning_rate &#x3d; 0.010, 0.004, &#x201c;depth&#x201d; &#x3d; 4, leaf_reg&#x2019;, 1.0, min_child_samples &#x3d; 1, 4, 8, 16, 32, iterations &#x3d; 3000, random_state &#x3d; 42</td>
</tr>
<tr>
<td align="left">LightGBM</td>
<td align="left">boosting_type &#x3d; &#x201c;lgbm&#x201d;, class_weight &#x3d; Auto, min_child_weight &#x3d; 0.01, random_state &#x3d; 124, num_leaves &#x3d; 11, n_estimators &#x3d; 1500, n_jobs &#x3d; 6</td>
</tr>
<tr>
<td align="left">AdaBoost</td>
<td align="left">learning_rate &#x3d; [0.0001, 0.001, 0.01, 0.1, 1.0], base_estimator &#x3d; base, grid_search &#x3d; GridSearchCV, param_grid &#x3d; grid, parameters, cv &#x3d; 5, n_jobs &#x3d; n_jobs</td>
</tr>
<tr>
<td align="left">Gradient boosting</td>
<td align="left">learning_rate &#x3d; 0.01, n_estimators &#x3d; 100000, max_depth &#x3d; 8, colsample_bytree &#x3d; 0.8, reg_alpha &#x3d; 0.002, scoring &#x3d; roc_curve, weight &#x3d; 4, subsample &#x3d; 0.6, seed &#x3d; 23</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4-7">
<title>4.7 Feature importance</title>
<p>Based on their contribution to forecasting the output feature (target variable), the feature significance procedure assigns scores to input attributes (predicate variables) (<xref ref-type="bibr" rid="B4">Dutta et al., 2019</xref>). This phase is essential for machine learning or ensemble learning models to produce better predictions.</p>
<p>The feature significance score (F-score), which measures how frequently an attribute is used for splitting during training, is employed in this study. A characteristic, such as DPF (Diabetes Pedigree Function), with a higher F-score is considered an essential attribute since it contributes more significantly to the prediction process.</p>
<p>According to their relative F-scores for each boosting algorithm, <xref ref-type="fig" rid="F7">Figure 7</xref> displays the contribution of all attributes to the prediction task. It can be observed that overall, age, BMI, and skin thickness are the most common indicators of the patient having diabetes. Increased glucose level and high blood pressure are also a matter of concern. Out of eight features, none of them were found to be absolutely insignificant for diabetes.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Feature importance for prediction using <bold>(A)</bold> XGBoost, <bold>(B)</bold> CatBoost, <bold>(C)</bold> LightGBM, <bold>(D)</bold> AdaBoost, and <bold>(E)</bold> Gradient boosting.</p>
</caption>
<graphic xlink:href="fgene-14-1252159-g007.tif"/>
</fig>
</sec>
<sec id="s4-8">
<title>4.8 Accuracy of classifiers</title>
<p>The testing accuracy (calculated using Eq. <xref ref-type="disp-formula" rid="e2">2</xref>) (<xref ref-type="bibr" rid="B21">Pramanik et al., 2020</xref>) of the boosting algorithms (i.e., XGBoost, CatBoost, LightGBM, AdaBoost, and gradient boosting) is presented in <xref ref-type="fig" rid="F8">Figure 8</xref>. The figure presents a comparison of the accuracy of the considered algorithms before and after conducting data processing, augmentation and hyperparameter tuning. It can be observed that before data processing CatBoost performed best with the highest accuracy of 81.81%. In comparison, gradient boosting emerged as the top performer after data processing, with the highest accuracy of 96.75%.<disp-formula id="e2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where TN: true negative, TP: true positive, FN: false negative, and FP: false positive.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Accuracy of all the boosting algorithms before and after data processing, augmentation and hyperparameter tuning.</p>
</caption>
<graphic xlink:href="fgene-14-1252159-g008.tif"/>
</fig>
</sec>
<sec id="s4-9">
<title>4.9 Confusion matrices</title>
<p>The performance evaluation of all classifiers was evaluated using a confusion matrix. The confusion matrices of all considered boosting algorithms are shown in <xref ref-type="fig" rid="F9">Figure 9</xref>.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Confusion matrices of <bold>(A)</bold> XGBoost, <bold>(B)</bold> CatBoost, <bold>(C)</bold> LightGBM, <bold>(D)</bold> AdaBoost, and <bold>(E)</bold> Gradient boosting.</p>
</caption>
<graphic xlink:href="fgene-14-1252159-g009.tif"/>
</fig>
</sec>
<sec id="s4-10">
<title>4.10 Other measurements</title>
<p>The precision (Eq. <xref ref-type="disp-formula" rid="e3">3</xref>) (<xref ref-type="bibr" rid="B7">Ganie et al., 2022a</xref>), recall (Eq. <xref ref-type="disp-formula" rid="e4">4</xref>) (<xref ref-type="bibr" rid="B6">Ganie and Malik, 2022b</xref>), and f1-score (Eq. <xref ref-type="disp-formula" rid="e5">5</xref>) (<xref ref-type="bibr" rid="B8">Ganie et al., 2022b</xref>) of the five considered classifiers were calculated. Furthermore, the macro average and the weighted average were measured for both classes (0: no diabetes, 1: diabetes), as shown in <xref ref-type="fig" rid="F10">Figure 10</xref>. On average, gradient boosting exhibited better results than other models in all respects. However, in the case no diabetes precision the performance of gradient boosting is at par with XGBoost and Light GBM. In most of the cases, XGBoost and Light GBM exhibited similar performances while in some cases, the performance of CatBoost and AdaBoost are found equivalent.<disp-formula id="e3">
<mml:math id="m3">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
<disp-formula id="e4">
<mml:math id="m4">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
<disp-formula id="e5">
<mml:math id="m5">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>/</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>Other measurements of classifiers.</p>
</caption>
<graphic xlink:href="fgene-14-1252159-g010.tif"/>
</fig>
</sec>
<sec id="s4-11">
<title>4.11 ROC curve</title>
<p>The prediction ability of the discussed boosting algorithms is evaluated at various levels using the receiver operating characteristic (ROC) curve. On the y-axis, it displays the true-positive rate (TPR) (Eq. <xref ref-type="disp-formula" rid="e7">7</xref>) (<xref ref-type="bibr" rid="B21">Pramanik et al., 2020</xref>)) and on the x-axis, the false-positive rate (FPR) (Eq. <xref ref-type="disp-formula" rid="e6">6</xref>) (<xref ref-type="bibr" rid="B21">Pramanik et al., 2020</xref>). We may assess how well the models can differentiate between the two classes&#x2014;0 (non-diabetic) and 1 (diabetic)&#x2014;by examining the ROC curve.<disp-formula id="e6">
<mml:math id="m6">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>R</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>/</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
<disp-formula id="e7">
<mml:math id="m7">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>R</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>/</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
</p>
<p>A higher ROC curve indicates that the model performs well in differentiating between the two classes (<xref ref-type="bibr" rid="B9">Ganie et al., 2023</xref>). Moreover, the area under the ROC curve (AUC) is used as a measure of separability. An AUC value close to 1 indicates a good separability measure, while a value close to 0 signifies a poor measure of discrimination. A value of 0.5 suggests that the model is not effectively separating the classes.</p>
<p>
<xref ref-type="fig" rid="F11">Figure 11</xref> displays the ROC curves for XGBoost, CatBoost, LightGBM, AdaBoost, and gradient boosting. Based on the curves, gradient boosting performed the best, while AdaBoost exhibited the poorest performance among the considered boosting algorithms.</p>
<fig id="F11" position="float">
<label>FIGURE 11</label>
<caption>
<p>The ROC curves for <bold>(A)</bold> XGBoost, <bold>(B)</bold> CatBoost, <bold>(C)</bold> LightGBM, <bold>(D)</bold> AdaBoost, and <bold>(E)</bold> Gradient boosting.</p>
</caption>
<graphic xlink:href="fgene-14-1252159-g011.tif"/>
</fig>
</sec>
<sec id="s4-12">
<title>4.12 AUPRC</title>
<p>Area Under the Precision-Recall Curve (AUPRC) metric is employed to assess the machine learning model&#x2019;s performance, distinguishing between a positive class and a negative class. It illustrates the relationship between precision, representing the positive predictive value, and recall, indicating sensitivity or the genuine positive rate. This curve is constructed by considering different probability thresholds for the positive class. The AUPRC for our proposed model is shown in <xref ref-type="fig" rid="F12">Figure 12</xref>, from which it is observed that gradient boosting and AdaBoost have the best and worst performances, respectively.</p>
<fig id="F12" position="float">
<label>FIGURE 12</label>
<caption>
<p>The AUPRC for the experimented boosting algorithms.</p>
</caption>
<graphic xlink:href="fgene-14-1252159-g012.tif"/>
</fig>
</sec>
</sec>
<sec id="s5">
<title>5 Comparative analysis</title>
<p>
<xref ref-type="fig" rid="F13">Figure 13</xref> presents a comparative analysis of the five boosting algorithms considered in the experiment. The algorithms were compared in terms of accuracy, AUC value, and runtimes. Among these algorithms, gradient boosting achieved the highest accuracy rate, reaching a maximum accuracy of 96.75%. Following gradient boosting, LightGBM achieved an accuracy rate of 94.15%, AdaBoost achieved 91.55%, CatBoost achieved 92.2%, and XGBoost achieved 93.5%. In addition, gradient boosting also excels in terms of AUC. However, in terms of runtime XGBoost outperforms others, requiring the least runtime.</p>
<fig id="F13" position="float">
<label>FIGURE 13</label>
<caption>
<p>Comparative analysis of the considered algorithms in terms of <bold>(A)</bold> accuracy, <bold>(B)</bold> AUC, and <bold>(C)</bold> runtime (in seconds).</p>
</caption>
<graphic xlink:href="fgene-14-1252159-g013.tif"/>
</fig>
<p>We compared the highest accuracy achieved by our proposed method (i.e., using gradient boosting) with several relevant literature in terms of accuracy, as shown in <xref ref-type="table" rid="T4">Table 4</xref>. The implemented processes, such as data imputation for handling missing values, detection, and box plotting for outlier elimination, could be credited with the reason for achieving improved accuracy.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Comparison of the proposed work with existing similar works.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Research work</th>
<th align="left">Adopted ensemble methods</th>
<th align="left">Dataset used</th>
<th align="left">Highest accuracy</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<xref ref-type="bibr" rid="B17">Li et al. (2020)</xref>
</td>
<td align="left">XGBoost, XGBoost &#x2b; logistic regression, data feature stitching &#x2b; XGBoost</td>
<td align="left">PIMA Indian diabetes dataset</td>
<td align="left">80.20% with data feature stitching &#x2b; XGBoost</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B19">Mahabub (2019)</xref>
</td>
<td align="left">kNN, AdaBoost, decision tree, random forest, support vector classification, gradient boosting, multilayer perceptron, XGBoost, gaussian naive Bayes</td>
<td align="left">Do</td>
<td align="left">84.42% with multilayer perceptron</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B20">Mushtaq et al. (2022)</xref>
</td>
<td align="left">kNN, random forest, naive Bayes, SVM, gradient boosting, logistic regression, and voting classifier</td>
<td align="left">Do</td>
<td align="left">81.30% with voting classifier</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B2">Beschi Raja et al. (2019)</xref>
</td>
<td align="left">Neural networks, random forest, and GBC</td>
<td align="left">Do</td>
<td align="left">76.10% with GBC</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B14">Khan et al. (2021)</xref>
</td>
<td align="left">Gradient boosting, hybrid K-mean, J48, decision tree, deep learning, naive Bayes, and ANN</td>
<td align="left">Do</td>
<td align="left">92% with gradient boosting algorithm</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B24">Singh et al. (2021)</xref>
</td>
<td align="left">XGBoost, random forest, SVM, neural network, and decision tree</td>
<td align="left">Do</td>
<td align="left">95% with XGBoost and random forest</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B11">Hasan et al. (2020)</xref>
</td>
<td align="left">kNN, decision trees, random forest, AdaBoost, naive Bayes, XGBoost, and multilayer perceptron</td>
<td align="left">Do</td>
<td align="left">88.84 with AdaBoost &#x2b; XGboost</td>
</tr>
<tr>
<td align="left">This paper</td>
<td align="left">XGBoost, CatBoost, LightGBM, AdaBoost, and gradient boosting</td>
<td align="left">Do</td>
<td align="left">96.75% with gradient boosting</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s6">
<title>6 Conclusion and future scope</title>
<p>In this research, we investigated the effectiveness of five boosting algorithms, namely, XGBoost, CatBoost, LightGBM, AdaBoost, and gradient boosting, for predicting diabetes disease. Various preprocessing techniques, such as imputation, Z-score, and cleaning methods, were applied to improve the quality of the dataset. Additionally, to enhance disease prediction, data normalisation, upsampling, and hyperparameter tuning were performed.</p>
<p>According to the experimental findings, gradient boosting had the greatest accuracy rate of 96%. Additionally, it did well in terms of other evaluation criteria like ROC curve, precision, recall, and f1-score. The feature importance technique revealed how independent features contributed to the outcome of the final prediction.</p>
<p>Furthermore, when compared to similar related efforts, the suggested framework performed better than existing systems. Other ensemble learning strategies, such as bagging and stacking, can be added to further increase the quality of the outcomes. To increase the scope of this research, the proposed method can also be used for other healthcare datasets with comparable features.</p>
<p>In future studies, exploring deep learning techniques could lead to better detection and prediction of diabetes. These advancements in machine learning and deep learning can contribute to more accurate and efficient healthcare solutions.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/Supplementary Material, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s8">
<title>Author contributions</title>
<p>SG and PP conceived the method and design. SG, PP, and MB conducted the experiment, and PP and MB analyzed the results. SG, PP, and SM wrote the manuscript. SM and HQ reviewed and edited the manuscript. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec id="s9">
<title>Funding</title>
<p>HQ thanks USA NSF 1761839 and 2200138, a catalyst award from the USA National Academy of Medicine, AI Tennessee Initiative, and internal support of the University of Tennessee at Chattanooga.</p>
</sec>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Aziz</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Akhir</surname>
<given-names>E. A. P.</given-names>
</name>
<name>
<surname>Aziz</surname>
<given-names>I. A.</given-names>
</name>
<name>
<surname>Jaafar</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hasan</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Abas</surname>
<given-names>A. N. C.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>A study on gradient boosting algorithms for development of AI monitoring and prediction systems</article-title>,&#x201d; in <conf-name>Proceedings of the International Conference on Computational Intelligence (ICCI)</conf-name>, <conf-loc>Malaysia</conf-loc>, <conf-date>October 2020</conf-date>, <fpage>11</fpage>&#x2013;<lpage>16</lpage>.</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Beschi Raja</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Anitha</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Sujatha</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Roopa</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Sam Peter</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Diabetics prediction using gradient boosted classifier</article-title>. <source>Int. J. Eng. Adv. Technol.</source> <volume>9</volume> (<issue>1</issue>), <fpage>3181</fpage>&#x2013;<lpage>3183</lpage>. <pub-id pub-id-type="doi">10.35940/ijeat.a9898.109119</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="book">
<collab>Diabetes Federation International and IDF</collab> (<year>2019</year>). <source>IDF diabetes Atlas 2019</source>.</citation>
</ref>
<ref id="B4">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Dutta</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Paul</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ghosh</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Analysing feature importances for diabetes prediction using machine learning</article-title>,&#x201d; in <conf-name>Proceedings of the 2018 IEEE 9th Annual Information Technology, Electronics and Mobile Communication Conference (IEMCON)</conf-name>, <conf-loc>Vancouver, Canada</conf-loc>, <conf-date>November 2018</conf-date>, <fpage>924</fpage>&#x2013;<lpage>928</lpage>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ganie</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Malik</surname>
<given-names>M. B.</given-names>
</name>
</person-group> (<year>2022a</year>). <article-title>Comparative analysis of various supervised machine learning algorithms for the early prediction of type-II diabetes mellitus</article-title>. <source>Int. J. Med. Eng. Inf.</source> <volume>14</volume> (<issue>6</issue>), <fpage>473</fpage>&#x2013;<lpage>483</lpage>. <pub-id pub-id-type="doi">10.1504/ijmei.2022.126519</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ganie</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Malik</surname>
<given-names>M. B.</given-names>
</name>
</person-group> (<year>2022b</year>). <article-title>An ensemble machine Learning approach for predicting Type-II diabetes mellitus based on lifestyle indicators</article-title>. <source>Healthc. Anal.</source> <volume>2</volume>, <fpage>100092</fpage>. <pub-id pub-id-type="doi">10.1016/j.health.2022.100092</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ganie</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Malik</surname>
<given-names>M. B.</given-names>
</name>
<name>
<surname>Arif</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2022a</year>). <article-title>Performance analysis and prediction of type 2 diabetes mellitus based on lifestyle data using machine learning approaches</article-title>. <source>J. Diabetes &#x26; Metabolic Disord.</source> <volume>21</volume>, <fpage>339</fpage>&#x2013;<lpage>352</lpage>. <pub-id pub-id-type="doi">10.1007/s40200-022-00981-w</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ganie</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Malik</surname>
<given-names>M. B.</given-names>
</name>
<name>
<surname>Arif</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2022b</year>). &#x201c;<article-title>Machine learning techniques for diagnosis of type 2 diabetes using lifestyle data</article-title>,&#x201d; in <conf-name>Proceedings of the International Conference on Innovative Computing and Communications</conf-name>, <conf-loc>New Delhi, India</conf-loc>, <conf-date>August 2021</conf-date>, <fpage>487</fpage>&#x2013;<lpage>497</lpage>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ganie</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Pramanik</surname>
<given-names>P. K. D.</given-names>
</name>
<name>
<surname>Malik</surname>
<given-names>M. B.</given-names>
</name>
<name>
<surname>Nayyar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kwak</surname>
<given-names>K. S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>An improved ensemble learning approach for heart disease prediction using boosting algorithms</article-title>. <source>Comput. Syst. Sci. Eng.</source> <volume>46</volume> (<issue>3</issue>), <fpage>3993</fpage>&#x2013;<lpage>4006</lpage>. <pub-id pub-id-type="doi">10.32604/csse.2023.035244</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hancock</surname>
<given-names>J. T.</given-names>
</name>
<name>
<surname>Khoshgoftaar</surname>
<given-names>T. M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>CatBoost for big data: an interdisciplinary review</article-title>. <source>J. Big Data</source> <volume>7</volume> (<issue>1</issue>), <fpage>94</fpage>. <pub-id pub-id-type="doi">10.1186/s40537-020-00369-8</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hasan</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Alam</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Das</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Hossain</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Hasan</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Diabetes prediction using ensembling of different machine learning classifiers</article-title>. <source>IEEE Access</source> <volume>8</volume>, <fpage>76516</fpage>&#x2013;<lpage>76531</lpage>. <pub-id pub-id-type="doi">10.1109/access.2020.2989857</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Hussain</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Naaz</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Prediction of diabetes mellitus: comparative study of various machine learning models</article-title>,&#x201d; in <conf-name>Proceedings of the International Conference on Innovative Computing and Communications. Advances in Intelligent Systems and Computing</conf-name>, <conf-loc>Singapore</conf-loc>, <conf-date>July 2021</conf-date> (<publisher-name>Springer</publisher-name>), <fpage>103</fpage>&#x2013;<lpage>115</lpage>.</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jazayeri</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>O. S.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>C. C.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Imputation of missing data in electronic health records based on patients&#x27; similarities</article-title>. <source>J. Healthc. Inf. Res.</source> <volume>4</volume> (<issue>3</issue>), <fpage>295</fpage>&#x2013;<lpage>307</lpage>. <pub-id pub-id-type="doi">10.1007/s41666-020-00073-5</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Khan</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Qayyum</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liaqat</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Ahmad</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Nawaz</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Younis</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Optimised prediction model for type 2 diabetes mellitus using gradient boosting algorithm</article-title>,&#x201d; in <conf-name>Proceedings of the 2021 Mohammad Ali Jinnah University International Conference on Computing (MAJICC)</conf-name>, <conf-loc>Karachi, Pakistan</conf-loc>, <conf-date>July 2021</conf-date>, <fpage>1</fpage>&#x2013;<lpage>6</lpage>.</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kharroubi</surname>
<given-names>A. T.</given-names>
</name>
<name>
<surname>Darwish</surname>
<given-names>H. M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Diabetes mellitus: the epidemic of the century</article-title>. <source>World J. Diabetes</source> <volume>6</volume> (<issue>6</issue>), <fpage>850</fpage>&#x2013;<lpage>867</lpage>. <pub-id pub-id-type="doi">10.4239/wjd.v6.i6.850</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lai</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Keshavjee</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Guergachi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Predictive models for diabetes mellitus using machine learning techniques</article-title>. <source>BMC Endocr. Disord.</source> <volume>19</volume> (<issue>1</issue>), <fpage>101</fpage>&#x2013;<lpage>109</lpage>. <pub-id pub-id-type="doi">10.1186/s12902-019-0436-6</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Diabetes prediction based on XGBoost algorithm</article-title>. <source>IOP Conf. Ser. Mater. Sci. Eng.</source> <volume>768</volume> (<issue>7</issue>), <fpage>072093</fpage>. <pub-id pub-id-type="doi">10.1088/1757-899x/768/7/072093</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Machado</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Karray</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>de Sousa</surname>
<given-names>I. T.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>LightGBM: an effective decision tree gradient boosting method to predict customer loyalty in the finance industry</article-title>,&#x201d; in <conf-name>Proceedings of the International Conference on Computer Science &#x26; Education (ICCSE)</conf-name>, <conf-loc>Toronto, Canada</conf-loc>, <conf-date>August 2019</conf-date>, <fpage>1111</fpage>&#x2013;<lpage>1116</lpage>.</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mahabub</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A robust voting approach for diabetes prediction using traditional machine learning techniques</article-title>. <source>SN Appl. Sci.</source> <volume>1</volume> (<issue>12</issue>), <fpage>1667</fpage>&#x2013;<lpage>1712</lpage>. <pub-id pub-id-type="doi">10.1007/s42452-019-1759-7</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mushtaq</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ramzan</surname>
<given-names>M. F.</given-names>
</name>
<name>
<surname>Ali</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Baseer</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Samad</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Husnain</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Voting classification-based diabetes mellitus prediction using hypertuned machine-learning techniques</article-title>. <source>Mob. Inf. Syst.</source> <volume>2022</volume>, <fpage>1</fpage>&#x2013;<lpage>16</lpage>. <pub-id pub-id-type="doi">10.1155/2022/6521532</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pramanik</surname>
<given-names>P. K. D.</given-names>
</name>
<name>
<surname>Bandyopadhyay</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Choudhury</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Predicting relative topological stability of mobile users in a P2P mobile cloud</article-title>. <source>SN Appl. Sci.</source> <volume>2</volume> (<issue>11</issue>), <fpage>1827</fpage>. <pub-id pub-id-type="doi">10.1007/s42452-020-03584-3</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Santhanam</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Uzir</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Raman</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Banerjee</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Experimenting XGBoost algorithm for prediction and classification of different datasets</article-title>. <source>Int. J. Control Theory Appl.</source> <volume>9</volume> (<issue>40</issue>), <fpage>651</fpage>&#x2013;<lpage>662</lpage>.</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sevinc</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>An empowered AdaBoost algorithm implementation: a COVID-19 dataset study</article-title>. <source>Comput. Industrial Eng.</source> <volume>165</volume>, <fpage>107912</fpage>. <pub-id pub-id-type="doi">10.1016/j.cie.2021.107912</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Singh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dhillon</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Hossain</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Muhammad</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>eDiaPredict: an ensemble-based framework for diabetes prediction</article-title>. <source>ACM Trans. Multimedia Comput. Commun. Appl.</source> <volume>17</volume> (<issue>2</issue>), <fpage>1</fpage>&#x2013;<lpage>26</lpage>. <pub-id pub-id-type="doi">10.1145/3415155</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sneha</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Gangil</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Analysis of diabetes mellitus for early prediction using optimal features selection</article-title>. <source>J. Big Data</source> <volume>6</volume> (<issue>1</issue>), <fpage>13</fpage>. <pub-id pub-id-type="doi">10.1186/s40537-019-0175-6</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Webber</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>International diabetes federation</article-title>. <source>Diabetes Res. Clin. Pract.</source> <volume>102</volume> (<issue>2</issue>).</citation>
</ref>
</ref-list>
</back>
</article>