<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Public Health</journal-id>
<journal-title>Frontiers in Public Health</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Public Health</abbrev-journal-title>
<issn pub-type="epub">2296-2565</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpubh.2024.1341279</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Public Health</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Machine learning prediction of adolescent HIV testing services in Ethiopia</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Alie</surname>
<given-names>Melsew Setegn</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/874517/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Negesse</surname>
<given-names>Yilkal</given-names>
</name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1690922/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Public Health, School of Public Health, College of Medicine and Health Science, Mizan-Tepi University</institution>, <addr-line>Mizan-Aman</addr-line>, <country>Ethiopia</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Public Health, College of Medicine and Health Science, Debre-Markos University</institution>, <addr-line>Gojjam</addr-line>, <country>Ethiopia</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0002">
<p>Edited by: Diego Ripamonti, Papa Giovanni XXIII Hospital, Italy</p>
</fn>
<fn fn-type="edited-by" id="fn0003">
<p>Reviewed by: Laboni Akter, Khulna University of Engineering and Technology, Bangladesh</p>
<p>Archana Singh, Amity University, India</p>
<p>Haithem Taha Mohammed Ali, University of Zakho, Iraq</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Melsew Setegn Alie, <email>melsewsetegn2010@gmail.com</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>15</day>
<month>03</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>12</volume>
<elocation-id>1341279</elocation-id>
<history>
<date date-type="received">
<day>20</day>
<month>11</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>04</day>
<month>03</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2024 Alie and Negesse.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Alie and Negesse</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec id="sec1">
<title>Background</title>
<p>Despite endeavors to achieve the Joint United Nations Programme on HIV/AIDS 95-95-95 fast track targets established in 2014 for HIV prevention, progress has fallen short. Hence, it is imperative to identify factors that can serve as predictors of an adolescent&#x2019;s HIV status. This identification would enable the implementation of targeted screening interventions and the enhancement of healthcare services. Our primary objective was to identify these predictors to facilitate the improvement of HIV testing services for adolescents in Ethiopia.</p>
</sec>
<sec id="sec2">
<title>Methods</title>
<p>A study was conducted by utilizing eight different machine learning techniques to develop models using demographic and health data from 4,502 adolescent respondents. The dataset consisted of 31 variables and variable selection was done using different selection methods. To train and validate the models, the data was randomly split into 80% for training and validation, and 20% for testing. The algorithms were evaluated, and the one with the highest accuracy and mean f1 score was selected for further training using the most predictive variables.</p>
</sec>
<sec id="sec3">
<title>Results</title>
<p>The J48 decision tree algorithm has proven to be remarkably successful in accurately detecting HIV positivity, outperforming seven other algorithms with an impressive accuracy rate of 81.29% and a Receiver Operating Characteristic (ROC) curve of 86.3%. The algorithm owes its success to its remarkable capability to identify crucial predictor features, with the top five being age, knowledge of HIV testing locations, age at first sexual encounter, recent sexual activity, and exposure to family planning. Interestingly, the model&#x2019;s performance witnessed a significant improvement when utilizing only twenty variables as opposed to including all variables.</p>
</sec>
<sec id="sec4">
<title>Conclusion</title>
<p>Our research findings indicate that the J48 decision tree algorithm, when combined with demographic and health-related data, is a highly effective tool for identifying potential predictors of HIV testing. This approach allows us to accurately predict which adolescents are at a high risk of infection, enabling the implementation of targeted screening strategies for early detection and intervention. To improve the testing status of adolescents in the country, we recommend considering demographic factors such as age, age at first sexual encounter, exposure to family planning, recent sexual activity, and other identified predictors.</p>
</sec>
</abstract>
<kwd-group>
<kwd>HIV</kwd>
<kwd>machine learning</kwd>
<kwd>adolescent</kwd>
<kwd>Ethiopia</kwd>
<kwd>ML</kwd>
</kwd-group>
<counts>
<fig-count count="9"/>
<table-count count="8"/>
<equation-count count="0"/>
<ref-count count="62"/>
<page-count count="19"/>
<word-count count="11281"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Infectious Diseases: Epidemiology and Prevention</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec5">
<title>Introduction</title>
<p>In 2016, 2.1 million adolescents were infected with HIV, with 1.7 million in sub-Saharan Africa (<xref ref-type="bibr" rid="ref1">1</xref>). Effective antiretroviral treatment has decreased perinatal HIV infection, but challenges remain in treatment and care. Sub-Saharan Africa has the highest burden of HIV, with young women at higher risk (<xref ref-type="bibr" rid="ref2">2</xref>&#x2013;<xref ref-type="bibr" rid="ref4">4</xref>). HIV testing is crucial for diagnosis, treatment, and prevention, including the prevention of mother-to-child transmission. A well-functioning HIV testing service is essential for reducing HIV-related illnesses and deaths (<xref ref-type="bibr" rid="ref5">5</xref>&#x2013;<xref ref-type="bibr" rid="ref7">7</xref>).</p>
<p>HIV testing is a crucial public health program aimed at reducing the spread of HIV/AIDS and mitigating its impact on communities and national economies (<xref ref-type="bibr" rid="ref8">8</xref>, <xref ref-type="bibr" rid="ref9">9</xref>). It serves as a critical entry point for HIV detection, care, treatment, prevention, and support services (<xref ref-type="bibr" rid="ref10">10</xref>, <xref ref-type="bibr" rid="ref11">11</xref>). HIV testing protects individuals who have been exposed to an HIV-positive partner and their infants from infection (<xref ref-type="bibr" rid="ref12">12</xref>). The Sustainable Development Goals prioritize ending the HIV/AIDS epidemic by 2030, making prevention and control of the disease a critical agenda item (<xref ref-type="bibr" rid="ref13">13</xref>). Studies show that HIV testing is the most cost-effective measure for preventing and controlling HIV transmission in Africa (<xref ref-type="bibr" rid="ref14">14</xref>, <xref ref-type="bibr" rid="ref15">15</xref>). The Ethiopian government has embraced voluntary HIV counseling and testing as a key component of the country&#x2019;s HIV/AIDS prevention and control efforts (<xref ref-type="bibr" rid="ref16">16</xref>).</p>
<p>Regular HIV testing is crucial for meeting the Joint United Nations Programme on HIV/AIDS (UNAIDS) 95-95-95 targets by 2030, which aim to ensure that 95% of people living with HIV are diagnosed, 95% of those diagnosed are receiving antiretroviral therapy (ART), and 95% of those on ART are virally suppressed (<xref ref-type="bibr" rid="ref17">17</xref>). In 2016, the World Health Organization (WHO) expanded ART eligibility guidelines to ensure that all people living with HIV have access to treatment, with a focus on same-day ART initiation whenever feasible. This approach, known as &#x201C;test and treat,&#x201D; is an essential step towards achieving universal ART coverage and ending the HIV epidemic (<xref ref-type="bibr" rid="ref18">18</xref>).</p>
<p>Despite women being at a higher risk of HIV infection, access to HIV testing remains uneven, particularly in sub-Saharan Africa where the prevalence of HIV among adults is alarmingly high. The uptake of HIV testing and counseling among women, including adolescent girls, remains low, even though they account for two-thirds of new infections globally (<xref ref-type="bibr" rid="ref19">19</xref>&#x2013;<xref ref-type="bibr" rid="ref23">23</xref>). In Ethiopia, HIV voluntary counseling and testing (VCT) has been a key strategy in the country&#x2019;s efforts to prevent and control HIV/AIDS. However, the utilization of VCT services among males, females, adults, and rural residents in Ethiopia is still inadequate (<xref ref-type="bibr" rid="ref24">24</xref>).</p>
<p>Adolescents face barriers in accessing information and services related to HIV and reproductive health due to factors like age and socioeconomic status (<xref ref-type="bibr" rid="ref22">22</xref>, <xref ref-type="bibr" rid="ref25">25</xref>). Given the high HIV burden in many countries, adolescence presents an opportunity for early intervention. Comprehensive data is essential for shaping accurate HIV-related messages and services before risky behaviors become entrenched. Socioeconomic and structural challenges such as poverty, limited education, and gender inequality can increase the risk of HIV infection (<xref ref-type="bibr" rid="ref26">26</xref>, <xref ref-type="bibr" rid="ref27">27</xref>). While risk awareness is important, it is not enough to address the challenges of HIV/AIDs (<xref ref-type="bibr" rid="ref5">5</xref>). Having a positive attitude towards HIV knowledge is influenced by factors like education, social status, and gender, which can contribute to better sexual and reproductive health policies (<xref ref-type="bibr" rid="ref6">6</xref>). Previous studies conducted in various regions of the world have identified several factors that are significantly associated with HIV testing among men. These factors include marital status, age, educational level, region of residence, having multiple sexual partners, wealth index, condom use, exposure to mass media, and age at first sexual encounter (<xref ref-type="bibr" rid="ref25">25</xref>, <xref ref-type="bibr" rid="ref27">27</xref>&#x2013;<xref ref-type="bibr" rid="ref31">31</xref>). Additionally, stigma towards HIV patients, comprehensive HIV knowledge, and engaging in risky sexual behavior have been found to have positive associations with HIV testing and counseling (<xref ref-type="bibr" rid="ref6">6</xref>, <xref ref-type="bibr" rid="ref19">19</xref>, <xref ref-type="bibr" rid="ref22">22</xref>, <xref ref-type="bibr" rid="ref28">28</xref>, <xref ref-type="bibr" rid="ref31">31</xref>, <xref ref-type="bibr" rid="ref32">32</xref>).</p>
<p>The global coverage of HIV testing and counseling among adolescents is currently low (<xref ref-type="bibr" rid="ref16">16</xref>). In Africa, where the age of first sexual encounter is decreasing, many adolescents are also contracting other sexually transmitted infections (STIs) such as gonorrhea, chlamydia, and (<xref ref-type="bibr" rid="ref33">33</xref>) syphilis, which can potentially increase the risk of acquiring and transmitting HIV (<xref ref-type="bibr" rid="ref2">2</xref>). Additionally, the lack of knowledge among adolescents about STI symptoms and modes of transmission further compounds these health challenges (<xref ref-type="bibr" rid="ref17">17</xref>). While there have been a few studies conducted on these issues in Ethiopia (<xref ref-type="bibr" rid="ref7">7</xref>, <xref ref-type="bibr" rid="ref8">8</xref>), there is currently no study in the country that assesses HIV testing and counseling by using a machine learning algorithm. The use of algorithm-based prediction for HIV testing and identifying the most influential factors in adolescent HIV testing is a novel approach that has not been explored yet. Currently, there is a lack of research on machine learning prediction of adolescent HIV testing specifically in Ethiopia. Therefore, the main objective of this study is to utilize machine learning techniques to predict HIV testing and counseling among adolescents aged 15&#x2013;24&#x2009;years in Ethiopia.</p>
</sec>
<sec sec-type="methods" id="sec6">
<title>Methods</title>
<sec id="sec7">
<title>Patient selection</title>
<p>The research-utilized data from a publicly available demographic and health survey conducted in the country. Out of the 12,688 data points collected, 8,186 were excluded from the study as they pertained to individuals who were either younger than 15 or older than 24&#x2009;years old. Following the exclusion of these data points, the analysis concentrated on a group of 4,502 adolescents who were included in the study.</p>
</sec>
<sec id="sec8">
<title>Outcome variable</title>
<p>In this study, we referred to the outcome variable as &#x201C;tested,&#x201D; which indicated whether the adolescent had undergone HIV testing within the past 5&#x2009;years. If the adolescent had been tested during the survey period, we coded the variable as &#x201C;Yes.&#x201D; On the other hand, if the adolescent had not been tested for HIV within the past 5&#x2009;years, we coded the variable as &#x201C;No&#x201D;.</p>
</sec>
<sec id="sec9">
<title>Data preprocessing</title>
<p>In this specific study, we made the deliberate choice to include individuals aged 15&#x2013;24&#x2009;years old. We extracted from publically available data set of demographic and health survey of 2016.<xref ref-type="fn" rid="fn0001"><sup>1</sup></xref> Then we filtered the data of adolescent in this data set. Our data was sourced from a de-identified demographic and health survey database, encompassing information from a total of 4,502 adolescent individuals. To ensure the integrity of our data, we enlisted the expertise of two epidemiologists who collaborated on the project. Their valuable insights were instrumental in identifying and addressing any discrepancies such as noisy or abnormal values, errors, duplicates, and irrelevant data. Additionally, we meticulously reviewed the initial list of parameters to guarantee consistency in the data preprocessing stage. Ultimately, our analysis focused solely on the data pertaining to the 4,502 adolescents within the age range of 15&#x2013;24&#x2009;years old. To enhance clarity and coherence, we assigned appropriate labels to both nominal and continuous variables based on previous literature.</p>
</sec>
<sec id="sec10">
<title>Random forest</title>
<p>The random forest (RF) algorithm is known for significantly improving the classification accuracy of a model. This is achieved by generating multiple decision trees. Each decision tree produces a result for a given sample, and the final result is determined based on the majority of the decision trees&#x2019; results (<xref ref-type="bibr" rid="ref34">34</xref>). To maximize the performance of the RF algorithm, we conducted hyperparameter tuning and 10-fold cross-validation for all our experiments. We focused on tuning the same set of hyperparameters. These hyperparameters include the maximum depth values of the decision trees and the minimum number of samples required in a leaf node. In order to assess the quality of each node&#x2019;s split, we used two criteria: the Gini index and entropy. Since the RF algorithm generates multiple decision trees, we utilized the &#x201C;number of estimators&#x201D; hyperparameter to control the number of trees created in the forest. The values we experimented with for this hyperparameter were 100, 200, and 300. It&#x2019;s worth noting that each decision tree in the RF algorithm learns from random subsets of samples drawn from the dataset. The use of bootstrap sampling, which involves drawing samples with replacement, ensures the diversity and robustness of the decision trees. Additionally, we tuned the &#x201C;bootstrap&#x201D; hyperparameter, which determines whether the entire dataset is used to build a decision tree. When set to &#x201C;False,&#x201D; the bootstrap parameter ensures that the entire dataset is used. On the other hand, setting it to &#x201C;True&#x201D; allows bootstrap sampling to take place. By carefully tuning these hyperparameters, we aimed to achieve the highest possible performance for the RF algorithm in our experiments.</p>
</sec>
<sec id="sec11">
<title>Imbalanced data handing</title>
<p>In machine learning, dealing with imbalanced data can be a significant challenge. This occurs when the distribution of classes in a dataset is uneven, which can lead to biased results in favor of the majority class. In the current dataset being analyzed, there is a substantial imbalance between the &#x201C;not tested&#x201D; and &#x201C;tested&#x201D; classes, with 2,950 and 1,552 cases, respectively. This imbalance can lead to inaccurate results and make it likely for new observations to be categorized into the majority class. To address this issue, the study utilized a method called synthetic minority over-sampling technique (SMOTE) from the imbalanced-learn toolbox. SMOTE generates synthetic samples for the minority class by interpolating between existing minority class samples. By applying SMOTE, the dataset was balanced, allowing for more accurate and unbiased training of machine learning models. If you are interested in learning more about the imbalanced-learn toolbox and the SMOTE method, you can visit their website at <ext-link xlink:href="https://imbalanced-learn.org/stable/" ext-link-type="uri">https://imbalanced-learn.org/stable/</ext-link>.</p>
<p>Predictive accuracy is a commonly used metric to evaluate the performance of machine learning algorithms. However, when working with imbalanced datasets, accuracy can be misleading and hinder the identification of underlying causes such as HIV testing. In this particular study, researchers employed various techniques to address the class imbalance issue in their dataset. They utilized Synthetic Minority Oversampling Technique (SMOTE) (<xref ref-type="bibr" rid="ref35">35</xref>), which generates new samples by interpolating between existing samples and their neighbors (<xref ref-type="bibr" rid="ref36">36</xref>, <xref ref-type="bibr" rid="ref37">37</xref>). Additionally, they employed random under-sampling, which involves discarding samples from the majority class until the minority class reaches a predetermined percentage of the majority class (<xref ref-type="bibr" rid="ref35">35</xref>). Another method used was Adaptive Synthetic (ADASYN), which generates synthetic data for harder-to-learn minority class samples, thereby reducing bias introduced by imbalanced data distribution. Through the application of these techniques, the researchers successfully achieved a balanced dataset (<xref ref-type="bibr" rid="ref38">38</xref>). The outcome of their endeavors is discussed in detail in the result section of the study (<xref ref-type="bibr" rid="ref39">39</xref>).</p>
</sec>
<sec id="sec12">
<title>Feature selection</title>
<p>In the initial phase of our study, our primary aim was to identify key features that could accurately predict the HIV testing status of adolescents in Ethiopia. To achieve this, we conducted a thorough review of scientific literature by searching various databases. The findings from this review were then used to create a comprehensive questionnaire, which covered a wide range of predictors for HIV testing services. By incorporating these identified predictors, we aimed to develop a reliable tool for predicting HIV testing status. The SHAP analysis was done for the features. SHAP is a powerful technique that is independent of any specific machine learning model. It is used to calculate the Shapley values of the different features in a model, providing explanations for the model&#x2019;s predictions. SHAP feature importance is determined by taking the average of the absolute values of the Shapley values for each feature. This approach allows for a comprehensive understanding of the impact each feature has on the model&#x2019;s predictions.</p>
<p>During this stage, our main objective was to identify specific features that could accurately predict HIV testing status. We began by conducting a comprehensive review of scientific databases to determine the most relevant features. Based on this review, we developed a questionnaire that included predictors. We carefully compiled all the pertinent information to create the final version of our data collection tool. By doing so, we aimed to ensure that our study would yield accurate and reliable results that could help inform future efforts to improve HIV testing services in Ethiopia. WEKA version 3.9 was used to select important features, R version 4.0.2 and python version 3.2 was used for the analysis of data. <xref ref-type="fig" rid="fig1">Figure 1</xref> displays a flowchart that outlines the process used to select the final variables for a machine-learning model. This process involved five distinct steps. The first step involved removing features from the dataset that had a missing value greater than 30%. In the second step, features that were deemed irrelevant to the final outcome variable, such as reference date, patient ID, and accompanying information, were eliminated. The third step addressed collinearity, which can lead to duplicated features and skew the model&#x2019;s results. Features with a collinearity greater than 0.95 were removed from the dataset. By implementing these procedures, the most relevant and informative features for the machine learning model were identified, as shown in <xref ref-type="fig" rid="fig1">Figure 1</xref>. In this study, various feature selection methods were employed to identify the most relevant predictive features. These methods included recursive feature elimination (RFE), random forest feature importance, and the Boruta feature selection method. RFE is a technique used for feature selection, which begins with all the features in the training dataset and iteratively eliminates features until the desired number of features is reached. This method is particularly effective in reducing model complexity and improving the efficiency of machine learning algorithms. By utilizing these feature selection methods, the study aimed to identify the most informative features that contribute significantly to the predictive power of the model (<xref ref-type="bibr" rid="ref40">40</xref>). This approach helps to streamline the data processing and enhance the accuracy of machine learning algorithms.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>The flowchart of variable selection for machine-learning algorithm model.</p>
</caption>
<graphic xlink:href="fpubh-12-1341279-g001.tif"/>
</fig>
</sec>
<sec id="sec13">
<title>Model development</title>
<p>A comprehensive literature review was conducted to develop accurate predictive classifier models for HIV testing status (<xref ref-type="bibr" rid="ref19">19</xref>, <xref ref-type="bibr" rid="ref33">33</xref>, <xref ref-type="bibr" rid="ref41">41</xref>&#x2013;<xref ref-type="bibr" rid="ref45">45</xref>). The selection of suitable machine learning (ML) algorithms was based on the type and quality of the dataset utilized. Eight ML algorithms were employed to construct the individual testing prediction model: J48 decision tree, random forest (RF), k-nearest neighborhood (k-NN), Support Vector Machine (SVM), multi-layer perceptron (MLP), Na&#x00EF;ve Bayes (NB), logistic gradient boosting (logit Boost), and logistic regression (LR). The data were analyzed using Weka software v3.9.2 python, and R software to implement the algorithms, analyze and calculate curves and criteria, and draw the confusion matrix.</p>
</sec>
<sec id="sec14">
<title>Cross-validation</title>
<p>We used the EXPLORER module in WEKA to find the best hyperparameters for our models and evaluated their performance using tenfold cross-validation. We ran experiments with WEKA&#x2019;s EXPERIMENTER module and repeated the cross-validation process to ensure reliable results. We used an 80:20 ratio for training and testing, and calculated average performance metrics across ten runs. We choose stratified tenfold cross-validation to accurately estimate accuracy. Our approach aims to minimize errors and biases by using more data for training and validation. We used WEKA&#x2019;s EXPLORER and EXPERIMENTER modules along with tenfold cross-validation for a robust evaluation and comparison of classification models. We selected stratified tenfold cross-validation as it strikes a favorable balance between bias and variance, making it a preferred technique for accurately estimating accuracy. It is worth emphasizing that tenfold cross-validation is widely employed in the fields of machine learning and data mining due to its advantages over traditional instance splitting methods. This provides a reliable and robust method for evaluating and comparing the effectiveness of classification models. The summary of machine learning pipeline presented on <xref ref-type="fig" rid="fig2">Figure 2</xref>.</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>Workflow of machine learning for adolescent HIV testing prediction.</p>
</caption>
<graphic xlink:href="fpubh-12-1341279-g002.tif"/>
</fig>
</sec>
<sec id="sec15">
<title>Model evaluation</title>
<p>Evaluating the performance of a machine-learning model is essential for its success. In our study, we thoroughly assessed the performance of our predictive models using various performance metrics, as outlined in <xref ref-type="table" rid="tab1">Table 1</xref>. These metrics encompassed accuracy, specificity, precision, sensitivity, and the receiver operating characteristic (ROC) chart criteria. By leveraging these metrics, we effectively measured the effectiveness of our models in predicting HIV testing services of adolescents. To determine the best model for predicting HIV testing service of adolescent, we compared the performance of each model using the aforementioned evaluation criteria. The results of this comparison are summarized in <xref ref-type="table" rid="tab2">Table 2</xref>. Through a meticulous analysis and comparison of these evaluation criteria, we successfully identified the model that demonstrated the highest performance in predicting HIV testing. Our comprehensive evaluation process enabled us to select the most effective model and gain valuable insights and confidence in its predictive capabilities.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Confusion matrix.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Output</th>
<th align="left" valign="top">Predicted value</th>
<th/>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Actual value</td>
<td align="left" valign="top">Tested (+)</td>
<td align="left" valign="top">Not tested (&#x2212;)</td>
</tr>
<tr>
<td align="left" valign="top">Tested (+)</td>
<td align="left" valign="top">True positive (TP)</td>
<td align="left" valign="top">False negative (FN)</td>
</tr>
<tr>
<td align="left" valign="top">Not tested (&#x2212;)</td>
<td align="left" valign="top">False positive (FP)</td>
<td align="left" valign="top">True negative (TN)</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>The performance evaluation measures.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Performance criteria</th>
<th align="left" valign="top">Calculation</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Accuracy</td>
<td align="left" valign="top">(TP&#x2009;+&#x2009;TN)/(TP&#x2009;+&#x2009;TN&#x2009;+&#x2009;FP&#x2009;+&#x2009;FN)</td>
</tr>
<tr>
<td align="left" valign="top">Sensitivity/recall</td>
<td align="left" valign="top">TP/(TP&#x2009;+&#x2009;FP)</td>
</tr>
<tr>
<td align="left" valign="top">Precision</td>
<td align="left" valign="top">TP/(TP&#x2009;+&#x2009;FN)</td>
</tr>
<tr>
<td align="left" valign="top">Specificity</td>
<td align="left" valign="top">TN/(TN&#x2009;+&#x2009;FP)</td>
</tr>
<tr>
<td align="left" valign="top">F1 score</td>
<td align="left" valign="top">
<inline-formula>
<mml:math id="M1">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mn>2</mml:mn>
<mml:mo>&#x2217;</mml:mo>
</mml:msup>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:msup>
<mml:mi>n</mml:mi>
<mml:mo>&#x2217;</mml:mo>
</mml:msup>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="sec16">
<title>Association rule</title>
<p>The technique of association rule mining examines the correlations between multiple variables in a group, and it was first developed by Agarwal and Srikanth (<xref ref-type="bibr" rid="ref46">46</xref>). In this study, an additional method was utilized to support the classification of machine learning algorithms for predicting adolescent HIV testing using R software. The apriori algorithm was employed to uncover associations between the selected features and the target feature (<xref ref-type="bibr" rid="ref47">47</xref>). A minimal support degree of 0.00095 and a minimum confidence threshold of 90% were set to identify all potential association rules. This is because a rule is considered reliable if its confidence level is more than 80% (<xref ref-type="bibr" rid="ref48">48</xref>). In this study, the main focus was on identifying features that are associated with adolescent HIV testing through the use of association rules. Specifically, the study utilized a technique known as classification association rules (<xref ref-type="bibr" rid="ref49">49</xref>), which involves analyzing the features that are implied by the target features (Antecedent =&#x2009;&#x003E;&#x2009;Consequent). The goal of this approach was to classify all the variables that contribute to HIV testing among adolescents and to identify the predictors that each category contributes to the testing. To evaluate the strength of each rule, the study used the metrics of Support, Confidence, and Lift. It is important to note that in this context, the sets of features represented by X and Y are mutually exclusive.</p>
<p>Rule X =&#x2009;&#x003E;&#x2009;Y:</p>
<p>Support&#x2009;=&#x2009;<inline-formula>
<mml:math id="M2">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">Fequence</mml:mi>
<mml:mspace width="thickmathspace"/>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">X,</mml:mi>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mi mathvariant="normal">Y</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi mathvariant="normal">N</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula>, Confidence&#x2009;=&#x2009;<inline-formula>
<mml:math id="M3">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">Frequency</mml:mi>
<mml:mspace width="thickmathspace"/>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">X,Y</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">Fequency</mml:mi>
<mml:mspace width="thickmathspace"/>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi mathvariant="normal">X</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula>, Lift&#x2009;=&#x2009;<inline-formula>
<mml:math id="M4">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">Frequency</mml:mi>
<mml:mspace width="thickmathspace"/>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">X,Y</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">Fequency</mml:mi>
<mml:mspace width="thickmathspace"/>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi mathvariant="normal">X</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mspace width="thickmathspace"/>
<mml:mi mathvariant="normal">Frequency</mml:mi>
<mml:mspace width="thickmathspace"/>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi mathvariant="normal">Y</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula></p>
</sec>
</sec>
<sec sec-type="results" id="sec17">
<title>Results</title>
<sec id="sec18">
<title>Patient characteristics and descriptive statistics</title>
<p>After applying our exclusion criteria and conducting a quantitative analysis of case records, we have identified a total of 4,502 adolescents in the database who met the eligibility criteria for our study. Among these participants, 3,339 (74.2%) adolescents were male, while 1,163 (25.8%) adolescents were female. The mean age of the study participants was 19.09 (&#x00B1;2.841) years old. A majority of the study participants, 3,137 (69.7%), reported residing in rural areas. Additionally, 55.2% of the total 4,502 study participants had attended primary education. An overwhelming majority, 4,292 (95.3%) of the adolescents, had previously heard about AIDS. Out of the total study participants, 79.4% of them knew the place for HIV testing. Regarding awareness of sexually transmitted infections (STIs), the majority (95.7%) of the adolescents were aware of STIs (<xref ref-type="table" rid="tab3">Table 3</xref>).</p>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption>
<p>Descriptive statistics of the current study conducted in Ethiopia.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Feature</th>
<th align="left" valign="top">Value</th>
<th align="center" valign="top">Frequencies</th>
<th align="left" valign="top">Feature</th>
<th align="left" valign="top">Value</th>
<th align="center" valign="top">Frequencies</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top" rowspan="2">Residence</td>
<td align="left" valign="top">Rural</td>
<td align="center" valign="top">3,137 (69.7%)</td>
<td align="left" valign="top" rowspan="2">Age</td>
<td align="left" valign="top" rowspan="2">Mean&#x2009;&#x00B1;&#x2009;SD</td>
<td align="center" valign="top" rowspan="2">19.09&#x2009;&#x00B1;&#x2009;2.841</td>
</tr>
<tr>
<td align="left" valign="top">Urban</td>
<td align="center" valign="top">1,365 (30.3%)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="5">Educational level</td>
<td align="left" valign="top">No education</td>
<td align="center" valign="top">513 (11.4%)</td>
<td align="left" valign="top" rowspan="5">Total children ever born</td>
<td align="left" valign="top" rowspan="2">Zero</td>
<td align="center" valign="top" rowspan="2">4,207 (93.4%)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">Primary</td>
<td align="center" valign="top" rowspan="2">2,484 (55.2%)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="3">&#x003E;&#x2009;=&#x2009;1 child</td>
<td align="center" valign="top" rowspan="3">295 (6.6%)</td>
</tr>
<tr>
<td align="left" valign="top">Secondary</td>
<td align="center" valign="top">1,102 (24.5%)</td>
</tr>
<tr>
<td align="left" valign="top">Higher</td>
<td align="center" valign="top">403 (9.0%)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="9">Region</td>
<td align="left" valign="top">Tigray</td>
<td align="center" valign="top">504 (11.2%)</td>
<td align="left" valign="top" rowspan="9">Number of household member</td>
<td align="left" valign="top" rowspan="2">1&#x2013;3 members</td>
<td align="center" valign="top" rowspan="2">979 (21.7%)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">Afar</td>
<td align="center" valign="top" rowspan="2">252 (5.6%)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">4&#x2013;6 members</td>
<td align="center" valign="top" rowspan="2">1876 (41.7%)</td>
</tr>
<tr>
<td align="left" valign="top">Amhara</td>
<td align="center" valign="top">588 (13.1%)</td>
</tr>
<tr>
<td align="left" valign="top">Oromia</td>
<td align="center" valign="top">603 (13.4%)</td>
<td align="left" valign="top" rowspan="5">&#x2265;7 members</td>
<td align="center" valign="top" rowspan="5">1,647 (36.6%)</td>
</tr>
<tr>
<td align="left" valign="top">Somali</td>
<td align="center" valign="top">378 (8.4%)</td>
</tr>
<tr>
<td align="left" valign="top">Benishangul</td>
<td align="center" valign="top">327 (7.3%)</td>
</tr>
<tr>
<td align="left" valign="top">SNNPR</td>
<td align="center" valign="top">569 (12.6%)</td>
</tr>
<tr>
<td align="left" valign="top">Gambella</td>
<td align="center" valign="top">338 (7.5%)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">Sex of household head</td>
<td align="left" valign="top">Male</td>
<td align="center" valign="top">3,339 (74.2%)</td>
<td align="left" valign="top" rowspan="2">Number of living children</td>
<td align="left" valign="top">Have no child</td>
<td align="center" valign="top">4,219 (93.7%)</td>
</tr>
<tr>
<td align="left" valign="top">Female</td>
<td align="center" valign="top">1,163 (25.8%)</td>
<td align="left" valign="top">One and above</td>
<td align="center" valign="top">283 (6.3%)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="7">Literacy</td>
<td align="left" valign="top">Cannot read at all</td>
<td align="center" valign="top">839 (18.9%)</td>
<td align="left" valign="top" rowspan="7">Recent sexual activity</td>
<td align="left" valign="top" rowspan="2">Never had sex</td>
<td align="center" valign="top" rowspan="2">3,166 (70.3%)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">Able to read only part of sentences</td>
<td align="center" valign="top" rowspan="2">729 (16.2%)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">Active in the last 4&#x2009;weeks</td>
<td align="center" valign="top" rowspan="2">622 (13.8%)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">Able to read whole part of sentences</td>
<td align="center" valign="top" rowspan="2">2,878 (63.9%)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="3">Not active in the last 4&#x2009;weeks</td>
<td align="center" valign="top" rowspan="3">714 (15.9%)</td>
</tr>
<tr>
<td align="left" valign="top">No card on required language</td>
<td align="center" valign="top">53 (1.2%)</td>
</tr>
<tr>
<td align="left" valign="top">Blind/visual impaired</td>
<td align="center" valign="top">3 (0.1%)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="5">Wealth index</td>
<td align="left" valign="top">Poorest</td>
<td align="center" valign="top">993 (22.1%)</td>
<td align="left" valign="top" rowspan="5">Currently working</td>
<td align="left" valign="top" rowspan="2">Yes</td>
<td align="center" valign="top" rowspan="2">1,519 (33.7%)</td>
</tr>
<tr>
<td align="left" valign="top">Poorer</td>
<td align="center" valign="top">608 (13.5%)</td>
</tr>
<tr>
<td align="left" valign="top">Middle</td>
<td align="center" valign="top">625 (13.9%)</td>
<td align="left" valign="top" rowspan="3">No</td>
<td align="center" valign="top" rowspan="3">2,983 (66.3%)</td>
</tr>
<tr>
<td align="left" valign="top">Richer</td>
<td align="center" valign="top">759 (16.9%)</td>
</tr>
<tr>
<td align="left" valign="top">Richest</td>
<td align="center" valign="top">1,517 (33.7%)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">Knowledge of ovulatory cycle</td>
<td align="left" valign="top">Did not have knowledge</td>
<td align="center" valign="top">1,097 (24.4%)</td>
<td align="left" valign="top" rowspan="2">Awareness on STI</td>
<td align="left" valign="top">Yes</td>
<td align="center" valign="top">4,307 (95.7%)</td>
</tr>
<tr>
<td align="left" valign="top">Have knowledge</td>
<td align="center" valign="top">3,405 (75.6%)</td>
<td align="left" valign="top">No</td>
<td align="center" valign="top">195 (4.3)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="3">Knowledge of any contraceptive method</td>
<td align="left" valign="top">Know no method</td>
<td align="center" valign="top">209 (4.6%)</td>
<td align="left" valign="top" rowspan="3">Worked in last 12&#x2009;months</td>
<td align="left" valign="top">No</td>
<td align="center" valign="top">1,257 (27.9%)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">Know any methods</td>
<td align="center" valign="top" rowspan="2">4,293 (95.4%)</td>
<td align="left" valign="top">In past year</td>
<td align="center" valign="top">262 (5.8%)</td>
</tr>
<tr>
<td align="left" valign="top">Currently working</td>
<td align="center" valign="top">2,983 (66.3%)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="3">Respondent circumcised</td>
<td align="left" valign="top">No</td>
<td align="center" valign="top">455 (10.1%)</td>
<td align="left" valign="top" rowspan="3">Ever heard of AIDS</td>
<td align="left" valign="top" rowspan="2">Yes</td>
<td align="center" valign="top" rowspan="2">4,292 (95.3%)</td>
</tr>
<tr>
<td align="left" valign="top">Yes</td>
<td align="center" valign="top">4,033 (89.9%)</td>
</tr>
<tr>
<td align="left" valign="top">I do not know</td>
<td align="center" valign="top">14 (0.3%)</td>
<td align="left" valign="top">No</td>
<td align="center" valign="top">210 (4.7%)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">Current marital status</td>
<td align="left" valign="top">Never in union</td>
<td align="center" valign="top">4,014 (89.2%)</td>
<td align="left" valign="top" rowspan="2">Alcohol drinking</td>
<td align="left" valign="top">Yes</td>
<td align="center" valign="top">1764 (39.3%)</td>
</tr>
<tr>
<td align="left" valign="top">Married</td>
<td align="center" valign="top">488 (10.8%)</td>
<td align="left" valign="top">No</td>
<td align="center" valign="top">2,722 (60.7%)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="3">Age at first sex</td>
<td align="left" valign="top">Not had sex</td>
<td align="center" valign="top">3,166 (70.3%)</td>
<td align="left" valign="top" rowspan="3">Intimate partner violence</td>
<td align="left" valign="top">Yes</td>
<td align="center" valign="top">1,472 (32.7%)</td>
</tr>
<tr>
<td align="left" valign="top">Between 8 and 15&#x2009;years</td>
<td align="center" valign="top">248 (5.5%)</td>
<td align="left" valign="top" rowspan="2">No</td>
<td align="center" valign="top" rowspan="2">3,030 (67.3%)</td>
</tr>
<tr>
<td align="left" valign="top">Between 16 and 24&#x2009;years</td>
<td align="center" valign="top">1,088 (24.2%)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">Ever been tested for HIV</td>
<td align="left" valign="top">Yes</td>
<td align="center" valign="top">1,552 (34.5%)</td>
<td align="left" valign="top" rowspan="2">Resource control</td>
<td align="left" valign="top">Yes</td>
<td align="center" valign="top">901 (20.0%)</td>
</tr>
<tr>
<td align="left" valign="top">No</td>
<td align="center" valign="top">2,950 (65.5%)</td>
<td align="left" valign="top">No</td>
<td align="center" valign="top">3,601 (80.0%)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">Know a place to get HIV test</td>
<td align="left" valign="top">Yes</td>
<td align="center" valign="top">3,576 (79.4%)</td>
<td align="left" valign="top" rowspan="2">Media exposure</td>
<td align="left" valign="top">Yes</td>
<td align="center" valign="top">3,108 (69.0%)</td>
</tr>
<tr>
<td align="left" valign="top">No</td>
<td align="center" valign="top">926 (20.6%)</td>
<td align="left" valign="top">No</td>
<td align="center" valign="top">1,394 (31.0%)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">Have you ever chewed chat?</td>
<td align="left" valign="top">Yes</td>
<td align="center" valign="top">3,648 (81.0%)</td>
<td align="left" valign="top" rowspan="2">Fear of Stigma and discrimination</td>
<td align="left" valign="top">Yes</td>
<td align="center" valign="top">4,046 (94.3%)</td>
</tr>
<tr>
<td align="left" valign="top">No</td>
<td align="center" valign="top">854 (19.0%)</td>
<td align="left" valign="top">No</td>
<td align="center" valign="top">246 (5.7%)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="3">Exposure to family planning</td>
<td align="left" valign="top" rowspan="2">Yes</td>
<td align="center" valign="top" rowspan="2">2,768 (61.5%)</td>
<td align="left" valign="top" rowspan="3">Number of Sexual partner</td>
<td align="left" valign="top">Have no sexual partner</td>
<td align="center" valign="top">2,924 (87.2%)</td>
</tr>
<tr>
<td align="left" valign="top">Only one</td>
<td align="center" valign="top">487 (10.8%)</td>
</tr>
<tr>
<td align="left" valign="top">No</td>
<td align="center" valign="top">1734 (38.5%)</td>
<td align="left" valign="top">Two and above</td>
<td align="center" valign="top">91 (2.0%)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">Smoking status</td>
<td align="left" valign="top">Yes</td>
<td align="center" valign="top">71 (1.6%)</td>
<td align="left" valign="top" rowspan="2">Health insurance</td>
<td align="left" valign="top">Yes</td>
<td align="center" valign="top">4,295 (95.4%)</td>
</tr>
<tr>
<td align="left" valign="top">No</td>
<td align="center" valign="top">4,431 (98.4%)</td>
<td align="left" valign="top">No</td>
<td align="center" valign="top">207 (4.6%)</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="sec19">
<title>Data pre-processing results</title>
<p>In this particular study, various pre-processing steps were undertaken to handle missing or null values, encode categorical labels, and balance the dataset. Initially, the incomplete raw data and missing values were addressed using the imputation technique. <xref rid="SM1" ref-type="supplementary-material">Supplementary Figure S1</xref> provides a breakdown of the percentage of missing values for each feature, which shows that the feature know the place of HIV testing 210 (4.66%), alcohol drinking 16 (0.3554%), and stigma 210 (4.66%) having the highest and the only percentage missing data in adolescent HIV testing data set. These missing values were imputed using the simple imputation technique. To prepare the data for the model, it was necessary to convert input and output features into numerical values. This was accomplished by utilizing the one-hot encoder to encode categorical variables found in the dataset. The dataset consisted of thirty one features with a single target feature. One-hot encoding is a valuable encoding technique for classification tasks. This technique transformed each categorical value into a new column with one-hot encoding, and the label values were created as new columns with values of either 1 or 0.</p>
</sec>
<sec id="sec20">
<title>Imbalance data handling</title>
<p>The current study aimed to address the issue of data imbalance and enhance the effectiveness of machine learning algorithms. Various techniques were employed to balance the dataset. The outcome feature revealed that the majority of observations (65.5%) were classified as not tested for HIV, while a smaller portion (34.5%) were tested. To balance the dataset, under-resampling, SMOTE, and ADASYN techniques were utilized, each employing a different approach to either maximize the minority class or decrease the majority class. The performance of each balancing technique was compared using selected supervised classification machine learning algorithms, with a focus on accuracy and AUC. In the unbalanced dataset, the random forest algorithm achieved a higher AUC of 84.9% compared to other classifiers, while the J48 decision tree had a higher accuracy of 82.4%. In the ADASYN approach, the J48 decision tree classifier achieved a higher accuracy of 83.1%, while logistic regression achieved a higher AUC of 79.9%. When comparing different balanced sampling methods using a J48 decision tree classifier, the SMOTE sampling technique performed the best, with an accuracy of 89.3% and an AUC of 86.3%. ADASYN was the second most effective method, with an accuracy and AUC of 83.1 and 76.9%, respectively. However, the under-sampling technique was found to be the least effective, with an AUC of 73.4% and an accuracy of 78.2% (<xref ref-type="table" rid="tab4">Table 4</xref>).</p>
<table-wrap position="float" id="tab4">
<label>Table 4</label>
<caption>
<p>Compares imbalanced data handling techniques using accuracy and area under the curve.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Algorithms</th>
<th align="left" valign="top">Comparison method</th>
<th align="center" valign="top">Unbalanced</th>
<th align="center" valign="top">SMOTE</th>
<th align="center" valign="top">Under-sampling</th>
<th align="center" valign="top">ADASYN</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top" rowspan="2">Random forest</td>
<td align="left" valign="top">Accuracy</td>
<td align="center" valign="top">69.7</td>
<td align="center" valign="top">72.3</td>
<td align="center" valign="top">81.0</td>
<td align="center" valign="top">75.4</td>
</tr>
<tr>
<td align="left" valign="top">AUC</td>
<td align="center" valign="top">71.9</td>
<td align="center" valign="top">77.9</td>
<td align="center" valign="top">69.6</td>
<td align="center" valign="top">70.6</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">Logit Boost</td>
<td align="left" valign="top">Accuracy</td>
<td align="center" valign="top">78.2</td>
<td align="center" valign="top">74.8</td>
<td align="center" valign="top">82.6</td>
<td align="center" valign="top">75.8</td>
</tr>
<tr>
<td align="left" valign="top">AUC</td>
<td align="center" valign="top">74.0</td>
<td align="center" valign="top">81.0</td>
<td align="center" valign="top">69.8</td>
<td align="center" valign="top">72.7</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">J48 decision tree</td>
<td align="left" valign="top">Accuracy</td>
<td align="center" valign="top">82.4</td>
<td align="center" valign="top">89.3</td>
<td align="center" valign="top">78.2</td>
<td align="center" valign="top">83.1</td>
</tr>
<tr>
<td align="left" valign="top">AUC</td>
<td align="center" valign="top">79.6</td>
<td align="center" valign="top">86.3</td>
<td align="center" valign="top">73.4</td>
<td align="center" valign="top">76.9</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">Na&#x00EF;ve Bayes</td>
<td align="left" valign="top">Accuracy</td>
<td align="center" valign="top">65.7</td>
<td align="center" valign="top">73.9</td>
<td align="center" valign="top">69.7</td>
<td align="center" valign="top">82.5</td>
</tr>
<tr>
<td align="left" valign="top">AUC</td>
<td align="center" valign="top">69.2</td>
<td align="center" valign="top">81.3</td>
<td align="center" valign="top">75.3</td>
<td align="center" valign="top">71.0</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">SVM</td>
<td align="left" valign="top">Accuracy</td>
<td align="center" valign="top">67.0</td>
<td align="center" valign="top">72.7</td>
<td align="center" valign="top">76.4</td>
<td align="center" valign="top">65.4</td>
</tr>
<tr>
<td align="left" valign="top">AUC</td>
<td align="center" valign="top">70.8</td>
<td align="center" valign="top">66.8</td>
<td align="center" valign="top">73.5</td>
<td align="center" valign="top">62.5</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">KNN</td>
<td align="left" valign="top">Accuracy</td>
<td align="center" valign="top">70.5</td>
<td align="center" valign="top">68.7</td>
<td align="center" valign="top">69.0</td>
<td align="center" valign="top">70.3</td>
</tr>
<tr>
<td align="left" valign="top">AUC</td>
<td align="center" valign="top">69.8</td>
<td align="center" valign="top">67.8</td>
<td align="center" valign="top">70.3</td>
<td align="center" valign="top">67.3</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">LR</td>
<td align="left" valign="top">Accuracy</td>
<td align="center" valign="top">75.0</td>
<td align="center" valign="top">75.2</td>
<td align="center" valign="top">78.8</td>
<td align="center" valign="top">69.9</td>
</tr>
<tr>
<td align="left" valign="top">AUC</td>
<td align="center" valign="top">702</td>
<td align="center" valign="top">82.1</td>
<td align="center" valign="top">65.3</td>
<td align="center" valign="top">73.0</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">MLP</td>
<td align="left" valign="top">Accuracy</td>
<td align="center" valign="top">63.7</td>
<td align="center" valign="top">71.4</td>
<td align="center" valign="top">75.6</td>
<td align="center" valign="top">70.8</td>
</tr>
<tr>
<td align="left" valign="top">AUC</td>
<td align="center" valign="top">81.0</td>
<td align="center" valign="top">76.4</td>
<td align="center" valign="top">73.4</td>
<td align="center" valign="top">67.5</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Overall, the study demonstrated that employing balanced sampling methods can significantly improve the performance of machine learning algorithms on imbalanced datasets. The results suggest that SMOTE and ADASYN are effective techniques for balancing imbalanced data, while under-sampling may not be the most effective approach. The study in question faced a significant issue of imbalanced data, which could potentially hinder the performance of the classifying algorithm. To tackle this problem, a balanced sampling method was deemed crucial. Unbalanced data poses a challenge to machine learning, as values from the minority class or rarely occurring classes may be mistakenly classified as instances of the majority class. To address this issue, the study applied SMOTE to the unbalanced dataset, resulting in an increase in the total number of records (<xref ref-type="fig" rid="fig3">Figure 3</xref>). The classifier and the balanced sampling method were compared with other classifiers using training accuracy and AUC. Therefore, the study primarily relied on AUC to compare the classifier and the balanced sampling method. Overall, the use of a balanced sampling method proved to be effective in overcoming the issue of imbalanced data, and the study&#x2019;s findings highlight the importance of considering AUC when evaluating the performance of classification algorithms on imbalanced datasets.</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>Before unbalanced and after balancing data of the target features.</p>
</caption>
<graphic xlink:href="fpubh-12-1341279-g003.tif"/>
</fig>
</sec>
<sec id="sec21">
<title>J48 decision tree model performance</title>
<p>In this experiment, the objective was to evaluate the effectiveness of different classifiers in predicting the HIV testing status of adolescents. The primary goal of the analysis was to assess the accuracy of the predictions made by the chosen classifier. Among the classifiers that were selected, the random forest classifier exhibited particularly robust performance on the balanced dataset compared to the others. Once the best model was identified, hyper-parameter tuning and feature selection were performed. Additionally, the significance of the predictor for viral failure was determined to further evaluate the model&#x2019;s performance.</p>
</sec>
<sec id="sec22">
<title>J48 decision tree with selected features</title>
<p>This study aimed to assess the effectiveness of the J48 decision tree classifier in predicting HIV testing among adolescents and compare it with two other feature selection methods: Boruta feature selection and recursive feature elimination. The findings, as depicted in <xref ref-type="fig" rid="fig4">Figure 4</xref>, revealed that the J48 decision tree feature importance method outperformed both Boruta feature selection and recursive feature elimination. It achieved impressive results, including a sensitivity of 81.3%, specificity of 80.9%, precision of 81.0%, f1-score of 81.14%, and an AUC of 0.863. These outcomes strongly suggest that the J48 decision tree feature importance method is the most suitable approach for accurately predicting adolescent HIV testing. Conversely, Boruta feature selection and recursive feature elimination were found to be less effective in this context. Consequently, the J48 decision tree classifier was selected as the feature selection method for this study. Among all the features examined, the most significant predictors of HIV testing were age, knowledge of the place for HIV testing, age at first sexual encounter, recent sexual activity, exposure to family planning, and the number of sexual partners (see <xref ref-type="table" rid="tab5">Table 5</xref>).</p>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption>
<p>Comparison features selection method in adolescent HIV testing in Ethiopia.</p>
</caption>
<graphic xlink:href="fpubh-12-1341279-g004.tif"/>
</fig>
<table-wrap position="float" id="tab5">
<label>Table 5</label>
<caption>
<p>Features degree of importance in predicting HIV testing status of the adolescents in Ethiopia.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">S/N</th>
<th align="left" valign="top">Features name</th>
<th align="center" valign="top">Feature value</th>
<th align="center" valign="top">S/N</th>
<th align="left" valign="top">Features name</th>
<th align="center" valign="top">Feature value</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">1</td>
<td align="left" valign="top">Age</td>
<td align="center" valign="top">0.31733</td>
<td align="center" valign="top">11</td>
<td align="left" valign="top">Awareness on STI</td>
<td align="center" valign="top">0.15434</td>
</tr>
<tr>
<td align="left" valign="top">2</td>
<td align="left" valign="top">Know place of HIV testing</td>
<td align="center" valign="top">0.31543</td>
<td align="center" valign="top">12</td>
<td align="left" valign="top">Knowledge on contraceptive</td>
<td align="center" valign="top">0.14893</td>
</tr>
<tr>
<td align="left" valign="top">3</td>
<td align="left" valign="top">Age at first sex</td>
<td align="center" valign="top">0.28301</td>
<td align="center" valign="top">13</td>
<td align="left" valign="top">Knowledge on ovulatory cycle</td>
<td align="center" valign="top">0.14607</td>
</tr>
<tr>
<td align="left" valign="top">4</td>
<td align="left" valign="top">Recent sexual activity</td>
<td align="center" valign="top">0.26688</td>
<td align="center" valign="top">14</td>
<td align="left" valign="top">Literacy</td>
<td align="center" valign="top">0.13507</td>
</tr>
<tr>
<td align="left" valign="top">5</td>
<td align="left" valign="top">Exposure to family planning</td>
<td align="center" valign="top">0.24512</td>
<td align="center" valign="top">15</td>
<td align="left" valign="top">Wealth index</td>
<td align="center" valign="top">0.1285</td>
</tr>
<tr>
<td align="left" valign="top">6</td>
<td align="left" valign="top">Number of sexual partner</td>
<td align="center" valign="top">0.2015</td>
<td align="center" valign="top">16</td>
<td align="left" valign="top">Marital status</td>
<td align="center" valign="top">0.12292</td>
</tr>
<tr>
<td align="left" valign="top">7</td>
<td align="left" valign="top">Media exposure</td>
<td align="center" valign="top">0.1997</td>
<td align="center" valign="top">17</td>
<td align="left" valign="top">Alcohol drinking</td>
<td align="center" valign="top">0.11765</td>
</tr>
<tr>
<td align="left" valign="top">8</td>
<td align="left" valign="top">Residence</td>
<td align="center" valign="top">0.1916</td>
<td align="center" valign="top">18</td>
<td align="left" valign="top">Intimate partner violence</td>
<td align="center" valign="top">0.11203</td>
</tr>
<tr>
<td align="left" valign="top">9</td>
<td align="left" valign="top">Educational level</td>
<td align="center" valign="top">0.18108</td>
<td align="center" valign="top">19</td>
<td align="left" valign="top">Household members</td>
<td align="center" valign="top">0.08716</td>
</tr>
<tr>
<td align="left" valign="top">10</td>
<td align="left" valign="top">Awareness on AIDS</td>
<td align="center" valign="top">0.16044</td>
<td align="center" valign="top">20</td>
<td align="left" valign="top">Number of children</td>
<td align="center" valign="top">0.08367</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="sec23">
<title>J48 decision tree with hyperparameter tuning</title>
<p>After carefully selecting the most suitable classifier, this study proceeded to apply hyperparameter tuning or optimization to compare it with the default hyperparameter settings. <xref ref-type="fig" rid="fig5">Figure 5</xref> visually represents the performance of the default hyperparameter configuration against the hyperparameter tuning approach when using the random forest classifier. The results indicated that the default hyperparameter configuration performed better than the tuned hyperparameters. Based on these findings, it was concluded that the random forest classifier with default hyperparameters outperformed the one with tuned hyperparameters. Consequently, the study decided to utilize the random forest classifier with default hyperparameter settings. The J48 decision tree classifier, after undergoing hyperparameter tuning, achieved impressive metrics, including a precision of 88.6%, an f1-score of 91.1%, a sensitivity of 89.3%, and a specificity of 80.9%. It is worth noting that the J48 decision tree classifier with hyperparameter tuning outperformed all other classifiers in terms of the area under the curve (AUC). In summary, the study ultimately presented the default hyperparameter tuning and hyperparameter tuned shown in <xref ref-type="fig" rid="fig5">Figure 5</xref>.</p>
<fig position="float" id="fig5">
<label>Figure 5</label>
<caption>
<p>Comparison of tuned and default hyper parameter classifier.</p>
</caption>
<graphic xlink:href="fpubh-12-1341279-g005.tif"/>
</fig>
</sec>
<sec id="sec24">
<title>Feature selection methods</title>
<p>In our analysis of the dataset, we utilized the Pearson Correlation Coefficient (PCC) as another feature selection method. PCC is a statistical measure that helps determine the correlation between two random attributes. The correlation value, represented by &#x201C;r,&#x201D; ranges from &#x2212;1 to +1. By applying PCC to measure the correlation between the attributes and the target variable, which in this case is &#x201C;adolescent HIV testing,&#x201D; we were able to identify which features were positively or negatively correlated with the target class. The results of this analysis are presented in <xref ref-type="table" rid="tab6">Table 6</xref>, which lists the correlation values of each feature with respect to the target variable. The feature with the highest correlation value is &#x201C;know the place of HIV testing,&#x201D; with a value of 0.337. Additionally, &#x201C;age&#x201D; exhibits a correlation of 0.317, while &#x201C;age at first sex&#x201D; has a correlation of 0.306 (<xref ref-type="table" rid="tab6">Table 6</xref>). A positive correlation indicates that the variables are positively associated, meaning that as the value of x increases, the value of y also increases, and vice versa.</p>
<table-wrap position="float" id="tab6">
<label>Table 6</label>
<caption>
<p>The correlation of predictors to adolescent HIV testing attributes.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Feature</th>
<th align="center" valign="top">Correlation value</th>
<th align="left" valign="top">Feature</th>
<th align="center" valign="top">Correlation value</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Residence</td>
<td align="center" valign="top">&#x2212;0.192</td>
<td align="left" valign="top">Age</td>
<td align="center" valign="top">0.317</td>
</tr>
<tr>
<td align="left" valign="top">Educational level</td>
<td align="center" valign="top">0.294</td>
<td align="left" valign="top">Total children ever born</td>
<td align="center" valign="top">0.084</td>
</tr>
<tr>
<td align="left" valign="top">Region</td>
<td align="center" valign="top">0.130</td>
<td align="left" valign="top">Number of household member</td>
<td align="center" valign="top">0.165</td>
</tr>
<tr>
<td align="left" valign="top">Sex of household head</td>
<td align="center" valign="top">&#x2212;0.008</td>
<td align="left" valign="top">Number of living children</td>
<td align="center" valign="top">0.080</td>
</tr>
<tr>
<td align="left" valign="top">Literacy</td>
<td align="center" valign="top">0.171</td>
<td align="left" valign="top">Recent sexual activity</td>
<td align="center" valign="top">0.271</td>
</tr>
<tr>
<td align="left" valign="top">Wealth index</td>
<td align="center" valign="top">0.236</td>
<td align="left" valign="top">Currently working</td>
<td align="center" valign="top">0.059</td>
</tr>
<tr>
<td align="left" valign="top">Knowledge of ovulatory cycle</td>
<td align="center" valign="top">0.146</td>
<td align="left" valign="top">Awareness on STI</td>
<td align="center" valign="top">0.154</td>
</tr>
<tr>
<td align="left" valign="top">Knowledge of any contraceptive method</td>
<td align="center" valign="top">0.149</td>
<td align="left" valign="top">Worked in last 12&#x2009;months</td>
<td align="center" valign="top">0.062</td>
</tr>
<tr>
<td align="left" valign="top">Respondent circumcised</td>
<td align="center" valign="top">0.013</td>
<td align="left" valign="top">Ever heard of AIDS</td>
<td align="center" valign="top">0.160</td>
</tr>
<tr>
<td align="left" valign="top">Current marital status</td>
<td align="center" valign="top">0.123</td>
<td align="left" valign="top">Alcohol drinking</td>
<td align="center" valign="top">0.117</td>
</tr>
<tr>
<td align="left" valign="top">Age at first sex</td>
<td align="center" valign="top">0.306</td>
<td align="left" valign="top">Intimate partner violence</td>
<td align="center" valign="top">&#x2212;0.112</td>
</tr>
<tr>
<td align="left" valign="top">Know a place to get HIV test</td>
<td align="center" valign="top">0.337</td>
<td align="left" valign="top">Media exposure</td>
<td align="center" valign="top">0.200</td>
</tr>
<tr>
<td align="left" valign="top">Have you ever chewed Chat?</td>
<td align="center" valign="top">0.057</td>
<td align="left" valign="top">Stigma and discrimination</td>
<td align="center" valign="top">0.050</td>
</tr>
<tr>
<td align="left" valign="top">Exposure to family planning</td>
<td align="center" valign="top">0.245</td>
<td align="left" valign="top">Number of Sexual partner</td>
<td align="center" valign="top">0.198</td>
</tr>
<tr>
<td align="left" valign="top">Smoking status</td>
<td align="center" valign="top">0.032</td>
<td align="left" valign="top">Health insurance</td>
<td align="center" valign="top">0.026</td>
</tr>
<tr>
<td align="left" valign="top">Resource control</td>
<td align="center" valign="top">0.071</td>
<td/>
<td/>
</tr>
</tbody>
</table>
</table-wrap>
<p>After studying the correlation of the predictors to the target, we also took into consideration the collinearity. Collinearity happens when two predictors are linearly associated or having a high correlation to each other, and both were used as predictors of the target variable (<xref ref-type="bibr" rid="ref21">21</xref>). Multicollinearity may also happen, which is a situation wherein the variable has collinearity with more than one predictors in the dataset. We used the Variance Inflation Factor (VIF) to detect the collinearity of the predictors in the dataset. The VIF starts from 1 to infinity, and the value of 1 means that the features were not correlated. VIF values less than 5 are moderately correlated, while VIF values of 10 and above are highly correlated and a cause of concern (<xref ref-type="bibr" rid="ref21">21</xref>). The VIF values of each predictor in the dataset can be seen in <xref ref-type="table" rid="tab7">Table 7</xref>.</p>
<table-wrap position="float" id="tab7">
<label>Table 7</label>
<caption>
<p>VIF of predictors in the dataset.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Feature</th>
<th align="center" valign="top">VIF value</th>
<th align="left" valign="top">Feature</th>
<th align="center" valign="top">VIF value</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Residence</td>
<td align="center" valign="top">2.063</td>
<td align="left" valign="top">Age</td>
<td align="center" valign="top">1.442</td>
</tr>
<tr>
<td align="left" valign="top">Educational level</td>
<td align="center" valign="top">1.383</td>
<td align="left" valign="top">Total children ever born</td>
<td align="center" valign="top">25.109</td>
</tr>
<tr>
<td align="left" valign="top">Region</td>
<td align="center" valign="top">1.312</td>
<td align="left" valign="top">Number of household member</td>
<td align="center" valign="top">1.108</td>
</tr>
<tr>
<td align="left" valign="top">Sex of household head</td>
<td align="center" valign="top">1.024</td>
<td align="left" valign="top">Number of living children</td>
<td align="center" valign="top">1.301</td>
</tr>
<tr>
<td align="left" valign="top">Literacy</td>
<td align="center" valign="top">1.460</td>
<td align="left" valign="top">Recent sexual activity</td>
<td align="center" valign="top">3.624</td>
</tr>
<tr>
<td align="left" valign="top">Wealth index</td>
<td align="center" valign="top">1.415</td>
<td align="left" valign="top">Currently working</td>
<td align="center" valign="top">1.192</td>
</tr>
<tr>
<td align="left" valign="top">Knowledge of ovulatory cycle</td>
<td align="center" valign="top">1.057</td>
<td align="left" valign="top">Awareness on STI</td>
<td align="center" valign="top">1.215</td>
</tr>
<tr>
<td align="left" valign="top">Knowledge of any contraceptive method</td>
<td align="center" valign="top">1.097</td>
<td align="left" valign="top">Worked in last 12&#x2009;months</td>
<td align="center" valign="top">16.599</td>
</tr>
<tr>
<td align="left" valign="top">Respondent circumcised</td>
<td align="center" valign="top">1.015</td>
<td align="left" valign="top">Ever heard of AIDS</td>
<td align="center" valign="top">1.213</td>
</tr>
<tr>
<td align="left" valign="top">Current marital status</td>
<td align="center" valign="top">3.294</td>
<td align="left" valign="top">Alcohol drinking</td>
<td align="center" valign="top">1.145</td>
</tr>
<tr>
<td align="left" valign="top">Age at first sex</td>
<td align="center" valign="top">1.375</td>
<td align="left" valign="top">Intimate partner violence</td>
<td align="center" valign="top">1.088</td>
</tr>
<tr>
<td align="left" valign="top">Know a place to get HIV test</td>
<td align="center" valign="top">1.109</td>
<td align="left" valign="top">Media exposure</td>
<td align="center" valign="top">1.372</td>
</tr>
<tr>
<td align="left" valign="top">Have you ever chewed Chat?</td>
<td align="center" valign="top">1.144</td>
<td align="left" valign="top">Stigma and discrimination</td>
<td align="center" valign="top">1.034</td>
</tr>
<tr>
<td align="left" valign="top">Exposure to family planning</td>
<td align="center" valign="top">1.214</td>
<td align="left" valign="top">Number of Sexual partner</td>
<td align="center" valign="top">1.596</td>
</tr>
<tr>
<td align="left" valign="top">Smoking status</td>
<td align="center" valign="top">1.014</td>
<td align="left" valign="top">Health insurance</td>
<td align="center" valign="top">1.062</td>
</tr>
<tr>
<td align="left" valign="top">Resource control</td>
<td align="center" valign="top">1.378</td>
<td/>
<td/>
</tr>
</tbody>
</table>
</table-wrap>
<p><xref ref-type="table" rid="tab7">Table 7</xref> presents the Variance Inflation Factor (VIF) for each predictor in the dataset. The highest VIF value is observed for the variable &#x201C;total children ever born,&#x201D; which stands at 25.109. Other predictors such as &#x201C;working in the last 12&#x2009;months,&#x201D; &#x201C;current marital status,&#x201D; &#x201C;recent sexual activity,&#x201D; and &#x201C;residence&#x201D; also exhibit relatively high VIF scores, although they are lower than 5. A VIF between 1 and 5 suggests that these predictors are not strongly correlated and can be considered for inclusion in the adolescent HIV testing model. Therefore, it is recommended to include &#x201C;working in the last 12&#x2009;months,&#x201D; &#x201C;current marital status,&#x201D; &#x201C;recent sexual activity,&#x201D; and &#x201C;residence&#x201D; as predictors when building the adolescent HIV testing model. In addition, we used SHAP plot of feature selection. Based on SHAP plot age, wealth, sexual activity, educational status, exposure to family planning, IPV, know the place of HIV testing and awareness on STI and AIDS were identified (<xref ref-type="fig" rid="fig6">Figure 6</xref>).</p>
<fig position="float" id="fig6">
<label>Figure 6</label>
<caption>
<p>SHAP plot of features selected.</p>
</caption>
<graphic xlink:href="fpubh-12-1341279-g006.tif"/>
</fig>
</sec>
<sec id="sec25">
<title>Data analysis and feature selection</title>
<p>A comprehensive literature review has thoroughly examined 31 features that contribute to HIV testing services for adolescents. Through a feature evaluator, the significance of these factors was assessed, resulting in the identification of 20 highly important variables during the feature selection process. However, only 20 features were included in the analysis, while others were excluded based on specific criteria outlined in <xref ref-type="fig" rid="fig1">Figure 1</xref>.</p>
<p>To predict the HIV testing status of adolescents, the significance of each factor was calculated, leading to the selection of 20 predictors for machine learning (ML) algorithms. Among these predictors, age emerged as the most important factor for HIV testing services, with a value of 0.31733. On the other hand, the total number of children born was found to be the least important predictor, with a value of 0.08367. The importance of each feature in the dataset was calculated and presented in <xref ref-type="table" rid="tab5">Table 5</xref>, displaying the variables in descending order of ranking.</p>
</sec>
<sec id="sec26">
<title>Developing and evaluating models</title>
<p>In this study, our objective was to predict HIV testing among adolescents by selecting the most optimal features and utilizing eight different machine learning (ML) algorithms. These algorithms included J48, RF, LR, MLP, logit Boost, k-NN, SVM, and NB. To evaluate the performance of each algorithm, we conducted 10-fold cross-validation with a seed value of two and assessed various metrics, such as sensitivity, specificity, accuracy, precision, and the receiver operating characteristic (ROC) curve. The results of the cross-validation are presented in <xref ref-type="table" rid="tab8">Table 8</xref>. Our experimental findings revealed that the J48 decision tree algorithm outperformed the other ML algorithms in accurately predicting adolescent HIV testing. It achieved impressive performance metrics, including a sensitivity of 81.30%, specificity of 80.90%, accuracy of 81.3%, precision of 81.0%, and an ROC value of 86.3%. <xref ref-type="fig" rid="fig7">Figure 7</xref> visually depicts the performance metrics of the ML algorithms used in this study, while <xref ref-type="fig" rid="fig8">Figure 8</xref> presents a comparison of the area under the ROC curve for these algorithms. Notably, the SVM algorithm exhibited the lowest performance with an ROC value of 66.8% according to the ROC analysis. <xref ref-type="fig" rid="fig9">Figure 9</xref> presents the false positive rate and true positive rate of each algorithm. For a comprehensive summary of the performance evaluation of each algorithm, please refer to <xref ref-type="table" rid="tab8">Table 8</xref>.</p>
<table-wrap position="float" id="tab8">
<label>Table 8</label>
<caption>
<p>Performance evaluation of the selected ML algorithms for HIV testing prediction.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Algorithm</th>
<th align="center" valign="top">Sensitivity (%)</th>
<th align="center" valign="top">Specificity (%)</th>
<th align="center" valign="top">Precision (%)</th>
<th align="center" valign="top">Accuracy (%)</th>
<th align="center" valign="top">Area under ROC (%)</th>
<th align="center" valign="top">F1 score</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Random forest</td>
<td align="center" valign="top">72.90</td>
<td align="center" valign="top">61.8</td>
<td align="center" valign="top">72.3</td>
<td align="center" valign="top">72.3234</td>
<td align="center" valign="top">77.9</td>
<td align="center" valign="top">72.598</td>
</tr>
<tr>
<td align="left" valign="top">Logit boost</td>
<td align="center" valign="top">74.80</td>
<td align="center" valign="top">73.96</td>
<td align="center" valign="top">74.10</td>
<td align="center" valign="top">74.8334</td>
<td align="center" valign="top">81.0</td>
<td align="center" valign="top">74.448</td>
</tr>
<tr>
<td align="left" valign="top">KNN</td>
<td align="center" valign="top">68.70</td>
<td align="center" valign="top">55.32</td>
<td align="center" valign="top">68.1</td>
<td align="center" valign="top">68.725</td>
<td align="center" valign="top">67.80</td>
<td align="center" valign="top">68.398</td>
</tr>
<tr>
<td align="left" valign="top">MLP</td>
<td align="center" valign="top">71.4</td>
<td align="center" valign="top">67.26</td>
<td align="center" valign="top">71.1</td>
<td align="center" valign="top">71.4349</td>
<td align="center" valign="top">76.4</td>
<td align="center" valign="top">71.249</td>
</tr>
<tr>
<td align="left" valign="top">LR</td>
<td align="center" valign="top">75.2</td>
<td align="center" valign="top">78.47</td>
<td align="center" valign="top">74.6</td>
<td align="center" valign="top">75.2332</td>
<td align="center" valign="top">82.1</td>
<td align="center" valign="top">74.898</td>
</tr>
<tr>
<td align="left" valign="top">J48 decision tree</td>
<td align="center" valign="top">81.3</td>
<td align="center" valign="top">80.9</td>
<td align="center" valign="top">81.0</td>
<td align="center" valign="top">81.2972</td>
<td align="center" valign="top">86.3</td>
<td align="center" valign="top">81.149</td>
</tr>
<tr>
<td align="left" valign="top">Na&#x00EF;ve Bayes</td>
<td align="center" valign="top">73.9</td>
<td align="center" valign="top">87.80</td>
<td align="center" valign="top">74.5</td>
<td align="center" valign="top">73.9449</td>
<td align="center" valign="top">81.3</td>
<td align="center" valign="top">74.198</td>
</tr>
<tr>
<td align="left" valign="top">SVM</td>
<td align="center" valign="top">72.7</td>
<td align="center" valign="top">60.53</td>
<td align="center" valign="top">71.7</td>
<td align="center" valign="top">72.701</td>
<td align="center" valign="top">66.8</td>
<td align="center" valign="top">72.196</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig position="float" id="fig7">
<label>Figure 7</label>
<caption>
<p>Visual comparisons of ML algorithm capabilities for adolescent HIV testing services.</p>
</caption>
<graphic xlink:href="fpubh-12-1341279-g007.tif"/>
</fig>
<fig position="float" id="fig8">
<label>Figure 8</label>
<caption>
<p>Comparison of ROC area under the curve in the study conducted among adolescents HIV testing status in Ethiopia.</p>
</caption>
<graphic xlink:href="fpubh-12-1341279-g008.tif"/>
</fig>
<fig position="float" id="fig9">
<label>Figure 9</label>
<caption>
<p>The true positive rate and false positive rate of the eight algorithms.</p>
</caption>
<graphic xlink:href="fpubh-12-1341279-g009.tif"/>
</fig>
</sec>
<sec id="sec27">
<title>Association rule result</title>
<p>This study utilized the J48 decision tree feature importance method to select relevant features. Subsequently, the association mining rules were applied using the apriori algorithm for interpretation and a comparison of the best-selected features. From the association mining rules, a total of nine rules were identified with a confidence value of over 90% and the highest lift or interestingness. Among these rules, the twenty most significant ones were chosen for predicting adolescent HIV testing. The absolute minimum support count of the apriori algorithm was 2,251 in support of 0.5 and confidence of 0.9. The summary of quality measures of association rule mining analysis indicated that minimum support value&#x2009;=&#x2009;0.5018, confidence&#x2009;=&#x2009;0.9461, lift&#x2009;=&#x2009;0.9924. The total rules of this apriori algorithm of this study was 712,602 done by 48&#x2009;s.</p>
<p>Rule 1 indicated that individuals know the place of HIV testing (yes), intimate partner violence (=no), had media exposure (=yes), number of sexual partner (=more than one), awareness on AIDS (=yes), awareness on STI (=yes), sexual activity (=active in last 4&#x2009;weeks), age at first sex (=between16 to 24&#x2009;years), marital status (=married), knowledge on contraceptive (=knowledgeable), number of children (=one and above), knowledge on ovulatory cycle (=yes) and exposure to family planning (=yes) then the possibility of HIV testing was 95.33% of confidence with support value&#x2009;=&#x2009;95% and Lift value&#x2009;=&#x2009;31.444.</p>
<p>Rule 2 means that being rural residence, awareness on AIDS (=yes), awareness on STI (=yes), sexual activity (=active in last 4&#x2009;weeks), age at first sex (=between16 to 24&#x2009;years), marital status (=married), knowledge on contraceptive (=knowledgeable), number of children (=one and above), knowledge on ovulatory cycle (=yes) and literacy (able to read and write) the adolescent will be 92.24% chance of HIV testing with support&#x2009;=&#x2009;93% and Lift&#x2009;=&#x2009;33.334.</p>
<p>Rule 3 means that if age greater than 18&#x2009;years old, education (=secondary), age at first sex (16&#x2013;24&#x2009;years), recent sexual activity (active in last 4&#x2009;weeks), exposure to family planning (=yes), and had media exposure the probability of adolescent HIV testing is 100% of confidence with support of 0.00255, Lift&#x2009;=&#x2009;31.233%.</p>
<p>Rule 4 showed that if the adolescent attended secondary education, age above 18&#x2009;years, intimate partner violence (=no), had media exposure, number of sexual partner (more than one), had awareness on AIDS, and had awareness on STI a then the possibility of the adolescent for testing HIV is 100% of confidence, support&#x2009;=&#x2009;0.00222, Lift&#x2009;=&#x2009;33.377.</p>
<p>Rule 5 means that if the adolescent attended secondary education, know the place of HIV testing, age at first sex (<xref ref-type="bibr" rid="ref16">16</xref>&#x2013;<xref ref-type="bibr" rid="ref24">24</xref>), urban residence, and had knowledge on contraceptives then the possibility of adolescent HIV testing is 100% confidence and support&#x2009;=&#x2009;0.001033, Lift&#x2009;=&#x2009;32.64.</p>
<p>According to the association rule analysis, it was found that several factors strongly predict adolescent HIV testing in Ethiopia. These factors include knowing the place of HIV testing, being above 18&#x2009;years old, having a secondary education or higher, engaging in first sexual activity above the age of 16, living in urban areas, having recent sexual activity, being exposed to family planning information, having more than one sexual partner, being aware of STIs and AIDS, and having media exposure. The association rules indicate a strong relationship between these independent features and the dependent feature (adolescent HIV testing). This is supported by the lift values, which are greater than one. This suggests that the presence of these factors increases the likelihood of adolescents getting tested for HIV in Ethiopia. In summary, the association rule analysis highlights the significant predictors of adolescent HIV testing in Ethiopia, emphasizing the interconnections between various factors and the importance of these factors in promoting HIV testing among adolescents.</p>
</sec>
</sec>
<sec sec-type="discussion" id="sec28">
<title>Discussion</title>
<p>In this study, the researchers aimed to develop a highly accurate machine-learning model for predicting HIV testing among adolescents in Ethiopia. What sets this research apart is its focus on exploring new predictor variables that have not been previously investigated. To achieve this, the researchers analyzed secondary data from the demographic and health survey of Ethiopia, which provided them with relevant information on the testing status of adolescents, their medical history, and demographic characteristics.</p>
<p>The researchers conducted a comprehensive study on predicting HIV testing status using various statistical analysis techniques and feature selection methods. They utilized a range of machine-learning models, including J48 decision tree, RF, k-NN, MLP, NB, logit Boost, SVM, and LR models. Among these techniques, the J48 decision tree model demonstrated the highest performance, achieving an accuracy of 81.29%. It also showed a sensitivity of 81.30%, precision of 81.0%, specificity of 80.90%, and an ROC of approximately 86.30%. These results indicate that the J48 decision tree is an exceptionally effective machine-learning technique for this specific task. The study also revealed that the J48 decision tree, KNN, MLP, LR, Na&#x00EF;ve Bayes, and XGBoost models exhibited good prediction performance, with ROC values above 76.4%. These models also demonstrated superior diagnostic efficiency compared to other models trained with the same parameters. Overall, this research provides valuable insights into the development of machine-learning models for predicting HIV testing services among adolescents. Implementing these models could potentially enhance adolescent HIV testing status and reduce the number of undiagnosed chronic HIV carriers in the country.</p>
<p>Age has the most significant impact on an individual&#x2019;s HIV testing status compared to other factors. Any change in age can have a more pronounced influence than other variables. Older individuals are more likely to undergo HIV testing. Moreover, individuals who are aware of the locations where HIV testing is available have a higher likelihood of getting tested. Additionally, the age at which adolescents first engage in sexual activity also plays a role in their HIV testing behavior in Ethiopia. This could be due to increased access to information and improved decision-making abilities among adolescents. Furthermore, it appears that those who are sexually active are more likely to seek HIV testing compared to their counterparts. Individuals with lower levels of education tend to have less knowledge about HIV risk mitigation measures, thereby increasing their vulnerability to HIV. This highlights the importance of HIV testing services in providing individuals with knowledge and awareness about HIV prevention.</p>
<p>Recent studies have also investigated the potential predictive algorithms for HIV testing services among adolescents. For instance, Mutai et al. (<xref ref-type="bibr" rid="ref43">43</xref>) developed a prediction model using six machine-learning techniques, with the XGBoost algorithm showing promising results. They identified factors such as sex, age, relationship with family head, highest level of education, highest grade at school level, work for payment, avoiding pregnancy, age at first experience of sex, and wealth quintile as having the highest weights in their model. These findings align with our own research, which predicts HIV testing status among adolescents based on nationally representative data. Another study involving 6,346 men who have sex with men indicated that the RF algorithm performed best, with an AUC of 0.942 compared to other three machine learning algorithms (<xref ref-type="bibr" rid="ref50">50</xref>). Overall, these studies contribute valuable knowledge to the field of HIV testing prediction models, offering insights into the potential factors and machine-learning techniques that can be utilized to improve testing services among adolescents.</p>
<p>Several research studies have investigated the application of machine learning (ML) techniques for predicting HIV testing and HIV infection among different population category in different countries. A longitudinal study conducted among 2,564 adolescent indicated that Random forest was the best model of predicting risky behaviour of the adolescents with AUC&#x2009;=&#x2009;0.84 on training data and 0.87 on testing data as compared with other four algorithm of the study (<xref ref-type="bibr" rid="ref51">51</xref>). Based on other study conducted on HIV and STI testing at a clinic, ten algorithms were used for prediction testing (<xref ref-type="bibr" rid="ref52">52</xref>). Out of those ten algorithms, the XGBoost model demonstrated the highest accuracy with an AUC of 62.8% and an F1 score of 70.8%. In another study that involved 55,151 males and 69,626 females in East and South Africa, the gradient boosting trees algorithm was found to be the most effective in predicting HIV status (<xref ref-type="bibr" rid="ref53">53</xref>). This implies that machine-learning was the best predictor of testing services and identifies the best features of the outcome.</p>
<p>A 81.3% sensitivity was required in ensuring that 81.3% of individuals tested and knew their status. With the J48 decision algorithm, we utilized 20 most predictive variables accordingly to establish the number required to screen to know one individual with HIV. There community-based and facility-based screening were studies in previous literatures (<xref ref-type="bibr" rid="ref54">54</xref>). This implies J48 decision tree predicts the testing of adolescent in Ethiopia. Our study revealed that the J48 decision tree model algorithm outperformed other models, emerging as the top performer. Interestingly, our findings differ from Orel&#x2019;s research in terms of the predictors identified, except for the consistent identification of individual age and wealth as predictors of the disease (<xref ref-type="bibr" rid="ref55">55</xref>). The possible justification for variation could be variation in demographic characteristics of the study participants.</p>
<p>There are alternative screening methods available; however, they do come with certain limitations. Universal screening, for instance, involves conducting tests on all patients in healthcare facilities. While this approach can be effective, it may not be cost-effective in situations where the incidence of the condition being screened for is low like Ethiopia (<xref ref-type="bibr" rid="ref56">56</xref>). Indicator-condition-guided testing, which fails to consider important factors such as age, sex, and medical conditions, overlooks their association with a reduced risk of HIV transmission (<xref ref-type="bibr" rid="ref57">57</xref>). In settings where HIV prevalence is high, it is effective to target well-established risk groups, such as families (through index contact elicitation), to reach individuals at high risk (<xref ref-type="bibr" rid="ref58">58</xref>). However, this approach may unintentionally neglect lesser-known or harder-to-define subgroups that are also vulnerable (<xref ref-type="bibr" rid="ref59">59</xref>), resulting in inefficient resource allocation (<xref ref-type="bibr" rid="ref60">60</xref>). Even in the absence of recognized risk factors, self-assessment provides a means of identifying individuals at high risk. Therefore, it is crucial to consider targeted testing and demographic factors like age in order to achieve the first 95% of the UNAIDs target in the country.</p>
<p>An individual&#x2019;s perception of risk is influenced by their level of awareness about HIV and its related factors (<xref ref-type="bibr" rid="ref61">61</xref>). Sometimes, unexpected or uncontrolled exposures to the virus can go unnoticed. In the case of a widespread outbreak, it may not be obvious which demographic subgroups should be targeted for prevention efforts. Merely providing Pre-Exposure Prophylaxis to established high-risk subgroups, such as young people or mobile populations, may not be effective. Therefore, it is important to consider a PrEP technique that takes into account individual characteristics in a more nuanced way (<xref ref-type="bibr" rid="ref62">62</xref>). This approach could help reduce the cost of preventing new HIV infections. Our method offers an alternative to the limitations mentioned earlier and could potentially complement existing strategies for identifying individuals who would benefit the most from enhanced mitigation measures.</p>
</sec>
<sec sec-type="conclusions" id="sec29">
<title>Conclusion</title>
<p>In our study, we utilized country-level demographic and health services data to develop a groundbreaking model for predicting adolescent HIV testing services. This model incorporates demographic characteristics that were carefully developed after an extensive review of a large dataset, showcasing its superior predictive capacity compared to existing literature. The main objective of our model is to prioritize early identification of barriers to testing high-risk patients and optimize the utilization of strained public healthcare systems.</p>
<p>We strongly believe that our proposed technique has the potential to significantly enhance healthcare systems&#x2019; decision-making processes, enabling precise and targeted HIV testing strategies in countries. This, in turn, empowers adolescents by promoting HIV testing and designing well-organized testing strategies. Our study specifically focused on creating and evaluating machine learning-based prediction models for HIV testing services in Ethiopia, utilizing 20 key predictors. Among the eight machine learning algorithms tested, the J48 decision tree model demonstrated the highest classification accuracy and precision. This suggests that our proposed model can effectively predict adolescent HIV testing, thereby optimizing the allocation of limited resources in the country and surpassing the achievement of the 95&#x2013;95-95 target. Importantly, our model identified age as the highest predictor of adolescent HIV testing, while the number of children ever born was the lowest predictor.</p>
<p>In conclusion, the integration of machine learning algorithms with comprehensive national-based data enables accurate classification of individuals for HIV testing services. This advancement holds great promise in improving healthcare outcomes and resource management, particularly during the ongoing pandemic. Surveys providing detailed individual-level data, including demographic characteristics, social history, laboratory tests, and disease results, have become more available. Leveraging these datasets through advanced approaches can significantly aid in the prevention, diagnosis, and testing of HIV and other diseases. By incorporating this approach into community-based or facility-based testing programs, it becomes possible to identify individuals at high risk. However, further studies are required to refine this model, effectively integrate it, and apply it in real-world primary care settings.</p>
<sec id="sec30">
<title>Limitation and strength</title>
<p>This retrospective study analyzed demographic and health survey data that exhibited irregularities and imbalances. To address this issue, the researchers took measures to balance the dataset by removing noisy and inadequate records. Specifically, they focused on tackling the problem of imbalanced classes, where the number of records related to the HIV-tested class was significantly lower than the tested or never tested (1,552 vs. 2,950).</p>
<p>To evaluate the performance of each machine learning algorithm, various criteria were utilized. Furthermore, external validation of the proposed model was conducted using multi-center country-level data, aiming to enhance the generalizability of the predictions. However, it is important to note that the researchers relied on self-reported data from the demographic and health survey, which may introduce inconclusiveness and potentially impact the training data.</p>
<p>For future research, the researchers recommend that behavioral features be incorporated to further enhance prediction accuracy. Additionally, it is crucial to monitor the dynamic variations of significant features over time to better identify adolescents HIV testing in the countries. While the study provides valuable insights using the available data, there are areas for improvement and avenues for further research to enhance the understanding and prediction of HIV testing services in this specific sub-population group in Ethiopia.</p>
</sec>
</sec>
<sec sec-type="data-availability" id="sec31">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found in the article/<xref rid="SM1" ref-type="supplementary-material">Supplementary material</xref>.</p>
</sec>
<sec sec-type="ethics-statement" id="sec32">
<title>Ethics statement</title>
<p>Ethical approval was not required for the study involving humans in accordance with the local legislation and institutional requirements. Written informed consent to participate in this study was not required from the participants or the participants&#x2019; legal guardians/next of kin in accordance with the national legislation and the institutional requirements.</p>
</sec>
<sec sec-type="author-contributions" id="sec33">
<title>Author contributions</title>
<p>MA: Conceptualization, Data curation, Formal analysis, Methodology, Resources, Software, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. YN: Data curation, Formal analysis, Investigation, Methodology, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing.</p>
</sec>
</body>
<back>
<sec sec-type="funding-information" id="sec34">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research, authorship, and/or publication of this article.</p>
</sec>
<ack>
<p>The authors would like to acknowledge the DHS for providing the data analyzed in this study.</p>
</ack>
<sec sec-type="COI-statement" id="sec35">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="sec100" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="sec161">
<title>Supplementary material</title>
<p>The Supplementary material for this article can be found online at: <ext-link xlink:href="https://www.frontiersin.org/articles/10.3389/fpubh.2024.1341279/full#supplementary-material" ext-link-type="uri">https://www.frontiersin.org/articles/10.3389/fpubh.2024.1341279/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Image_1.jpg" id="SM1" mimetype="image/jpeg" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<fn-group>
<title>Abbreviations</title>
<fn fn-type="abbr"><p>ART, antiretroviral therapy; HIV, human immunodeficiency virus; ML, machine learning; Logit GB, logistic gradient boosting; LR, logistic regression; KNN, K-nearest neighbor; RF, random forest; ROC, receiver operating characteristic; STI, sexually transmitted infection; SVM, support vector machine; USAIDS, Joint United Nations Programme on HIV/AIDS; VCT, voluntary counseling and testing; WHO, World Health Organization</p></fn>
</fn-group>

<fn-group>
<fn id="fn0001">
<p><sup>1</sup><ext-link xlink:href="https://dhsprogram.com/" ext-link-type="uri">https://dhsprogram.com/</ext-link>
</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="ref1">
<label>1.</label>
<citation citation-type="other"><person-group person-group-type="author">
<name>
<surname>Geza</surname>
<given-names>G.</given-names>
</name>
</person-group> <source>Evaluation of the Effect of Adolescent and Youth Friendly Services Implementation on HIV Testing Uptake among Youth (Aged 15&#x2013;24 Years) in Health Facilities of Amathole District, Eastern Cape</source>. (<year>2020</year>). <fpage>20</fpage>&#x2013;<lpage>29</lpage>. Available at: <ext-link xlink:href="http://hdl.handle.net/11394/7642" ext-link-type="uri">http://hdl.handle.net/11394/7642</ext-link></citation>
</ref>
<ref id="ref2">
<label>2.</label>
<citation citation-type="book"><person-group person-group-type="author">
<collab id="coll1">UNICEF</collab>
</person-group>. <source>Children and AIDS 2015 Statistical Update</source>. <publisher-loc>New York</publisher-loc>: <publisher-name>UNICEF</publisher-name> (<year>2015</year>).</citation>
</ref>
<ref id="ref3">
<label>3.</label>
<citation citation-type="book"><person-group person-group-type="author"><name>
<surname>Mofenson</surname>
<given-names>LM</given-names>
</name> <name>
<surname>Cotton</surname>
<given-names>MF</given-names>
</name></person-group>. <article-title>The challenges of success: adolescents with perinatal HIV infection</article-title>. <source>J Int AID Soc.</source> (<year>2013</year>). <volume>16</volume>:<fpage>18650</fpage>.</citation>
</ref>
<ref id="ref4">
<label>4.</label>
<citation citation-type="other"><person-group person-group-type="author">
<name>
<surname>Collaborators</surname>
<given-names>G.</given-names>
</name>
</person-group> <source>Global, Regional, and National Incidence, Prevalence, and Years Lived with Disability for 354 Diseases and Injuries for 195 Countries and Territories, 1990&#x2013;2017: A Systematic Analysis for the Global Burden of Disease Study 2017</source>. (<year>2018</year>) <volume>392</volume>:<fpage>1789</fpage>&#x2013;<lpage>858</lpage>.</citation>
</ref>
<ref id="ref5">
<label>5.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Mushamiri</surname>
<given-names>I</given-names>
</name> <name>
<surname>Adudans</surname>
<given-names>M</given-names>
</name> <name>
<surname>Apat</surname>
<given-names>D</given-names>
</name> <name>
<surname>Ben</surname>
<given-names>AY</given-names>
</name></person-group>. <article-title>Optimizing PMTCT efforts by repeat HIV testing during antenatal and perinatal care in resource-limited settings: a longitudinal assessment of HIV seroconversion</article-title>. <source>PLoS One</source>. (<year>2020</year>) <volume>15</volume>:<fpage>e0233396</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0233396</pub-id>, PMID: <pub-id pub-id-type="pmid">32470004</pub-id></citation>
</ref>
<ref id="ref6">
<label>6.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Hussen</surname>
<given-names>R</given-names>
</name> <name>
<surname>Zenebe</surname>
<given-names>WA</given-names>
</name> <name>
<surname>Mamo</surname>
<given-names>TT</given-names>
</name> <name>
<surname>Shaka</surname>
<given-names>MF</given-names>
</name></person-group>. <article-title>Determinants of HIV infection among children born from mothers on prevention of mother to child transmission programme of HIV in southern Ethiopia: a case&#x2013;control study</article-title>. <source>BMJ Open</source>. (<year>2022</year>) <volume>12</volume>:<fpage>e048491</fpage>. doi: <pub-id pub-id-type="doi">10.1136/bmjopen-2020-048491</pub-id>, PMID: <pub-id pub-id-type="pmid">35131814</pub-id></citation>
</ref>
<ref id="ref7">
<label>7.</label>
<citation citation-type="book"><person-group person-group-type="author">
<collab id="coll2">World Health Organization</collab>
</person-group>. <source>Consolidated Guidelines on HIV Testing Services: 5Cs: Consent, Confidentiality, Counselling, Correct Results and Connection 2015</source>. <publisher-loc>Geneva, Switzerland</publisher-loc>: <publisher-name>World Health Organization</publisher-name> (<year>2015</year>).</citation>
</ref>
<ref id="ref8">
<label>8.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Ojo</surname>
<given-names>K</given-names>
</name> <name>
<surname>Delaney</surname>
<given-names>M</given-names>
</name></person-group>. <article-title>Economic and demograhic consequences of AIDS in Namibia: rapid assessment of the costs</article-title>. <source>Int J Health Plann Manag</source>. (<year>1997</year>) <volume>12</volume>:<fpage>315</fpage>&#x2013;<lpage>26</lpage>. doi: <pub-id pub-id-type="doi">10.1002/(SICI)1099-1751(199710/12)12:4&#x003C;315::AID-HPM492&#x003E;3.0.CO;2-A</pub-id>, PMID: <pub-id pub-id-type="pmid">10177418</pub-id></citation>
</ref>
<ref id="ref9">
<label>9.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Beck</surname>
<given-names>EJ</given-names>
</name> <name>
<surname>Miners</surname>
<given-names>AH</given-names>
</name> <name>
<surname>Tolley</surname>
<given-names>K</given-names>
</name></person-group>. <article-title>The cost of HIV treatment and care: a global review</article-title>. <source>PharmacoEconomics</source>. (<year>2001</year>) <volume>19</volume>:<fpage>13</fpage>&#x2013;<lpage>39</lpage>. doi: <pub-id pub-id-type="doi">10.2165/00019053-200119010-00002</pub-id></citation>
</ref>
<ref id="ref10">
<label>10.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Muyunda</surname>
<given-names>B</given-names>
</name> <name>
<surname>Mee</surname>
<given-names>P</given-names>
</name> <name>
<surname>Todd</surname>
<given-names>J</given-names>
</name> <name>
<surname>Musonda</surname>
<given-names>P</given-names>
</name> <name>
<surname>Michelo</surname>
<given-names>C</given-names>
</name></person-group>. <article-title>Estimating levels of HIV testing coverage and use in prevention of mother-to-child transmission among women of reproductive age in Zambia</article-title>. <source>Arch Public Health</source>. (<year>2018</year>) <volume>76</volume>:<fpage>1</fpage>&#x2013;<lpage>9</lpage>. doi: <pub-id pub-id-type="doi">10.1186/s13690-018-0325-x</pub-id></citation>
</ref>
<ref id="ref11">
<label>11.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Muyunda</surname>
<given-names>B</given-names>
</name> <name>
<surname>Musonda</surname>
<given-names>P</given-names>
</name> <name>
<surname>Mee</surname>
<given-names>P</given-names>
</name> <name>
<surname>Todd</surname>
<given-names>J</given-names>
</name> <name>
<surname>Michelo</surname>
<given-names>C</given-names>
</name></person-group>. <article-title>Educational attainment as a predictor of HIV testing uptake among women of child-bearing age: analysis of 2014 demographic and health survey in Zambia</article-title>. <source>Front Public Health</source>. (<year>2018</year>) <volume>6</volume>:<fpage>192</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fpubh.2018.00192</pub-id>, PMID: <pub-id pub-id-type="pmid">30155454</pub-id></citation>
</ref>
<ref id="ref12">
<label>12.</label>
<citation citation-type="journal"><person-group person-group-type="author">
<name>
<surname>Painter</surname>
<given-names>TM</given-names>
</name>
</person-group>. <article-title>Voluntary counseling and testing for couples: a high-leverage intervention for HIV/AIDS prevention in sub-Saharan Africa</article-title>. <source>Soc Sci Med</source>. (<year>2001</year>) <volume>53</volume>:<fpage>1397</fpage>&#x2013;<lpage>411</lpage>. doi: <pub-id pub-id-type="doi">10.1016/S0277-9536(00)00427-5</pub-id>, PMID: <pub-id pub-id-type="pmid">11710416</pub-id></citation>
</ref>
<ref id="ref13">
<label>13.</label>
<citation citation-type="other"><person-group person-group-type="author">
<collab id="coll3">United Nations</collab>
</person-group>. Sustainable Development Goals: 17 Goals to Transform Our World. United Nations; (<year>2015</year>). <comment>Available at:</comment> <ext-link xlink:href="https://www.un.org/sustainabledevelopment/energy/" ext-link-type="uri">https://www.un.org/sustainabledevelopment/energy/</ext-link> (Accessed June 04, 2018).</citation>
</ref>
<ref id="ref14">
<label>14.</label>
<citation citation-type="journal"><person-group person-group-type="author">
<name>
<surname>Van de Perre</surname>
<given-names>P</given-names>
</name>
</person-group>. <article-title>HIV voluntary counselling and testing in community health services</article-title>. <source>Lancet</source>. (<year>2000</year>) <volume>356</volume>:<fpage>86</fpage>&#x2013;<lpage>7</lpage>. doi: <pub-id pub-id-type="doi">10.1016/S0140-6736(00)02462-4</pub-id></citation>
</ref>
<ref id="ref15">
<label>15.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Godfrey-Faussett</surname>
<given-names>P</given-names>
</name> <name>
<surname>Maher</surname>
<given-names>D</given-names>
</name> <name>
<surname>Mukadi</surname>
<given-names>YD</given-names>
</name> <name>
<surname>Nunn</surname>
<given-names>P</given-names>
</name> <name>
<surname>Perri&#x00EB;ns</surname>
<given-names>J</given-names>
</name> <name>
<surname>Raviglione</surname>
<given-names>M</given-names>
</name></person-group>. <article-title>How human immunodeficiency virus voluntary testing can contribute to tuberculosis control</article-title>. <source>Bull World Health Organ</source>. (<year>2002</year>) <volume>80</volume>:<fpage>939</fpage>&#x2013;<lpage>45</lpage>. PMID: <pub-id pub-id-type="pmid">12571721</pub-id></citation>
</ref>
<ref id="ref16">
<label>16.</label>
<citation citation-type="book"><person-group person-group-type="author">
<collab id="coll4">World Health Organization</collab>
</person-group>. <source>Consolidated Guidelines on HIV Prevention, Testing, Treatment, Service Delivery and Monitoring: Recommendations for a Public Health Approach</source>. <publisher-loc>Geneva, Switzerland</publisher-loc>: <publisher-name>World Health Organization</publisher-name> (<year>2021</year>).</citation>
</ref>
<ref id="ref17">
<label>17.</label>
<citation citation-type="book"><person-group person-group-type="author">
<collab id="coll5">UNAIDS</collab>
</person-group>. <source>Understanding Fast-Track: Accelerating Action to End the AIDS Epidemic by 2030</source>. <publisher-loc>Geneva</publisher-loc>: <publisher-name>UNAIDS</publisher-name> (<year>2015</year>).</citation>
</ref>
<ref id="ref18">
<label>18.</label>
<citation citation-type="book"><person-group person-group-type="author">
<collab id="coll6">World Health Organization</collab>
</person-group>. <source>Consolidated Guidelines on the Use of Antiretroviral Drugs for Treating and Preventing HIV Infection: Recommendations for a Public Health Approach</source>. <publisher-loc>Geneva, Switzerland</publisher-loc>: <publisher-name>World Health Organization</publisher-name> (<year>2016</year>).</citation>
</ref>
<ref id="ref19">
<label>19.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Asaolu</surname>
<given-names>IO</given-names>
</name> <name>
<surname>Gunn</surname>
<given-names>JK</given-names>
</name> <name>
<surname>Center</surname>
<given-names>KE</given-names>
</name> <name>
<surname>Koss</surname>
<given-names>MP</given-names>
</name> <name>
<surname>Iwelunmor</surname>
<given-names>JI</given-names>
</name> <name>
<surname>Ehiri</surname>
<given-names>JE</given-names>
</name></person-group>. <article-title>Predictors of HIV testing among youth in sub-Saharan Africa: a cross-sectional study</article-title>. <source>PLoS One</source>. (<year>2016</year>) <volume>11</volume>:<fpage>e0164052</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0164052</pub-id>, PMID: <pub-id pub-id-type="pmid">27706252</pub-id></citation>
</ref>
<ref id="ref20">
<label>20.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Rosenberg</surname>
<given-names>NE</given-names>
</name> <name>
<surname>Westreich</surname>
<given-names>D</given-names>
</name> <name>
<surname>B&#x00E4;rnighausen</surname>
<given-names>T</given-names>
</name> <name>
<surname>Miller</surname>
<given-names>WC</given-names>
</name> <name>
<surname>Behets</surname>
<given-names>F</given-names>
</name> <name>
<surname>Maman</surname>
<given-names>S</given-names>
</name> <etal/></person-group>. <article-title>Assessing the effect of HIV counseling and testing on HIV acquisition among south African youth</article-title>. <source>AIDS (London, England)</source>. (<year>2013</year>) <volume>27</volume>:<fpage>2765</fpage>&#x2013;<lpage>73</lpage>. doi: <pub-id pub-id-type="doi">10.1097/01.aids.0000432454.68357.6a</pub-id>, PMID: <pub-id pub-id-type="pmid">23887069</pub-id></citation>
</ref>
<ref id="ref21">
<label>21.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Kurth</surname>
<given-names>AE</given-names>
</name> <name>
<surname>Lally</surname>
<given-names>MA</given-names>
</name> <name>
<surname>Choko</surname>
<given-names>AT</given-names>
</name> <name>
<surname>Inwani</surname>
<given-names>IW</given-names>
</name> <name>
<surname>Fortenberry</surname>
<given-names>JD</given-names>
</name></person-group>. <article-title>HIV testing and linkage to services for youth</article-title>. <source>J Int AIDS Soc</source>. (<year>2015</year>) <volume>18</volume>:<fpage>19433</fpage>. doi: <pub-id pub-id-type="doi">10.7448/IAS.18.2.19433</pub-id>, PMID: <pub-id pub-id-type="pmid">25724506</pub-id></citation>
</ref>
<ref id="ref22">
<label>22.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Leta</surname>
<given-names>TH</given-names>
</name> <name>
<surname>Sand&#x00F8;y</surname>
<given-names>IF</given-names>
</name> <name>
<surname>Fylkesnes</surname>
<given-names>K</given-names>
</name></person-group>. <article-title>Factors affecting voluntary HIV counselling and testing among men in Ethiopia: a cross-sectional survey</article-title>. <source>BMC Public Health</source>. (<year>2012</year>) <volume>12</volume>:<fpage>1</fpage>&#x2013;<lpage>12</lpage>. doi: <pub-id pub-id-type="doi">10.1186/1471-2458-12-438</pub-id></citation>
</ref>
<ref id="ref23">
<label>23.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Teklehaimanot</surname>
<given-names>HD</given-names>
</name> <name>
<surname>Teklehaimanot</surname>
<given-names>A</given-names>
</name> <name>
<surname>Yohannes</surname>
<given-names>M</given-names>
</name> <name>
<surname>Biratu</surname>
<given-names>D</given-names>
</name></person-group>. <article-title>Factors influencing the uptake of voluntary HIV counseling and testing in rural Ethiopia: a cross sectional study</article-title>. <source>BMC Public Health</source>. (<year>2016</year>) <volume>16</volume>:<fpage>1</fpage>&#x2013;<lpage>13</lpage>. doi: <pub-id pub-id-type="doi">10.1186/s12889-016-2918-z</pub-id></citation>
</ref>
<ref id="ref24">
<label>24.</label>
<citation citation-type="journal"><person-group person-group-type="author">
<name>
<surname>Caron</surname>
<given-names>PO</given-names>
</name>
</person-group>. <article-title>Multilevel analysis of matching behavior</article-title>. <source>J Exp Anal Behav</source>. (<year>2019</year>) <volume>111</volume>:<fpage>183</fpage>&#x2013;<lpage>91</lpage>. doi: <pub-id pub-id-type="doi">10.1002/jeab.510</pub-id></citation>
</ref>
<ref id="ref25">
<label>25.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Lakhe</surname>
<given-names>NA</given-names>
</name> <name>
<surname>Diallo Mbaye</surname>
<given-names>K</given-names>
</name> <name>
<surname>Sylla</surname>
<given-names>K</given-names>
</name> <name>
<surname>Ndour</surname>
<given-names>CT</given-names>
</name></person-group>. <article-title>HIV screening in men and women in Senegal: coverage and associated factors; analysis of the 2017 demographic and health survey</article-title>. <source>BMC Infect Dis</source>. (<year>2020</year>) <volume>20</volume>:<fpage>1</fpage>&#x2013;<lpage>12</lpage>. doi: <pub-id pub-id-type="doi">10.1186/s12879-019-4717-5</pub-id>, PMID: <pub-id pub-id-type="pmid">31892320</pub-id></citation>
</ref>
<ref id="ref26">
<label>26.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Mandiwa</surname>
<given-names>C</given-names>
</name> <name>
<surname>Namondwe</surname>
<given-names>B</given-names>
</name></person-group>. <article-title>Uptake and correlates of HIV testing among men in Malawi: evidence from a national population&#x2013;based household survey</article-title>. <source>BMC Health Serv Res</source>. (<year>2019</year>) <volume>19</volume>:<fpage>1</fpage>&#x2013;<lpage>8</lpage>. doi: <pub-id pub-id-type="doi">10.1186/s12913-019-4031-3</pub-id></citation>
</ref>
<ref id="ref27">
<label>27.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Tetteh</surname>
<given-names>JK</given-names>
</name> <name>
<surname>Frimpong</surname>
<given-names>JB</given-names>
</name> <name>
<surname>Budu</surname>
<given-names>E</given-names>
</name> <name>
<surname>Adu</surname>
<given-names>C</given-names>
</name> <name>
<surname>Mohammed</surname>
<given-names>A</given-names>
</name> <name>
<surname>Ahinkorah</surname>
<given-names>BO</given-names>
</name> <etal/></person-group>. <article-title>Comprehensive HIV/AIDS knowledge and HIV testing among men in sub-Saharan Africa: a multilevel modelling</article-title>. <source>J Biosoc Sci</source>. (<year>2022</year>) <volume>54</volume>:<fpage>975</fpage>&#x2013;<lpage>90</lpage>. doi: <pub-id pub-id-type="doi">10.1017/S0021932021000560</pub-id>, PMID: <pub-id pub-id-type="pmid">34736542</pub-id></citation>
</ref>
<ref id="ref28">
<label>28.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Hensen</surname>
<given-names>B</given-names>
</name> <name>
<surname>Lewis</surname>
<given-names>J</given-names>
</name> <name>
<surname>Schaap</surname>
<given-names>A</given-names>
</name> <name>
<surname>Tembo</surname>
<given-names>M</given-names>
</name> <name>
<surname>Vera-Hern&#x00E1;ndez</surname>
<given-names>M</given-names>
</name> <name>
<surname>Mutale</surname>
<given-names>W</given-names>
</name> <etal/></person-group>. <article-title>Frequency of HIV-testing and factors associated with multiple lifetime HIV-testing among a rural population of Zambian men</article-title>. <source>BMC Public Health</source>. (<year>2015</year>) <volume>15</volume>:<fpage>1</fpage>&#x2013;<lpage>14</lpage>. doi: <pub-id pub-id-type="doi">10.1186/s12889-015-2259-3</pub-id></citation>
</ref>
<ref id="ref29">
<label>29.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Kabeta</surname>
<given-names>T</given-names>
</name> <name>
<surname>Belina</surname>
<given-names>M</given-names>
</name> <name>
<surname>Nigatu</surname>
<given-names>M</given-names>
</name></person-group>. <article-title>HIV voluntary counseling and testing uptake and associated factors among sexually active men in Ethiopia: analysis of the 2016 Ethiopian demographic and health survey data</article-title>. <source>HIV AIDS Res. Palliat. Care</source>. (<year>2020</year>) <volume>12</volume>:<fpage>351</fpage>&#x2013;<lpage>62</lpage>. doi: <pub-id pub-id-type="doi">10.2147/HIV.S263851</pub-id></citation>
</ref>
<ref id="ref30">
<label>30.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Conserve</surname>
<given-names>D</given-names>
</name> <name>
<surname>Sevilla</surname>
<given-names>L</given-names>
</name> <name>
<surname>Mbwambo</surname>
<given-names>J</given-names>
</name> <name>
<surname>King</surname>
<given-names>G</given-names>
</name></person-group>. <article-title>Determinants of previous HIV testing and knowledge of partner&#x2019;s HIV status among men attending a voluntary counseling and testing clinic in Dar Es Salaam</article-title>. <source>Tanzania Ame. J. Mens Health</source>. (<year>2013</year>) <volume>7</volume>:<fpage>450</fpage>&#x2013;<lpage>60</lpage>. doi: <pub-id pub-id-type="doi">10.1177/1557988312468146</pub-id>, PMID: <pub-id pub-id-type="pmid">23221684</pub-id></citation>
</ref>
<ref id="ref31">
<label>31.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Molla</surname>
<given-names>G</given-names>
</name> <name>
<surname>Huruy</surname>
<given-names>A</given-names>
</name> <name>
<surname>Mussie</surname>
<given-names>A</given-names>
</name> <name>
<surname>Wondowosen</surname>
<given-names>T</given-names>
</name></person-group>. <article-title>Factors associated with HIV counseling and testing among males and females in Ethiopia: evidence from Ethiopian demographic and health survey data</article-title>. <source>J AIDS Clin Res</source>. (<year>2015</year>) <volume>6</volume>:<fpage>429</fpage>. doi: <pub-id pub-id-type="doi">10.4172/2155-6113.1000429</pub-id></citation>
</ref>
<ref id="ref32">
<label>32.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Nabukenya</surname>
<given-names>AM</given-names>
</name> <name>
<surname>Matovu</surname>
<given-names>JK</given-names>
</name></person-group>. <article-title>Correlates of HIV status awareness among older adults in Uganda: results from a nationally representative survey</article-title>. <source>BMC Public Health</source>. (<year>2018</year>) <volume>18</volume>:<fpage>1</fpage>&#x2013;<lpage>8</lpage>. doi: <pub-id pub-id-type="doi">10.1186/s12889-018-6027-z</pub-id></citation>
</ref>
<ref id="ref33">
<label>33.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Eng</surname>
<given-names>CW</given-names>
</name> <name>
<surname>Tuot</surname>
<given-names>S</given-names>
</name> <name>
<surname>Chann</surname>
<given-names>N</given-names>
</name> <name>
<surname>Chhoun</surname>
<given-names>P</given-names>
</name> <name>
<surname>Mun</surname>
<given-names>P</given-names>
</name> <name>
<surname>Yi</surname>
<given-names>S</given-names>
</name></person-group>. <article-title>Recent HIV testing and associated factors among people who use drugs in Cambodia: a national cross-sectional study</article-title>. <source>BMJ Open</source>. (<year>2021</year>) <volume>11</volume>:<fpage>e045282</fpage>. doi: <pub-id pub-id-type="doi">10.1136/bmjopen-2020-045282</pub-id></citation>
</ref>
<ref id="ref34">
<label>34.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Schonlau</surname>
<given-names>M</given-names>
</name> <name>
<surname>Zou</surname>
<given-names>RY</given-names>
</name></person-group>. <article-title>The random forest algorithm for statistical learning</article-title>. <source>Stata J</source>. (<year>2020</year>) <volume>20</volume>:<fpage>3</fpage>&#x2013;<lpage>29</lpage>. doi: <pub-id pub-id-type="doi">10.1177/1536867X20909688</pub-id></citation>
</ref>
<ref id="ref35">
<label>35.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Chawla</surname>
<given-names>NV</given-names>
</name> <name>
<surname>Bowyer</surname>
<given-names>KW</given-names>
</name> <name>
<surname>Hall</surname>
<given-names>LO</given-names>
</name> <name>
<surname>Kegelmeyer</surname>
<given-names>WP</given-names>
</name></person-group>. <article-title>SMOTE: synthetic minority over-sampling technique</article-title>. <source>J Artif Intell Res</source>. (<year>2002</year>) <volume>16</volume>:<fpage>321</fpage>&#x2013;<lpage>57</lpage>. doi: <pub-id pub-id-type="doi">10.1613/jair.953</pub-id></citation>
</ref>
<ref id="ref36">
<label>36.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Desalegn</surname>
<given-names>M</given-names>
</name> <name>
<surname>Seyoum</surname>
<given-names>D</given-names>
</name> <name>
<surname>Tola</surname>
<given-names>EK</given-names>
</name> <name>
<surname>Tsegaye</surname>
<given-names>GR</given-names>
</name></person-group>. <article-title>Determinants of first-line antiretroviral treatment failure among adult HIV patients at Nekemte specialized hospital, Western Ethiopia: unmatched case-control study</article-title>. <source>SAGE Open Med</source>. (<year>2021</year>) <volume>9</volume>:<fpage>205031212110301</fpage>. doi: <pub-id pub-id-type="doi">10.1177/20503121211030182</pub-id></citation>
</ref>
<ref id="ref37">
<label>37.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Wang</surname>
<given-names>S</given-names>
</name> <name>
<surname>Dai</surname>
<given-names>Y</given-names>
</name> <name>
<surname>Shen</surname>
<given-names>J</given-names>
</name> <name>
<surname>Xuan</surname>
<given-names>J</given-names>
</name></person-group>. <article-title>Research on expansion and classification of imbalanced data based on SMOTE algorithm</article-title>. <source>Sci Rep</source>. (<year>2021</year>) <volume>11</volume>:<fpage>24039</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41598-021-03430-5</pub-id></citation>
</ref>
<ref id="ref38">
<label>38.</label>
<citation citation-type="other"><person-group person-group-type="editor"><name><surname>He</surname> <given-names>H</given-names></name> <name><surname>Bai</surname> <given-names>Y</given-names></name> <name><surname>Garcia</surname> <given-names>EA</given-names></name> <name><surname>Li</surname> <given-names>S</given-names></name></person-group>, (Eds.) ADASYN: Adaptive Synthetic Sampling Approach for Imbalanced Learning. 2008 IEEE International Joint Conference on Neural Networks (IEEE World Congress on Computational Intelligence); IEEE, (<year>2008</year>).</citation>
</ref>
<ref id="ref39">
<label>39.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Kaur</surname>
<given-names>H</given-names>
</name> <name>
<surname>Pannu</surname>
<given-names>HS</given-names>
</name> <name>
<surname>Malhi</surname>
<given-names>AK</given-names>
</name></person-group>. <article-title>A systematic review on imbalanced data challenges in machine learning: applications and solutions</article-title>. <source>ACM Comput Surv</source>. (<year>2019</year>) <volume>52</volume>:<fpage>1</fpage>&#x2013;<lpage>36</lpage>. doi: <pub-id pub-id-type="doi">10.1145/3343440</pub-id></citation>
</ref>
<ref id="ref40">
<label>40.</label>
<citation citation-type="other"><person-group person-group-type="author"><name>
<surname>Saxena</surname>
<given-names>A</given-names>
</name> <name>
<surname>Ganguly</surname>
<given-names>A</given-names>
</name> <name>
<surname>Shrivastava</surname>
<given-names>AK</given-names>
</name></person-group>. <source>Predicting Chronic Kidney Disease Risk Using Recursive Feature Elimination and Machine Learning</source>. (<year>2020</year>) <volume>4</volume>:<fpage>129</fpage>&#x2013;<lpage>32</lpage>.</citation>
</ref>
<ref id="ref41">
<label>41.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Maheu-Giroux</surname>
<given-names>M</given-names>
</name> <name>
<surname>Marsh</surname>
<given-names>K</given-names>
</name> <name>
<surname>Doyle</surname>
<given-names>CM</given-names>
</name> <name>
<surname>Godin</surname>
<given-names>A</given-names>
</name> <name>
<surname>Delaunay</surname>
<given-names>CL</given-names>
</name> <name>
<surname>Johnson</surname>
<given-names>LF</given-names>
</name> <etal/></person-group>. <article-title>National HIV testing and diagnosis coverage in sub-Saharan Africa: a new modeling tool for estimating the &#x2018;first 90&#x2019;from program and survey data</article-title>. <source>AIDS (London, England)</source>. (<year>2019</year>) <volume>33</volume>:<fpage>S255</fpage>&#x2013;<lpage>69</lpage>. doi: <pub-id pub-id-type="doi">10.1097/QAD.0000000000002386</pub-id>, PMID: <pub-id pub-id-type="pmid">31764066</pub-id></citation>
</ref>
<ref id="ref42">
<label>42.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Kidman</surname>
<given-names>R</given-names>
</name> <name>
<surname>Waidler</surname>
<given-names>J</given-names>
</name> <name>
<surname>Palermo</surname>
<given-names>T</given-names>
</name></person-group>. <article-title>Uptake of HIV testing among adolescents and associated adolescent-friendly services</article-title>. <source>BMC Health Serv Res</source>. (<year>2020</year>) <volume>20</volume>:<fpage>1</fpage>&#x2013;<lpage>10</lpage>. doi: <pub-id pub-id-type="doi">10.1186/s12913-020-05731-3</pub-id></citation>
</ref>
<ref id="ref43">
<label>43.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Mutai</surname>
<given-names>CK</given-names>
</name> <name>
<surname>McSharry</surname>
<given-names>PE</given-names>
</name> <name>
<surname>Ngaruye</surname>
<given-names>I</given-names>
</name> <name>
<surname>Musabanganji</surname>
<given-names>E</given-names>
</name></person-group>. <article-title>Use of machine learning techniques to identify HIV predictors for screening in sub-Saharan Africa</article-title>. <source>BMC Med Res Methodol</source>. (<year>2021</year>) <volume>21</volume>:<fpage>1</fpage>&#x2013;<lpage>11</lpage>. doi: <pub-id pub-id-type="doi">10.1186/s12874-021-01346-2</pub-id></citation>
</ref>
<ref id="ref44">
<label>44.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Myint</surname>
<given-names>WW</given-names>
</name> <name>
<surname>Washburn</surname>
<given-names>DJ</given-names>
</name> <name>
<surname>Colwell</surname>
<given-names>B</given-names>
</name> <name>
<surname>Maddock</surname>
<given-names>JE</given-names>
</name></person-group>. <article-title>Determinants of HIV testing uptake among women (aged 15-49 years) in the Philippines, Myanmar, and Cambodia</article-title>. <source>Int J Maternal Child Health AIDS</source>. (<year>2021</year>) <volume>10</volume>:<fpage>221</fpage>&#x2013;<lpage>30</lpage>. doi: <pub-id pub-id-type="doi">10.21106/ijma.525</pub-id></citation>
</ref>
<ref id="ref45">
<label>45.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Zegeye</surname>
<given-names>B</given-names>
</name> <name>
<surname>Adjei</surname>
<given-names>NK</given-names>
</name> <name>
<surname>Ahinkorah</surname>
<given-names>BO</given-names>
</name> <name>
<surname>Tesema</surname>
<given-names>GA</given-names>
</name> <name>
<surname>Ameyaw</surname>
<given-names>EK</given-names>
</name> <name>
<surname>Budu</surname>
<given-names>E</given-names>
</name> <etal/></person-group>. <article-title>HIV testing among women of reproductive age in 28 sub-Saharan African countries: a multilevel modelling</article-title>. <source>Int Health</source>. (<year>2023</year>) <volume>15</volume>:<fpage>573</fpage>&#x2013;<lpage>84</lpage>. doi: <pub-id pub-id-type="doi">10.1093/inthealth/ihad031</pub-id></citation>
</ref>
<ref id="ref46">
<label>46.</label>
<citation citation-type="other"><person-group person-group-type="editor"><name><surname>Agrawal</surname> <given-names>R</given-names></name> <name><surname>Imieli&#x0144;ski</surname> <given-names>T</given-names></name> <name><surname>Swami</surname> <given-names>A</given-names></name></person-group>, (Eds.) Mining Association Rules Between Sets of Items in Large Databases. Proceedings of the 1993 ACM SIGMOD International Conference on Management of Data; (<year>1993</year>).</citation>
</ref>
<ref id="ref47">
<label>47.</label>
<citation citation-type="other"><person-group person-group-type="editor"><name><surname>Harahap</surname> <given-names>M</given-names></name> <name><surname>Husein</surname> <given-names>A</given-names></name> <name><surname>Aisyah</surname> <given-names>S</given-names></name> <name><surname>Lubis</surname> <given-names>F</given-names></name> <name><surname>Wijaya</surname> <given-names>B</given-names></name></person-group>, (Eds.) Mining Association Rule Based on the Diseases Population for Recommendation of Medicine Need. Journal of Physics: Conference Series. IOP Publishing; (<year>2018</year>).</citation>
</ref>
<ref id="ref48">
<label>48.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Altaf</surname>
<given-names>W</given-names>
</name> <name>
<surname>Shahbaz</surname>
<given-names>M</given-names>
</name> <name>
<surname>Guergachi</surname>
<given-names>A</given-names>
</name></person-group>. <article-title>Applications of association rule mining in health informatics: a survey</article-title>. <source>Artif Intell Rev</source>. (<year>2017</year>) <volume>47</volume>:<fpage>313</fpage>&#x2013;<lpage>40</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10462-016-9483-9</pub-id></citation>
</ref>
<ref id="ref49">
<label>49.</label>
<citation citation-type="other"><person-group person-group-type="editor"><name><surname>Khare</surname> <given-names>S</given-names></name> <name><surname>Gupta</surname> <given-names>D</given-names></name></person-group>, (Eds.) Association rule Analysis in Cardiovascular Disease. 2016 Second International Conference on Cognitive Computing and Information Processing (CCIP). IEEE; (<year>2016</year>).</citation>
</ref>
<ref id="ref50">
<label>50.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>He</surname>
<given-names>J</given-names>
</name> <name>
<surname>Li</surname>
<given-names>J</given-names>
</name> <name>
<surname>Jiang</surname>
<given-names>S</given-names>
</name> <name>
<surname>Cheng</surname>
<given-names>W</given-names>
</name> <name>
<surname>Jiang</surname>
<given-names>J</given-names>
</name> <name>
<surname>Xu</surname>
<given-names>Y</given-names>
</name> <etal/></person-group>. <article-title>Application of machine learning algorithms in predicting HIV infection among men who have sex with men: model development and validation</article-title>. <source>Front Public Health</source>. (<year>2022</year>) <volume>10</volume>:<fpage>967681</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fpubh.2022.967681</pub-id>, PMID: <pub-id pub-id-type="pmid">36091522</pub-id></citation>
</ref>
<ref id="ref51">
<label>51.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Wang</surname>
<given-names>B</given-names>
</name> <name>
<surname>Liua</surname>
<given-names>F</given-names>
</name> <name>
<surname>Deveaux</surname>
<given-names>L</given-names>
</name> <name>
<surname>Ash</surname>
<given-names>A</given-names>
</name> <name>
<surname>Gosh</surname>
<given-names>S</given-names>
</name> <name>
<surname>Li</surname>
<given-names>X</given-names>
</name> <etal/></person-group>. <article-title>Adolescent HIV-related behavioural prediction using machine learning: a foundation for precision HIV prevention</article-title>. <source>AIDS (London, England)</source>. (<year>2021</year>) <volume>35</volume>:<fpage>S75</fpage>&#x2013;<lpage>84</lpage>. doi: <pub-id pub-id-type="doi">10.1097/QAD.0000000000002867</pub-id>, PMID: <pub-id pub-id-type="pmid">33867490</pub-id></citation>
</ref>
<ref id="ref52">
<label>52.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Xu</surname>
<given-names>X</given-names>
</name> <name>
<surname>Fairley</surname>
<given-names>CK</given-names>
</name> <name>
<surname>Chow</surname>
<given-names>EP</given-names>
</name> <name>
<surname>Lee</surname>
<given-names>D</given-names>
</name> <name>
<surname>Aung</surname>
<given-names>ET</given-names>
</name> <name>
<surname>Zhang</surname>
<given-names>L</given-names>
</name> <etal/></person-group>. <article-title>Using machine learning approaches to predict timely clinic attendance and the uptake of HIV/STI testing post clinic reminder messages</article-title>. <source>Sci Rep</source>. (<year>2022</year>) <volume>12</volume>:<fpage>8757</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41598-022-12033-7</pub-id>, PMID: <pub-id pub-id-type="pmid">35610227</pub-id></citation>
</ref>
<ref id="ref53">
<label>53.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Orel</surname>
<given-names>E</given-names>
</name> <name>
<surname>Esra</surname>
<given-names>R</given-names>
</name> <name>
<surname>Estill</surname>
<given-names>J</given-names>
</name> <name>
<surname>Thiabaud</surname>
<given-names>A</given-names>
</name> <name>
<surname>Marchand-Maillet</surname>
<given-names>S</given-names>
</name> <name>
<surname>Merzouki</surname>
<given-names>A</given-names>
</name> <etal/></person-group>. <article-title>Prediction of HIV status based on socio-behavioural characteristics in east and southern Africa</article-title>. <source>PLoS One</source>. (<year>2022</year>) <volume>17</volume>:<fpage>e0264429</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0264429</pub-id>, PMID: <pub-id pub-id-type="pmid">35239697</pub-id></citation>
</ref>
<ref id="ref54">
<label>54.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Suthar</surname>
<given-names>AB</given-names>
</name> <name>
<surname>Ford</surname>
<given-names>N</given-names>
</name> <name>
<surname>Bachanas</surname>
<given-names>PJ</given-names>
</name> <name>
<surname>Wong</surname>
<given-names>VJ</given-names>
</name> <name>
<surname>Rajan</surname>
<given-names>JS</given-names>
</name> <name>
<surname>Saltzman</surname>
<given-names>AK</given-names>
</name> <etal/></person-group>. <article-title>Towards universal voluntary HIV testing and counselling: a systematic review and meta-analysis of community-based approaches</article-title>. <source>PLoS Med</source>. (<year>2013</year>) <volume>10</volume>:<fpage>e1001496</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pmed.1001496</pub-id>, PMID: <pub-id pub-id-type="pmid">23966838</pub-id></citation>
</ref>
<ref id="ref55">
<label>55.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Orel</surname>
<given-names>E</given-names>
</name> <name>
<surname>Esra</surname>
<given-names>R</given-names>
</name> <name>
<surname>Estill</surname>
<given-names>J</given-names>
</name> <name>
<surname>Marchand-Maillet</surname>
<given-names>S</given-names>
</name> <name>
<surname>Merzouki</surname>
<given-names>A</given-names>
</name> <name>
<surname>Keiser</surname>
<given-names>O</given-names>
</name></person-group>. <article-title>Machine learning to identify socio-behavioural predictors of HIV positivity in east and southern Africa</article-title>. <source>medRxiv</source>. (<year>2020</year>) <volume>2020</volume>:<fpage>18242</fpage>. doi: <pub-id pub-id-type="doi">10.1101/2020.01.27.20018242</pub-id></citation>
</ref>
<ref id="ref56">
<label>56.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Yazdanpanah</surname>
<given-names>Y</given-names>
</name> <name>
<surname>Sloan</surname>
<given-names>CE</given-names>
</name> <name>
<surname>Charlois-Ou</surname>
<given-names>C</given-names>
</name> <name>
<surname>Le Vu</surname>
<given-names>S</given-names>
</name> <name>
<surname>Semaille</surname>
<given-names>C</given-names>
</name> <name>
<surname>Costagliola</surname>
<given-names>D</given-names>
</name> <etal/></person-group>. <article-title>Routine HIV screening in France: clinical impact and cost-effectiveness</article-title>. <source>PLoS One</source>. (<year>2010</year>) <volume>5</volume>:<fpage>e13132</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0013132</pub-id>, PMID: <pub-id pub-id-type="pmid">20976112</pub-id></citation>
</ref>
<ref id="ref57">
<label>57.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Sullivan</surname>
<given-names>AK</given-names>
</name> <name>
<surname>Raben</surname>
<given-names>D</given-names>
</name> <name>
<surname>Reekie</surname>
<given-names>J</given-names>
</name> <name>
<surname>Rayment</surname>
<given-names>M</given-names>
</name> <name>
<surname>Mocroft</surname>
<given-names>A</given-names>
</name> <name>
<surname>Esser</surname>
<given-names>S</given-names>
</name> <etal/></person-group>. <article-title>Feasibility and effectiveness of indicator condition-guided testing for HIV: results from HIDES I (HIV indicator diseases across Europe study)</article-title>. <source>PLoS One</source>. (<year>2013</year>) <volume>8</volume>:<fpage>e52845</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0052845</pub-id>, PMID: <pub-id pub-id-type="pmid">23341910</pub-id></citation>
</ref>
<ref id="ref58">
<label>58.</label>
<citation citation-type="book"><person-group person-group-type="author">
<collab id="coll7">World Health Organization</collab>
</person-group>. <source>Differentiated and Simplified Pre-Exposure Prophylaxis for HIV Prevention: Update to WHO Implementation Guidance: Technical Brief</source>. <publisher-loc>Geneva, Switzerland</publisher-loc>: <publisher-name>World Health Organization</publisher-name> (<year>2022</year>).</citation>
</ref>
<ref id="ref59">
<label>59.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Kagaayi</surname>
<given-names>J</given-names>
</name> <name>
<surname>Gray</surname>
<given-names>RH</given-names>
</name> <name>
<surname>Whalen</surname>
<given-names>C</given-names>
</name> <name>
<surname>Fu</surname>
<given-names>P</given-names>
</name> <name>
<surname>Neuhauser</surname>
<given-names>D</given-names>
</name> <name>
<surname>McGrath</surname>
<given-names>JW</given-names>
</name> <etal/></person-group>. <article-title>Indices to measure risk of HIV acquisition in Rakai, Uganda</article-title>. <source>PLoS One</source>. (<year>2014</year>) <volume>9</volume>:<fpage>e92015</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0092015</pub-id>, PMID: <pub-id pub-id-type="pmid">24704778</pub-id></citation>
</ref>
<ref id="ref60">
<label>60.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name>
<surname>Cambiano</surname>
<given-names>V</given-names>
</name> <name>
<surname>Miners</surname>
<given-names>A</given-names>
</name> <name>
<surname>Phillips</surname>
<given-names>A</given-names>
</name></person-group>. <article-title>What do we know about the cost&#x2013;effectiveness of HIV preexposure prophylaxis, and is it affordable?</article-title> <source>Curr Opin HIV AIDS</source>. (<year>2016</year>) <volume>11</volume>:<fpage>56</fpage>&#x2013;<lpage>66</lpage>. doi: <pub-id pub-id-type="doi">10.1097/COH.0000000000000217</pub-id>, PMID: <pub-id pub-id-type="pmid">26569182</pub-id></citation>
</ref>
<ref id="ref61">
<label>61.</label>
<citation citation-type="book"><person-group person-group-type="author">
<collab id="coll8">OGWENO OI</collab>
</person-group>. <source>Influence of School-Based Sexual Risk Avoidance Education on Sexual Behavior among Adolescent Girls in HOMABAY County</source>. <publisher-loc>KENYA</publisher-loc>: <publisher-name>Kenyatta University</publisher-name> (<year>2020</year>).</citation>
</ref>
<ref id="ref62">
<label>62.</label>
<citation citation-type="book"><person-group person-group-type="author">
<collab id="coll9">World Health Organization</collab>
</person-group>. <source>Guideline on When to Start Antiretroviral Therapy and On Pre-Exposure Prophylaxis for HIV</source>. <publisher-loc>Geneva, Switzerland</publisher-loc>: <publisher-name>World Health Organization</publisher-name> (<year>2015</year>).</citation>
</ref>
</ref-list>
</back>
</article>