<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Bioinform.</journal-id>
<journal-title>Frontiers in Bioinformatics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Bioinform.</abbrev-journal-title>
<issn pub-type="epub">2673-7647</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1522401</article-id>
<article-id pub-id-type="doi">10.3389/fbinf.2025.1522401</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Bioinformatics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Biomarker-driven drug repurposing for NAFLD-associated hepatocellular carcinoma using machine learning integrated ensemble feature selection</article-title>
<alt-title alt-title-type="left-running-head">Ghosh et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fbinf.2025.1522401">10.3389/fbinf.2025.1522401</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Ghosh</surname>
<given-names>Subhajit</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2864901/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Mandal</surname>
<given-names>Sukhen Das</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2918915/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Thakur</surname>
<given-names>Subarna</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2363001/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Bioinformatics</institution>, <institution>University of North Bengal</institution>, <addr-line>Darjeeling</addr-line>, <addr-line>West Bengal</addr-line>, <country>India</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Computer Science and Engineering</institution>, <institution>Ghani Khan Choudhury Institute of Engineering and Technology (GKCIET)</institution>, <addr-line>Malda</addr-line>, <country>India</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/147399/overview">Dharmendra Kumar Yadav</ext-link>, Gachon University, Republic of Korea</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2253852/overview">Bhavika Mam</ext-link>, Independent Researcher, Palo Alto, CA, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/999203/overview">Nam N. Phan</ext-link>, The University of Texas Health Science Center at San Antonio, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Subarna Thakur, <email>subarna.thakur@nbu.ac.in</email>; Sukhen Das Mandal, <email>sukhen@gkciet.ac.in</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>17</day>
<month>04</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>5</volume>
<elocation-id>1522401</elocation-id>
<history>
<date date-type="received">
<day>28</day>
<month>11</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>04</day>
<month>04</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Ghosh, Mandal and Thakur.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Ghosh, Mandal and Thakur</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>The incidence of non-alcoholic fatty liver disease (NAFLD), encompassing the more severe non-alcoholic steatohepatitis (NASH), is rising alongside the surges in diabetes and obesity. Increasing evidence indicates that NASH is responsible for a significant share of idiopathic hepatocellular carcinoma (HCC) cases, a fatal cancer with a 5-year survival rate below 22%. Biomarkers can facilitate early screening and monitoring of at-risk NAFLD/NASH patients and assist in identifying potential drug candidates for treatment. This study utilized an ensemble feature selection framework to analyze transcriptomic data, identifying biomarker genes associated with the stage-wise progression of NAFLD-related HCC. Seven machine learning algorithms were assessed for disease stage classification. Twelve feature selection methods including correlation-based techniques, mutual information-based methods, and embedded techniques were utilized to rank the top genes as features, through this approach, multiple feature selection methods were combined to yield more robust features important in this disease progression. Cox regression-based survival analysis was carried out to evaluate the biomarker potentiality of these genes. Furthermore, multiphase drug repurposing strategy and molecular docking were employed to identify potential drug candidates against these biomarkers. Among the seven machine learning models initially evaluated, DISCR resulted as the most accurate disease stage classifier. Ensemble feature selection identified ten top genes, among which eight were recognized as potential biomarkers based on survival analysis. These include genes ABAT, ABCB11, MBTPS1, and ZFP1 mostly involved in alanine and glutamate metabolism, butanoate metabolism, and ER protein processing. Through drug repurposing, 81 candidate drugs were found to be effective against these markers genes, with Diosmin, Esculin, Lapatinib, and Phenelzine as the best candidates screened through molecular docking and MMGBSA. The consensus derived from multiple methods enhances the accuracy of identifying relevant robust biomarkers for NAFLD-associated HCC. The use of these biomarkers in a multiphase drug repurposing strategy highlights potential therapeutic options for early intervention, which is essential to stop disease progression and improve outcomes.</p>
</abstract>
<kwd-group>
<kwd>NAFLD</kwd>
<kwd>HCC</kwd>
<kwd>machine learning</kwd>
<kwd>ensemble feature selection</kwd>
<kwd>drug repurposing</kwd>
<kwd>molecular docking</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Drug Discovery in Bioinformatics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>The alarming rise in the incidence of non-alcoholic fatty liver disease (NAFLD), triggered by obesity and the type 2 diabetes mellitus (T2D) epidemic, has increased the concerns within the healthcare system (<xref ref-type="bibr" rid="B68">Teng et al., 2023</xref>). NAFLD can progress to non-alcoholic steatohepatitis (NASH), liver fibrosis and cirrhosis, or even hepatocellular carcinoma (HCC), the most common form of primary liver cancer. HCC is associated with a poor prognosis, partly because it is often diagnosed at a late stage. Furthermore, the lack of approved pharmaceutical treatments specifically targeting NAFLD-induced HCC (<xref ref-type="bibr" rid="B76">Zhang and Yang, 2021</xref>) necessitates the urgent need for targeted therapeutic interventions. High-throughput omics technology has generated extensive gene expression datasets, allowing diverse new approaches to improve analysis and interpretation (<xref ref-type="bibr" rid="B16">Chen C. et al., 2023</xref>). The transition from NAFLD to HCC follows a prolonged timeline of 5&#x2013;15 years, posing difficulties in conducting patient research (<xref ref-type="bibr" rid="B65">Stra&#x15b; et al., 2020</xref>). This challenge can be addressed by integrating data from various samples, though identifying key regulators of disease progression remains difficult. Conventional methods analyze disease stages separately, but merging these results often leads to issues like overfitting, technical noise, and reduced robustness (<xref ref-type="bibr" rid="B58">Posekany et al., 2011</xref>). These challenges can significantly impact the reliability and generalizability of the results. Machine learning-based feature selection has recently emerged as a solution to this problem. Recently, a machine learning (ML)-based approach has been applied to breast cancer prediction and classification by the detection of malignant cells using models like XGBoost, logistic regression, K-nearest neighbor, etc., (<xref ref-type="bibr" rid="B17">Chen H. et al., 2023</xref>). This kind of method holds great potential for early diagnosis of cancer. The results of the ML-based studies using imaging, such as lung cancer CT scans, showed promising results in identifying cancer subtypes (<xref ref-type="bibr" rid="B51">Nazir et al., 2023</xref>). Biomarker selection for early detection using gene expression data can be accomplished through robust feature selection methods, which have recently been applied in breast cancer (<xref ref-type="bibr" rid="B60">Sarkar et al., 2021</xref>), gastric cancer (<xref ref-type="bibr" rid="B1">Azari et al., 2023</xref>), lung and colon cancers (<xref ref-type="bibr" rid="B66">Talukder et al., 2022</xref>), etc., In gene expression analysis, effective feature selection techniques can pinpoint the most relevant and unique genes or molecular characteristics (<xref ref-type="bibr" rid="B2">Barrera et al., 2007</xref>), which facilitates the development of robust and easily interpretable gene signature models. Ensemble feature selection is a recently introduced approach, that enhances the robustness and accuracy of selected features by combining the results of multiple feature selection methods (<xref ref-type="bibr" rid="B2">Barrera et al., 2007</xref>; <xref ref-type="bibr" rid="B10">Bol&#xf3;n-Canedo and Alonso-Betanzos, 2019</xref>). In the last few years, this approach has been found to be effective in cancer gene expression data analysis, and in the identification of key genes as the most relevant features (<xref ref-type="bibr" rid="B37">Koul and Manvi, 2020</xref>; <xref ref-type="bibr" rid="B35">Khatun et al., 2023</xref>). In this study, after selecting and preprocessing microarray datasets, seven machine-learning techniques such as DISCR (Discriminant Analysis), NB (Naive Bayes), RF (Random Forest), DT (Decision Tree), KNN (K-Nearest Neighbors), SVM (Support Vector Machine), and ANN (Artificial Neural Network) were applied for disease stage classification. The method with the highest classification accuracy was selected using a 10-fold cross-validation for further analysis. Next, twelve feature selection strategies were utilized to identify the most significant genes. Among these twelve techniques, CIFE (Conditional Informative Feature Extraction), JMI(Joint Mutual Information), and MIM(Mutual information maximization) were based on mutual information, which selects features based on their relevance and dependency on the target variable, aiming to capture the most informative attributes for classification or analysis purposes (<xref ref-type="bibr" rid="B26">Gao et al., 2018</xref>). The Kendall Tau, Pearson, and Spearman methods selected the features based on their correlation or association with the target variable, focusing on measuring the strength and direction of the relationship between variables (<xref ref-type="bibr" rid="B23">El-Hashash and Shiekh, 2022</xref>). The other embedded methods, LASSO (Least Absolute Shrinkage and Selection Operator), Ridge, and gradient boosting utilize regularization techniques to penalize the complexity of the model, encouraging simpler models that generalize well to new data (<xref ref-type="bibr" rid="B43">Li Y. et al., 2022</xref>; <xref ref-type="bibr" rid="B7">Bhandari et al., 2022</xref>). These methods aim to prevent overfitting by constraining the coefficients of the features, thus promoting sparsity and improving the model&#x2019;s interpretability and predictive performance. This study enhances feature selection, model regularization, and generalization by integrating various methods. This ensemble approach leverages diverse insights enhancing robustness and stability while revealing complex data patterns. (<xref ref-type="bibr" rid="B14">Castellanos-Garz&#xf3;n et al., 2017</xref>). This approach of feature selection helps in identifying the key genes as features that are subsequently evaluated for their prognostic potential as biomarkers in HCC applying the Cox proportional hazards model (<xref ref-type="bibr" rid="B47">Mohammed et al., 2021</xref>). Furthermore, it incorporates a biomarker-driven drug repurposing approach using identified biomarker genes as targets to screen out suitable drug candidates from the library of existing drugs. Drug repurposing has gained attention identification of novel uses of existing drugs, either through new combinations or in the treatment of different diseases (<xref ref-type="bibr" rid="B38">Krishnamurthy et al., 2022</xref>). Potential drugs were identified using connectivity map analysis, text mining, drug-gene interaction data, etc. These drug candidates were further assessed through molecular docking to evaluate their binding affinity with target proteins and to explore the therapeutic potential of these existing drugs against new targets.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<sec id="s2-1">
<title>2.1 Data collection and preparation</title>
<p>The microarray data and clinical information for Control, Healthy obese, NAFLD, NASH, and HCC samples were obtained from the NCBI Gene Expression Omnibus (GEO) (<xref ref-type="bibr" rid="B3">Barrett et al., 2013</xref>) and Array Express (<xref ref-type="bibr" rid="B54">Parkinson et al., 2007</xref>) database. This dataset comprised a total of 132 samples with GEO accession numbers: GSE48452 with platform ID - GPL11532 (Affymetrix Human Gene 1.1 ST Array); GSE25097 with platform ID - GPL10687 (Rosetta/Merck Human RSTA Affymetrix 1.0 microarray). Details of datasets and respective sample information were provided as <xref ref-type="sec" rid="s13">Supplementary Material S1</xref>. Following the merging of data, the R package &#x201c;imputeTS&#x201d; (version 3.3) (<xref ref-type="bibr" rid="B48">Moritz and Bartz-Beielstein, 2017</xref>) was utilized to impute missing values. This imputation process was only carried out on less than 5% of randomly missing values. Afterward, the &#x201c;Limma&#x201d; (version 3.57.3) package (<xref ref-type="bibr" rid="B59">Ritchie et al., 2015</xref>) was utilized to remove batch effects specific to each study, and data normalization was performed utilizing the robust multichip averaging (RMA) method (<xref ref-type="bibr" rid="B8">Bioconductor, 2023</xref>). <xref ref-type="fig" rid="F1">Figure 1</xref> presents a schematic diagram representing the methodology.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Schematic diagram depicting methodology.</p>
</caption>
<graphic xlink:href="fbinf-05-1522401-g001.tif"/>
</fig>
</sec>
<sec id="s2-2">
<title>2.2 Disease state classification using machine learning</title>
<p>The entire dataset, encompassing samples from various disease stages ranging from NAFLD to HCC, is utilized to identify a suitable machine-learning technique with the primary objective of effectively categorizing the data into distinct groups based on disease stages. Each of the following machine learning (ML) models is applied individually to the entire pre-processed dataset using 10-fold cross-validation: DISCR (Discriminant Analysis), NB (Naive Bayes), RF (Random Forest), DT (Decision Tree), KNN (K-Nearest Neighbors), SVM (Support Vector Machine), and ANN (Artificial Neural Network). This procedure guarantees a thorough assessment and reduces the risk of overfitting by training and verifying the models on distinct subsets of the data. Through this screening, the ML technique with the highest accuracy values and other pertinent performance indicators is chosen for further investigation.</p>
</sec>
<sec id="s2-3">
<title>2.3 Ensemble feature selection approach for identification of genes involved in disease progression</title>
<p>Twelve different feature selection methods are applied to the entire dataset. Each of these methods individually selects subsets of genes interactively, comprising varying quantities, such as the top 20, 30, 40, 50, and beyond. Subsequently, each subset generated by the different methods is utilized in the chosen machine learning technique, employing 10-fold cross-validation to calculate the classification accuracy. This guarantees that the performance of the chosen characteristics is comprehensively assessed. Furthermore, each feature selection method yields a subset of genes based on the highest level of accuracy attained. This technique enables the identification of the most pertinent genes that make a major contribution to classification, ensuring a strong and dependable selection of features. Among the feature selection techniques used, the ReliefF algorithm evaluates each feature&#x2019;s significance based on its ability to distinguish between similar cases. It iteratively selects a sample, comparing it with the nearest samples from both the same and different classes. Features that effectively differentiate between classes receive higher weights, while those that distinguish within the same class receive lower weights (<xref ref-type="bibr" rid="B70">Urbanowicz et al., 2018</xref>). The process is repeated across multiple instances to reliably estimate feature importance, making ReliefF effective for noisy, multi-class datasets. The basic equation for updating the feature weights in ReliefF is:<disp-formula id="equ1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mtext>diff</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>R</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mtext>Hit</mml:mtext>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x2260;</mml:mo>
<mml:mtext>class</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mfrac>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtext>class</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mtext>diff</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>R</mml:mi>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mtext>Miss</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mi>C</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Where,</p>
<p>W [A] This represents the updated weight or importance score of feature AAA.</p>
<p>
<inline-formula id="inf1">
<mml:math id="m2">
<mml:mrow>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">A</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> on the right-hand side: This term indicates the current weight of feature A.</p>
<p>
<inline-formula id="inf2">
<mml:math id="m3">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mtext>diff</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>R</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mtext>Hit</mml:mtext>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> This part calculates the average difference in feature A values between the Randomly decided on instance R and its K nearest neighbors that belong to the same class as R (denoted as <inline-formula id="inf3">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mtext>Hit</mml:mtext>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>).</p>
<p>Cram&#xe9;r&#x2019;s V is a filter-based technique that measures the correlation between two nominal variables. It involves creating a contingency table to compute the Chi-Squared statistic, which quantifies the strength of the association between the variables (<xref ref-type="bibr" rid="B34">Kearney, 2017</xref>). To account for the bias, the Chi-Squared estimate is reformulated as a Phi-Squared value. The degrees of freedom are subtracted from both the row and column counts to obtain the denominator of the Cram&#xe9;r&#x2019;s V formula. Taking the square root of the quotient resulting from dividing the corrected Phi-Squared value by the denominator yields Cram&#xe9;r&#x2019;s V in the range from 0 for no association up to 1 for the perfect association. This measure helps determine categorical data as it offers an idea of the strength with which variables are correlated without the assumption of any linear relation.</p>
<p>Chi-Squared Statistic <inline-formula id="inf4">
<mml:math id="m5">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3c7;</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>:<disp-formula id="equ2">
<mml:math id="m6">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="normal">&#x3c7;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2211;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">O</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">E</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">E</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf5">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">O</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the observed frequency and <inline-formula id="inf6">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">E</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the expected frequency.</p>
<p>Phi-Squared (<inline-formula id="inf7">
<mml:math id="m9">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="normal">&#x3d5;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>):<disp-formula id="equ3">
<mml:math id="m10">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="normal">&#x3d5;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="normal">&#x3c7;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf8">
<mml:math id="m11">
<mml:mrow>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the total number of observations.</p>
<p>Corrected Phi-Squared <inline-formula id="inf9">
<mml:math id="m12">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="normal">&#x3d5;</mml:mi>
<mml:mtext>corr</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>:<disp-formula id="equ4">
<mml:math id="m13">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="normal">&#x3d5;</mml:mi>
<mml:mtext>corr</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>max</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">&#x3d5;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf10">
<mml:math id="m14">
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf11">
<mml:math id="m15">
<mml:mrow>
<mml:mi mathvariant="normal">k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> are the number of rows and columns in the contingency table.</p>
<p>Corrected Row and Column Counts<disp-formula id="equ5">
<mml:math id="m16">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mtext>corr</mml:mtext>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ6">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:mi>k</mml:mi>
<mml:mtext>corr</mml:mtext>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Denominator:<disp-formula id="equ7">
<mml:math id="m18">
<mml:mrow>
<mml:mtext>denominator</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>min</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>k</mml:mi>
<mml:mtext>corr</mml:mtext>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mtext>corr</mml:mtext>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Cram&#xe9;r&#x2019;s V:<disp-formula id="equ8">
<mml:math id="m19">
<mml:mrow>
<mml:mi>V</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msqrt>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="normal">&#x3d5;</mml:mi>
<mml:mtext>corr</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
<mml:mrow>
<mml:mtext>denominator</mml:mtext>
</mml:mrow>
</mml:mfrac>
</mml:msqrt>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Pearson&#x2019;s correlation coefficient (r) quantifies the strength and direction of a linear relationship between two variables (<xref ref-type="bibr" rid="B50">Nasir et al., 2020</xref>), ranging from &#x2212;1 to &#x2b;1. A value of &#x2b;1 indicates a perfect positive correlation, where both variables increase together; &#x2212;1 signifies a perfect negative correlation, where one variable increases as the other decreases; and 0 denotes no linear correlation.</p>
<p>Pearson&#x2019;s correlation coefficient (<inline-formula id="inf12">
<mml:math id="m20">
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) is defined as:<disp-formula id="equ9">
<mml:math id="m21">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>x</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2211;</mml:mo>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2211;</mml:mo>
<mml:msup>
<mml:mi>y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Where:</p>
<list list-type="simple">
<list-item>
<p>&#x2022; <inline-formula id="inf13">
<mml:math id="m22">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the number of data points.</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf14">
<mml:math id="m23">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf15">
<mml:math id="m24">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> are the two variables being compared.</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf16">
<mml:math id="m25">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>x</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the sum of the product of <inline-formula id="inf17">
<mml:math id="m26">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf18">
<mml:math id="m27">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf19">
<mml:math id="m28">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the sum of <inline-formula id="inf20">
<mml:math id="m29">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf21">
<mml:math id="m30">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the sum of <inline-formula id="inf22">
<mml:math id="m31">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf23">
<mml:math id="m32">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the sum of the squares of <inline-formula id="inf24">
<mml:math id="m33">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf25">
<mml:math id="m34">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
<mml:msup>
<mml:mi>y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the sum of the squares of <inline-formula id="inf26">
<mml:math id="m35">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
</list>
<p>Kendall&#x2019;s Tau is a non-parametric measure that quantifies the ordinal correlation between two variables (<xref ref-type="bibr" rid="B71">Valencia et al., 2019</xref>), focusing on the direction and magnitude of their association. While Pearson&#x2019;s correlation coefficient measures only linear relationships, Kendall&#x2019;s Tau is a measure that is very useful in cases where observations do not meet the assumptions of normality or linearity. It counts concordant and discordant pairs of observations for its calculation. Values range from &#x2212;1 (perfect inverse correlation) up to &#x2b;1 for perfect direct correlation. A value of 0 indicates no association.</p>
<p>Kendall&#x2019;s Tau <inline-formula id="inf27">
<mml:math id="m36">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3c4;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> can be defined as:<disp-formula id="equ10">
<mml:math id="m37">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3c4;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtext>Number</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>of</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>concordant</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>pairs</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtext>Number</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>of</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>discordant</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>pairs</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mi>n</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>2</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Where:</p>
<list list-type="simple">
<list-item>
<p>&#x2022; Concordant pairs: For any two pairs of observations <inline-formula id="inf28">
<mml:math id="m38">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf29">
<mml:math id="m39">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> the pairs are concordant if the order of the elements is the same, i.e., <inline-formula id="inf30">
<mml:math id="m40">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mtext>&#x2009;and&#x2009;</mml:mtext>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mtext>&#x2009;or&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3e;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mtext>&#x2009;and&#x2009;</mml:mtext>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3e;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2022; Discordant pairs: For any two pairs of observations <inline-formula id="inf31">
<mml:math id="m41">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf32">
<mml:math id="m42">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>, the pairs are discordant if the order of the elements is opposite, i.e., <inline-formula id="inf33">
<mml:math id="m43">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mtext>&#x2009;and&#x2009;</mml:mtext>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3e;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mtext>&#x2009;or&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3e;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mtext>&#x2009;and&#x2009;</mml:mtext>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf34">
<mml:math id="m44">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mi>n</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>2</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> is the total number of pairs, calculated as <inline-formula id="inf35">
<mml:math id="m45">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:mfrac>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
</list>
<p>Another method, Spearman&#x2019;s rank correlation coefficient (&#x3c1;) is a statistical measure that quantifies the degree and direction of the relationship between two variables that have been ranked (<xref ref-type="bibr" rid="B62">Schober and Schwarte, 2018</xref>). It measures how well the relationship between two variables can be represented as a monotonic function. This means that the Spearman rank correlation varies from Pearson&#x2019;s correlation, as it considers the variables&#x2019; ranks while trying to figure out their relationship and not the variable&#x2019;s actual values. Hence, it is applicable in assessing ordinal data that deviate from the assumptions of linearity and normalcy.</p>
<p>Spearman&#x2019;s rank correlation coefficient <inline-formula id="inf36">
<mml:math id="m46">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3c1;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> is defined as:<disp-formula id="equ11">
<mml:math id="m47">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3c1;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>6</mml:mn>
<mml:mo>&#x2211;</mml:mo>
<mml:msubsup>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>n</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Where:</p>
<list list-type="simple">
<list-item>
<p>&#x2022; <inline-formula id="inf37">
<mml:math id="m48">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the number of data points.</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf38">
<mml:math id="m49">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the difference between the ranks of corresponding variables.</p>
</list-item>
</list>
<p>On the other hand, Conditional Infomax Feature Extraction (CIFE) selects features by maximizing a score J(S) (<xref ref-type="bibr" rid="B44">Lin and Tang, 2006</xref>). Features are incrementally added to the set <italic>S</italic> based on their significance and redundancy relative to already chosen features. In this way, the algorithm continues till it gets some pre-defined features or threshold scores. CIFE intends to offer an informative feature set that is concise in length, by managing relevance and redundancy towards better facilitation of subsequent ML models.</p>
<p>The CIFE can be expressed as<disp-formula id="equ12">
<mml:math id="m50">
<mml:mrow>
<mml:mi>J</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Where:</p>
<list list-type="simple">
<list-item>
<p>&#x2022; <inline-formula id="inf39">
<mml:math id="m51">
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the set of selected features.</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf40">
<mml:math id="m52">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>;</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the mutual information between feature <inline-formula id="inf41">
<mml:math id="m53">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the target variable <inline-formula id="inf42">
<mml:math id="m54">
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf43">
<mml:math id="m55">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>;</mml:mo>
<mml:mi>X</mml:mi>
<mml:mi>j</mml:mi>
<mml:mo>&#x2223;</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the conditional mutual information between features <inline-formula id="inf44">
<mml:math id="m56">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf45">
<mml:math id="m57">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> given the target variable <inline-formula id="inf46">
<mml:math id="m58">
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf47">
<mml:math id="m59">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is a parameter that controls the trade-off between relevance and redundancy.</p>
</list-item>
</list>
<p>Among mutual information-based feature selection methods, the Joint Mutual Information (JMI) criterion is used to identify the most relevant features for a specific task. JMI selects features based on their high mutual information with the target variable while considering the combined information among selected features (<xref ref-type="bibr" rid="B5">Bennasar et al., 2015</xref>). This way of approaching features ensures that they are individually relevant and also collectively informative to reduce redundancy while enhancing the predictive capability of the model. JMI helps build an efficient and effective feature set, resulting in balancing relevance with redundancy.</p>
<p>The JMI criterion can be defined as:<disp-formula id="equ13">
<mml:math id="m60">
<mml:mrow>
<mml:mi>J</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:mi>Y</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Where:</p>
<list list-type="simple">
<list-item>
<p>&#x2022; <inline-formula id="inf48">
<mml:math id="m61">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>;</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the mutual information between feature <inline-formula id="inf49">
<mml:math id="m62">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and the target variable <inline-formula id="inf50">
<mml:math id="m63">
<mml:mrow>
<mml:mi>Y</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf51">
<mml:math id="m64">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>;</mml:mo>
<mml:mi>X</mml:mi>
<mml:mi>j</mml:mi>
<mml:mo>&#x2223;</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the conditional mutual information between feature <inline-formula id="inf52">
<mml:math id="m65">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and an already-selected feature <inline-formula id="inf53">
<mml:math id="m66">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> given the target variable <inline-formula id="inf54">
<mml:math id="m67">
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf55">
<mml:math id="m68">
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the set of already selected features.</p>
</list-item>
</list>
<p>The goal is to maximize <inline-formula id="inf56">
<mml:math id="m69">
<mml:mrow>
<mml:mi>J</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>;</mml:mo>
<mml:mi>Y</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> to select features that contribute the most information about the target variable while considering the redundancy with already selected features.</p>
<p>Another mutual information-based method, The Maximum Relevance (MIM) criterion is a technique for selecting features that have the most mutual information with the target variable (<xref ref-type="bibr" rid="B15">Che et al., 2017</xref>). The main objective of MIM is to identify those attributes with the highest relevance to predict the outcome, such that it improves the performance of the machine learning models. This technique lessens the dimensionality of the data without losing the most informative attributes. MIM especially comes in handy in applications where the dataset comprises several features, and the identification of the most relevant ones can greatly enhance model efficiency and accuracy. The MIM criterion can be mathematically defined as:<disp-formula id="equ14">
<mml:math id="m70">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>;</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Where:</p>
<list list-type="simple">
<list-item>
<p>&#x2022; <inline-formula id="inf57">
<mml:math id="m71">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>;</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the mutual information between feature <inline-formula id="inf58">
<mml:math id="m72">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>&#x200b; and the target variable <inline-formula id="inf59">
<mml:math id="m73">
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
</list>
<p>Mutual information <inline-formula id="inf60">
<mml:math id="m74">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>;</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> measures the amount of information obtained about one variable through another variable and is defined as:<disp-formula id="equ15">
<mml:math id="m75">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mi mathvariant="italic">log</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>where:</p>
<list list-type="simple">
<list-item>
<p>&#x2022; <inline-formula id="inf61">
<mml:math id="m76">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the joint probability distribution function of <inline-formula id="inf62">
<mml:math id="m77">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf63">
<mml:math id="m78">
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf64">
<mml:math id="m79">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf65">
<mml:math id="m80">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> are the marginal probability distribution functions of <inline-formula id="inf66">
<mml:math id="m81">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf67">
<mml:math id="m82">
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> respectively.</p>
</list-item>
</list>
<p>Normalized Mutual Information Feature Selection (NMIFS) is a feature selection technique that aims to find and rank features based on their normalized mutual information in relation to the target variable (<xref ref-type="bibr" rid="B24">Est&#xe9;vez et al., 2009</xref>). Mutual information quantifies interdependence between variables, showing how much one can inform about another. By standardizing this metric, NMIFS accounts for variable scales and distributions, helping to identify the most informative features for predicting the target variable and enhancing the effectiveness and precision of machine learning models. The NMIFS score for a feature <inline-formula id="inf68">
<mml:math id="m83">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> with respect to the target variable <inline-formula id="inf69">
<mml:math id="m84">
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> can be defined as:<disp-formula id="equ16">
<mml:math id="m85">
<mml:mrow>
<mml:mtext>NMIFS</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Where:</p>
<list list-type="simple">
<list-item>
<p>&#x2022; <inline-formula id="inf70">
<mml:math id="m86">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>;</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the mutual information between feature <inline-formula id="inf71">
<mml:math id="m87">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and the target variable <inline-formula id="inf72">
<mml:math id="m88">
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf73">
<mml:math id="m89">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the entropy of the feature <inline-formula id="inf74">
<mml:math id="m90">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf75">
<mml:math id="m91">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the entropy of the target variable <inline-formula id="inf76">
<mml:math id="m92">
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
</list>
<p>Lasso, or Least Absolute Shrinkage and Selection Operator, is a regularization technique in linear regression that enhances feature selection and controls model complexity (<xref ref-type="bibr" rid="B49">Muthukrishnan and Rohini, 2017</xref>). By adding a penalty based on the absolute values of coefficients, Lasso reduces the coefficients of less important features to zero, promoting simpler models and avoiding overfitting. It is widely used in data science and machine learning to streamline models and focus on the most relevant features. The objective function minimized by Lasso is given by:<disp-formula id="equ17">
<mml:math id="m93">
<mml:mrow>
<mml:munder>
<mml:mi>min</mml:mi>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
</mml:munder>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>p</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">&#x3b1;</mml:mi>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>p</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Where:</p>
<list list-type="simple">
<list-item>
<p>&#x2022; <inline-formula id="inf77">
<mml:math id="m94">
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the number of samples.</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf78">
<mml:math id="m95">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the number of features.</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf79">
<mml:math id="m96">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the target variable for sample <inline-formula id="inf80">
<mml:math id="m97">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf81">
<mml:math id="m98">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the value of feature <inline-formula id="inf82">
<mml:math id="m99">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> for sample <inline-formula id="inf83">
<mml:math id="m100">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf84">
<mml:math id="m101">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the coefficient of feature <inline-formula id="inf85">
<mml:math id="m102">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf86">
<mml:math id="m103">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the regularization parameter that controls the strength of the penalty term <inline-formula id="inf87">
<mml:math id="m104">
<mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>p</mml:mi>
</mml:msubsup>
</mml:mstyle>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
</list>
<p>Ridge regression is a regularization method employed in linear regression models to mitigate overfitting and enhance generalization (<xref ref-type="bibr" rid="B55">Paul and Drineas, 2016</xref>). Ridge regression employs a <inline-formula id="inf88">
<mml:math id="m105">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> penalty on the regression coefficients, in contrast to Lasso regression which utilizes a <inline-formula id="inf89">
<mml:math id="m106">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> penalty. The penalty term is determined by a regularization parameter <inline-formula id="inf90">
<mml:math id="m107">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3b1;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>, which determines the trade-off between accurately fitting the data and punishing big coefficients.<disp-formula id="equ18">
<mml:math id="m108">
<mml:mrow>
<mml:munder>
<mml:mi>min</mml:mi>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
</mml:munder>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>p</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">&#x3b1;</mml:mi>
</mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>p</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msubsup>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
<mml:mi>j</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>
<inline-formula id="inf91">
<mml:math id="m109">
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the number of samples.</p>
<p>
<inline-formula id="inf92">
<mml:math id="m110">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the number of features.</p>
<p>
<inline-formula id="inf93">
<mml:math id="m111">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the target variable for sample <inline-formula id="inf94">
<mml:math id="m112">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>
<inline-formula id="inf95">
<mml:math id="m113">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the value of feature <inline-formula id="inf96">
<mml:math id="m114">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> e for sample <inline-formula id="inf97">
<mml:math id="m115">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>
<inline-formula id="inf98">
<mml:math id="m116">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the coefficient of feature <inline-formula id="inf99">
<mml:math id="m117">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>
<inline-formula id="inf100">
<mml:math id="m118">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the regularization parameter that controls the strength of the penalty term <inline-formula id="inf101">
<mml:math id="m119">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>p</mml:mi>
</mml:msubsup>
</mml:mstyle>
<mml:msubsup>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>j</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
<p>Gradient Boosting is an advanced ensemble learning method that iteratively combines decision trees. Unlike Random Forests, which build trees independently, Gradient Boosting constructs trees sequentially, with each tree correcting errors made by its predecessors (<xref ref-type="bibr" rid="B53">Otchere et al., 2022</xref>). It uses feature importance to identify significant features, revealing how much each feature contributes to accurate predictions. High feature importances allow for more contribution toward the overall impact of model performance while allowing some form of implicit feature selection, especially with complex datasets, making the model perform better and be more interpretable. The method includes several steps, such as -</p>
<p>Initialization of the model with a Constant:</p>
<p>Initialization of the model with a constant value, typically the mean of the target values for a regression problem. This can be represented as: <inline-formula id="inf102">
<mml:math id="m120">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>arg</mml:mi>
<mml:mrow>
<mml:munder>
<mml:mi>min</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
</mml:mstyle>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
<p>Where <inline-formula id="inf103">
<mml:math id="m121">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the loss function, <inline-formula id="inf104">
<mml:math id="m122">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the true target values, and <inline-formula id="inf105">
<mml:math id="m123">
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is a constant.</p>
<p>Iterative tree building:</p>
<p>For <inline-formula id="inf106">
<mml:math id="m124">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula id="inf107">
<mml:math id="m125">
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> (where <inline-formula id="inf108">
<mml:math id="m126">
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the number of trees to be built):<list list-type="simple">
<list-item>
<p>a. Computing the Pseudo-Residuals: <inline-formula id="inf109">
<mml:math id="m127">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
</list>where <inline-formula id="inf110">
<mml:math id="m128">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the pseudo-residuals for each instance <inline-formula id="inf111">
<mml:math id="m129">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> at iteration <inline-formula id="inf112">
<mml:math id="m130">
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf113">
<mml:math id="m131">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the prediction from the previous iteration.</p>
<list list-type="simple">
<list-item>
<p>b. Fitting a Base Learner: Fit a decision tree (base learner) <inline-formula id="inf114">
<mml:math id="m132">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>c. to the pseudo-residuals <inline-formula id="inf115">
<mml:math id="m133">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>arg</mml:mi>
<mml:munder>
<mml:mi>min</mml:mi>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
</mml:mstyle>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>d. Updating the Model: Updating the model by adding the newly fitted tree, scaled by a learning rate <inline-formula id="inf116">
<mml:math id="m134">
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b7;</mml:mi>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
</list>
<p>Combining the Trees:</p>
<p>The final model <inline-formula id="inf117">
<mml:math id="m135">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> after <inline-formula id="inf118">
<mml:math id="m136">
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> iterations is: <inline-formula id="inf119">
<mml:math id="m137">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>M</mml:mi>
</mml:msubsup>
</mml:mstyle>
<mml:mrow>
<mml:mi mathvariant="normal">&#x3b7;</mml:mi>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
<p>Next, Feature selection in Gradient Boosting is typically achieved by examining feature importances derived from the model. Each feature&#x2019;s importance is calculated based on its contribution to reducing the model&#x2019;s prediction error. Feature importance can be computed as follows:</p>
<list list-type="simple">
<list-item>
<p>&#x2022; Calculating the total reduction in the loss function due to splits involving each feature across all trees.</p>
</list-item>
<list-item>
<p>&#x2022; Aggregating these reductions to assign an importance score to each feature.</p>
</list-item>
</list>
<p>Mathematically, the importance of the feature <inline-formula id="inf120">
<mml:math id="m138">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> can be represented as:<disp-formula id="equ19">
<mml:math id="m139">
<mml:mrow>
<mml:mtext>Importance</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>M</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:munderover>
</mml:mstyle>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>m</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Where, <inline-formula id="inf121">
<mml:math id="m140">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the number of nodes in the tree, and is the reduction in the loss function due to the split on feature <inline-formula id="inf122">
<mml:math id="m141">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> at node <inline-formula id="inf123">
<mml:math id="m142">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> in tree <inline-formula id="inf124">
<mml:math id="m143">
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</sec>
<sec id="s2-4">
<title>2.4 Categorization of features using an ensemble of different subsets of features</title>
<p>The genes obtained from different feature selection approaches are used as subsets. These subsets were then used to create an ensemble by categorizing the genes by assigning values 12 to 1. A gene is classified as 12 if it appears in all twelve subsets, 11 if it is present in eleven of the twelve subsets, and so on. Subsequently, genes that fell into at least 6 subsets were selected for Cox regression-based survival analysis. This decision guarantees that a minimum of 50% of the feature selection methods will support the inclusion of these genes, hence strengthening the reliability subsequent to analysis.</p>
</sec>
<sec id="s2-5">
<title>2.5 GO and KEGG enrichment analysis</title>
<p>The ensemble approach yielded a consolidated list of genes, then annotated using DAVID tool version 6.8 (<ext-link ext-link-type="uri" xlink:href="https://david.ncifcrf.gov/">https://david.ncifcrf.gov/</ext-link>) (<xref ref-type="bibr" rid="B63">Sherman et al., 2022</xref>). The criteria for conducting Gene Ontology (GO) and KEGG pathway enrichment analysis were set as a p-value below 0.05 and a false discovery rate (FDR) below 0.05. The GO enrichment analysis was utilized to ascertain the biological activities of these genes. GO ontologies are divided into three categories: molecular function (MF), cellular component (CC), and biological process (BP). The KEGG pathway enrichment analysis identified metabolic pathways that showed a significant enrichment of genes, as compared to the total genome background. The SRPlot online toolkit (<ext-link ext-link-type="uri" xlink:href="http://www.bioinformatics.com.cn/srplot">http://www.bioinformatics.com.cn/srplot</ext-link>) (<xref ref-type="bibr" rid="B67">Tang et al., 2023</xref>) was used to display the findings of the Gene Ontology (GO) and Kyoto Encyclopedia of Genes and Genomes (KEGG) pathway enrichment studies.</p>
</sec>
<sec id="s2-6">
<title>2.6 Survival analysis</title>
<p>Survival analysis was performed using the TCGA-LIHC (Liver Hepatocellular Carcinoma) dataset to investigate the association between the expression level of the genes and overall survival. First, clinical data and gene expression data were retrieved from the TCGA-LIHC cohort using the &#x201c;TCGAbiolinks&#x201d; package (<xref ref-type="bibr" rid="B19">Colaprico et al., 2016</xref>). The gene expression data is preprocessed, and a Variance Stabilizing Transformation (VST) is applied using DESeq2 (<xref ref-type="bibr" rid="B45">Love et al., 2014</xref>). The expression levels of the genes were extracted, and a median value was calculated to stratify the samples into &#x201c;HIGH&#x201d; and &#x201c;LOW&#x201d; expression groups. Next, the clinical data is merged with the gene expression data, and a Cox proportional hazards regression model is fitted to estimate the hazard ratio (HR) and its 95% confidence interval (CI) using the survival package (<xref ref-type="bibr" rid="B69">Therneau, 2021</xref>). Finally, a Kaplan-Meier survival curve is plotted using the &#x201c;survminer&#x201d; package (<xref ref-type="bibr" rid="B56">Pawar et al., 2022</xref>).</p>
</sec>
<sec id="s2-7">
<title>2.7 Screening of possible drug candidates for repurposing</title>
<p>The biomarkers identified through survival analysis were used as input in two separate databases i.e., DGIdb (The Drug Gene Interaction Database) (<xref ref-type="bibr" rid="B13">Cannon et al., 2024</xref>) and Drug repurposing encyclopedia (<xref ref-type="bibr" rid="B42">Li et al., 2023</xref>). Additionally, clue.io COMMAND app web tool (<xref ref-type="bibr" rid="B74">Xie et al., 2022</xref>) was utilized to screen potential drug candidates. The DGIdb database uses a combination of expert curation and text-mining approaches to mine drug-gene interactions mined from DrugBank, PharmGKB, ChEMBL, Drug Target Commons, and others. On the other hand, The Drug Repurposing Encyclopedia utilizes the Molecular Signatures Database (MSigDB, <ext-link ext-link-type="uri" xlink:href="https://www.gsea-msigdb.org/gsea/msigdb/">https://www.gsea-msigdb.org/gsea/msigdb/</ext-link>), and consensus drug profiles from DREIMT (<ext-link ext-link-type="uri" xlink:href="http://www.dreimt.org/">http://www.dreimt.org/</ext-link>), which are derived from the Connectivity Map (CMap) LINCS gene expression dataset (<ext-link ext-link-type="uri" xlink:href="https://clue.io/">https://clue.io/</ext-link>). The command app also utilizes connectivity map (CMap) analysis to screen possible drugs.</p>
</sec>
<sec id="s2-8">
<title>2.8 Molecular docking and MMGBSA analysis</title>
<sec id="s2-8-1">
<title>2.8.1 Structure retrieval</title>
<p>The potential drugs screened through drug repurposing were used as ligands in the docking analysis against the targets implicated in disease progression. The chemical structures of the drugs were obtained from the PubChem database (<ext-link ext-link-type="uri" xlink:href="https://pubchem.ncbi.nlm.nih.gov/">https://pubchem.ncbi.nlm.nih.gov/</ext-link>) (<xref ref-type="bibr" rid="B72">Wang et al., 2009</xref>). The structures of the targets, namely, ABAT (PDB ID: 1OHW), ABCB11 (PDB ID: 6LRO), C8B (PDB ID: 3OJY), and FBX23 (PDB ID: 416J), were obtained from the RCSB database (<ext-link ext-link-type="uri" xlink:href="https://www.rcsb.org/">https://www.rcsb.org/</ext-link>) (<xref ref-type="bibr" rid="B22">Deshpande et al., 2005</xref>) in PDB format. The RCSB-PDB database (<ext-link ext-link-type="uri" xlink:href="https://www.rcsb.org/">https://www.rcsb.org/</ext-link>) (<xref ref-type="bibr" rid="B22">Deshpande et al., 2005</xref>) lacked monomeric structures for the proteins APOF, CENPV, MBTPS1, and ZFP1. Therefore, the protein sequences were obtained from the NCBI database, and a BLAST search was conducted in the SWISS-MODEL template library to identify structurally comparable homologous structures. The structures exhibiting similarity were subsequently acquired via the SWISS-MODEL tool (<xref ref-type="bibr" rid="B36">Kiefer et al., 2009</xref>).</p>
</sec>
<sec id="s2-8-2">
<title>2.8.2 Protein and ligand preparation</title>
<p>The proteins were prepared using the Schr&#xf6;dinger software (Maestro Version 12.5.139, Schr&#xf6;dinger, LLC, New York) with the OPLS3 force field (<xref ref-type="bibr" rid="B28">Harder et al., 2016</xref>) to ensure precise depiction. Similarly, the ligands were produced using the OPLS3 force field in LigPrep (Maestro Version 12.5.139, Schr&#xf6;dinger, LLC, New York). This preparatory step aimed to ensure that the docking analysis was consistent and reliable.</p>
</sec>
<sec id="s2-8-3">
<title>2.8.3 Docking procedure</title>
<p>Before docking analysis, receptor grids were created for each protein using Glide (Maestro Version 12.5.139, Schr&#xf6;dinger, LLC, New York) to ensure precise accommodation. The grids were constructed with accurate coordinates based on binding pocket predictions from PrankWeb (<ext-link ext-link-type="uri" xlink:href="https://prankweb.cz/">https://prankweb.cz/</ext-link>) (<xref ref-type="bibr" rid="B33">Jendele et al., 2019</xref>). After generating the grids, the docking scores were calculated systematically using Glide&#x2019;s Extra-precision (XP) docking model. The scores provided measurable data on the binding affinities between each ligand and its respective protein target.</p>
</sec>
<sec id="s2-8-4">
<title>2.8.4 Binding energy calculations using Prime/MM GBSA analysis</title>
<p>The binding free energies of the protein-ligand complex are evaluated using the Molecular Mechanics-Generalized Born Surface Area (MM-GBSA) approach, implemented in the Prime module of the Schr&#xf6;dinger suite (Maestro Version 12.5.139, Schr&#xf6;dinger, LLC, New York). The calculations utilize the OPLS 2005 force field and the VSGB solvation model for accurate energy estimations.</p>
</sec>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<sec id="s3-1">
<title>3.1 Key features selected by the ensemble feature selection approach</title>
<p>After data preparation, seven well-known machine learning techniques such as DISCR, NB, RF, DT, KNN, SVM, and ANN, were tested on the entire dataset. <xref ref-type="fig" rid="F2">Figure 2</xref> illustrates the accuracy scores of these machine learning techniques, each conducted independently on the whole preprocessed dataset using 10-fold cross-validation. Based on the accuracy values, DISCR outperformed other machine learning techniques, achieving the highest accuracy score of 0.90.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Box plot comparing the performance of different classifiers in terms of Accuracy (green) and F1 Score (yellow).</p>
</caption>
<graphic xlink:href="fbinf-05-1522401-g002.tif"/>
</fig>
<p>After selecting DISCR as the optimal classification method, twelve feature selection models were applied to identify the most significant gene subsets. These models included ReliefF, Cram&#xe9;r&#x2019;s V, Kendall&#x2019;s Tau, Pearson&#x2019;s correlation, Spearman&#x2019;s correlation coefficient, CIFE, JMI, MIM, NMIFS, LASSO, RIDGE, and Gradient Boosting. Iteratively, each of the twelve methods separately selects different subsets of genes like top 30, 40, 50, and so on. DISCR with 10-fold cross-validation is used for each subset of the different methods separately to check the classification accuracy. The accuracy details for different subsets of the genes, such as 30, 40, 50 to 150 genes, are reported in <xref ref-type="table" rid="T1">Table 1</xref>. Out of all the filter-based strategies, Relief F achieved the greatest accuracy of 0.70297 using a set of 30 characteristics. Similarly, the accuracy reached by Cram&#xe9;r&#x2019;s V, Kendall Tau, Pearson&#x2019;s correlation, and Spearman&#x2019;s correlation coefficient was 0.623762, 0.683168, 0.722772, and 0.673267, respectively. These accuracies were obtained using 40, 20, 20, and 10 features. Among the approaches that use mutual information, CIFE, JMI, MIM, and NMIFS achieved accuracies of 0.643564, 0.752475, 0.475248, and 0.712871 using 40, 30, 70, and 20 features, respectively. Within the set of embedded methods, the LASSO algorithm achieved the highest accuracy score of 1 while using 40 features. Ridge and elastic-net algorithms achieved an accuracy of 0.752475 and 0.712871, respectively, using a total of 30 features. <xref ref-type="fig" rid="F3">Figure 3</xref> illustrates the performance matrices of different feature selection models (A-D) (Performance metrics for different feature selection models (E-L) were provided as <xref ref-type="sec" rid="s13">Supplementary Material S2</xref>).</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Classification accuracy values of DISCR using twelve separate feature selection methods.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Number of features</th>
<th align="left">Relief F</th>
<th align="left">Cammers V</th>
<th align="left">Kendall tau</th>
<th align="left">Pearson correlation coefficient</th>
<th align="left">Spearman&#x2019;s rank correlation coefficient</th>
<th align="left">CIFE</th>
<th align="left">JMI</th>
<th align="left">MIM</th>
<th align="left">NMIFS</th>
<th align="left">LASSO</th>
<th align="left">RIDGE</th>
<th align="left">Gradient boosting</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">10</td>
<td align="center">0.603</td>
<td align="center">0.623</td>
<td align="center">0.663</td>
<td align="center">0.683</td>
<td align="center">0.673</td>
<td align="center">0.603</td>
<td align="center">0.663</td>
<td align="center">0.435</td>
<td align="center">0.663</td>
<td align="center">0.762</td>
<td align="center">0.663</td>
<td align="center">0.722</td>
</tr>
<tr>
<td align="center">20</td>
<td align="center">0.643</td>
<td align="center">0.594</td>
<td align="center">0.683</td>
<td align="center">0.722</td>
<td align="center">0.653</td>
<td align="center">0.643</td>
<td align="center">0.712</td>
<td align="center">0.376</td>
<td align="center">0.712</td>
<td align="center">0.831</td>
<td align="center">0.742</td>
<td align="center">0.693</td>
</tr>
<tr>
<td align="center">30</td>
<td align="center">0.702</td>
<td align="center">0.584</td>
<td align="center">0.663</td>
<td align="center">0.653</td>
<td align="center">0.653</td>
<td align="center">0.613</td>
<td align="center">0.752</td>
<td align="center">0.465</td>
<td align="center">0.663</td>
<td align="center">0.891</td>
<td align="center">0.752</td>
<td align="center">0.712</td>
</tr>
<tr>
<td align="center">40</td>
<td align="center">0.653</td>
<td align="center">0.623</td>
<td align="center">0.584</td>
<td align="center">0.574</td>
<td align="center">0.633</td>
<td align="center">0.643</td>
<td align="center">0.683</td>
<td align="center">0.465</td>
<td align="center">0.673</td>
<td align="center">1</td>
<td align="center">0.732</td>
<td align="center">0.712</td>
</tr>
<tr>
<td align="center">50</td>
<td align="center">0.554</td>
<td align="center">0.603</td>
<td align="center">0.574</td>
<td align="center">0.504</td>
<td align="center">0.574</td>
<td align="center">0.564</td>
<td align="center">0.683</td>
<td align="center">0.415</td>
<td align="center">0.683</td>
<td align="center">1</td>
<td align="center">0.673</td>
<td align="center">0.663</td>
</tr>
<tr>
<td align="center">60</td>
<td align="center">0.524</td>
<td align="center">0.554</td>
<td align="center">0.504</td>
<td align="center">0.435</td>
<td align="center">0.534</td>
<td align="center">0.495</td>
<td align="center">0.584</td>
<td align="center">0.455</td>
<td align="center">0.653</td>
<td align="center">0.99</td>
<td align="center">0.574</td>
<td align="center">0.613</td>
</tr>
<tr>
<td align="center">70</td>
<td align="center">0.514</td>
<td align="center">0.524</td>
<td align="center">0.425</td>
<td align="center">0.524</td>
<td align="center">0.485</td>
<td align="center">0.504</td>
<td align="center">0.554</td>
<td align="center">0.475</td>
<td align="center">0.574</td>
<td align="center">0.891</td>
<td align="center">0.524</td>
<td align="center">0.475</td>
</tr>
<tr>
<td align="center">80</td>
<td align="center">0.564</td>
<td align="center">0.495</td>
<td align="center">0.475</td>
<td align="center">0.455</td>
<td align="center">0.475</td>
<td align="center">0.495</td>
<td align="center">0.524</td>
<td align="center">0.475</td>
<td align="center">0.603</td>
<td align="center">0.643</td>
<td align="center">0.564</td>
<td align="center">0.524</td>
</tr>
<tr>
<td align="center">90</td>
<td align="center">0.455</td>
<td align="center">0.445</td>
<td align="center">0.504</td>
<td align="center">0.465</td>
<td align="center">0.485</td>
<td align="center">0.504</td>
<td align="center">0.504</td>
<td align="center">0.435</td>
<td align="center">0.603</td>
<td align="center">0.673</td>
<td align="center">0.603</td>
<td align="center">0.405</td>
</tr>
<tr>
<td align="center">100</td>
<td align="center">0.504</td>
<td align="center">0.386</td>
<td align="center">0.405</td>
<td align="center">0.495</td>
<td align="center">0.504</td>
<td align="center">0.534</td>
<td align="center">0.524</td>
<td align="center">0.465</td>
<td align="center">0.554</td>
<td align="center">0.504</td>
<td align="center">0.554</td>
<td align="center">0.504</td>
</tr>
<tr>
<td align="center">110</td>
<td align="center">0.465</td>
<td align="center">0.435</td>
<td align="center">0.465</td>
<td align="center">0.485</td>
<td align="center">0.495</td>
<td align="center">0.504</td>
<td align="center">0.504</td>
<td align="center">0.435</td>
<td align="center">0.534</td>
<td align="center">0.504</td>
<td align="center">0.594</td>
<td align="center">0.524</td>
</tr>
<tr>
<td align="center">120</td>
<td align="center">0.485</td>
<td align="center">0.524</td>
<td align="center">0.564</td>
<td align="center">0.534</td>
<td align="center">0.524</td>
<td align="center">0.485</td>
<td align="center">0.495</td>
<td align="center">0.455</td>
<td align="center">0.524</td>
<td align="center">0.504</td>
<td align="center">0.623</td>
<td align="center">0.455</td>
</tr>
<tr>
<td align="center">130</td>
<td align="center">0.465</td>
<td align="center">0.574</td>
<td align="center">0.534</td>
<td align="center">0.504</td>
<td align="center">0.495</td>
<td align="center">0.435</td>
<td align="center">0.475</td>
<td align="center">0.435</td>
<td align="center">0.475</td>
<td align="center">0.504</td>
<td align="center">0.594</td>
<td align="center">0.564</td>
</tr>
<tr>
<td align="center">140</td>
<td align="center">0.445</td>
<td align="center">0.564</td>
<td align="center">0.514</td>
<td align="center">0.524</td>
<td align="center">0.534</td>
<td align="center">0.504</td>
<td align="center">0.495</td>
<td align="center">0.425</td>
<td align="center">0.504</td>
<td align="center">0.504</td>
<td align="center">0.574</td>
<td align="center">0.475</td>
</tr>
<tr>
<td align="center">150</td>
<td align="center">0.415</td>
<td align="center">0.544</td>
<td align="center">0.504</td>
<td align="center">0.514</td>
<td align="center">0.564</td>
<td align="center">0.514</td>
<td align="center">0.495</td>
<td align="center">0.415</td>
<td align="center">0.445</td>
<td align="center">0.504</td>
<td align="center">0.594</td>
<td align="center">0.455</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Performance metrics for different feature selection models, <bold>(A)</bold> Relief F, <bold>(B)</bold> Cammers V, <bold>(C)</bold> Kendall Tau, and <bold>(D)</bold> Pearson Correlation. Metrics for the rest of the feature selection models (E-L) were provided as supplementary file 2. Here X-axis represents the numbers of features, and the Y-axis represents performance metrics.</p>
</caption>
<graphic xlink:href="fbinf-05-1522401-g003.tif"/>
</fig>
<p>The subsets of genes obtained from different feature selection approaches are subsequently merged to create an ensemble set of genes. These genes are then categorized from 12 to 1 using a consensus approach with those scoring at least 6 being considered as most important features. Through this ensemble approach a set of ten genes were identified. These genes include <italic>C8B</italic> (Complement C8 Beta Chain), <italic>APOF</italic> (Apolipoprotein F), <italic>FBXL3</italic> (F-Box and Leucine Rich Repeat Protein 3), <italic>ABAT</italic> (4-Aminobutyrate Aminotransferase), <italic>ZFP1</italic> (ZFP1 Zinc Finger Protein), <italic>MBTPS1</italic>(Membrane Bound Transcription Factor Peptidase, Site 1), <italic>CENPV</italic> (Centromere Protein V), <italic>METTL23</italic> (Methyltransferase 23, Arginine), <italic>RPL9</italic> (Ribosomal Protein L9), and <italic>ABCB11</italic> (ATP Binding Cassette Subfamily B Member 11).</p>
</sec>
<sec id="s3-2">
<title>3.2 Evaluating the biomarker potential through survival analysis</title>
<p>Kaplan-Meier survival analysis was used to evaluate the predicted survival probability over time in liver cancer patients, comparing normal and HCC samples based on the expression levels of key genes identified through an ensemble approach. Out of the ten identified genes, eight genes were found to have the worst overall survival rate for HCC. The identified genes were <italic>ABAT</italic> (HR: 1.69), <italic>C8B</italic> (HR: 1.69), <italic>FBXL3</italic> (HR: 1.43), <italic>ZFP1</italic> (HR: 1.35), <italic>ABCB11</italic>(HR: 1.33), <italic>MBTPS1</italic>(HR: 1.21), <italic>CENPV</italic> (HR: 1.2) and <italic>APOF</italic> (HR: 1.18). In the case of <italic>ABAT</italic> gene, the average hazard ratio stands at 1.69. This suggests that individuals with high gene expression face a 69% elevated risk of death compared to those with low gene expression, regardless of the circumstance as depicted in <xref ref-type="fig" rid="F4">Figure 4</xref>. Similarly, for <italic>C8B</italic> and <italic>FBXL3</italic>, high expression corresponds to a 69% and 43% increased risk of death, respectively, in comparison to scenarios with low expression of these genes. HR value greater than 1 indicates a higher hazard rate in the reference group compared to the comparison group, suggesting that genes with higher expression in the comparison group are associated with an increased risk of death. Therefore, these genes could serve as predictive markers for poor prognosis. Survival analysis results of the rest of the genes are provided in <xref ref-type="sec" rid="s13">Supplementary Material S3</xref>.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Survival curve analysis results of three hub genes <bold>(A)</bold>- ABAT, <bold>(B)</bold>- C8B, <bold>(C)</bold>- FBXL3 and <bold>(D)</bold>- ZFP1. X-axis represents time in months and Y-axis represents denotes survival probability, survival curve for other genes (E-J) are provided in supplementary file 3.</p>
</caption>
<graphic xlink:href="fbinf-05-1522401-g004.tif"/>
</fig>
</sec>
<sec id="s3-3">
<title>3.3 GO term enrichment analyses and KEGG pathway analyses</title>
<p>The gene ontology analysis using the ten identified genes from feature selection and based on the selected identifier - &#x201c;OFFICIAL GENE SYMBOL,&#x201d; and p-value &#x3c; 0.05 and FDR &#x3c;0.05 cut-offs yielded significant GO terms, depicted in <xref ref-type="fig" rid="F5">Figure 5</xref>, for the marker genes. The enriched GO terms related to biological processes found to be negative regulation of gamma-aminobutyric acid secretion (GO:0014053), histone H3-R17 methylation (GO:0034971), gamma-aminobutyric acid biosynthetic process (GO:0009449), positive regulation of catecholamine metabolic process (GO:0045915), regulation of cholesterol biosynthetic process (GO:0045540), regulation of sterol biosynthetic process (GO:0106118), etc. The results suggest the involvement of metabolism and epigenetic processes, indicating dynamic changes in the disease progression. The top GO terms associated with cellular components terms identified as membrane attack complex (GO:0005579), low-density lipoprotein particle (GO:0034362), pore complex (GO:0046930), plasma lipoprotein particle (GO:0034358), etc; suggesting the involvement of membrane structure changes, lipid transport, etc, that are key characteristics of changes due to lipid accumulation and oxidative damage in hepatic cells in cellular dysfunction. The GO terms associated with molecular functions are histone-arginine N-methyltransferase activity (GO:0008469), carbon-sulfur lyase activity (GO:0016846), cholesterol binding (GO:0015485), pyridoxal phosphate binding (GO:0030170), etc., indicating lipid interaction and metabolic processes as important factors in this disease progression.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>The results of the Gene Ontology (GO) enrichment analysis, visualized through a bubble chart, <bold>(A)</bold>-GO terms related to Biological Processes, <bold>(B)</bold>-GO terms associated with Cellular Components, and <bold>(C)</bold>-GO terms related to Molecular Functions. Each bubble on the Y-axis represents a different GO term, while the X-axis displays the enrichment score. The size of each bubble corresponds to the gene counts associated with the term. The colour of the bubbles reflects the P-value for each GO term, with the intensity of red indicating higher corrected P-values.</p>
</caption>
<graphic xlink:href="fbinf-05-1522401-g005.tif"/>
</fig>
<p>The genes were also subjected to KEGG pathway analysis to determine their association with their corresponding biological pathways. A total of 13 pathways-related KEGG terms were obtained from the database. The criteria of p-value less than 0.05 and FDR less than 0.05 were used for analysis. The top enriched pathways and the respective gene counts are depicted in <xref ref-type="fig" rid="F6">Figure 6</xref>. The KEGG analysis results showed that the genes are enriched in Butanoate metabolism (hsa00650), Propanoate metabolism (hsa00640), Alanine aspartate and glutamate metabolism (hsa00250), Complement and coagulation cascades (hsa04610), Protein processing in the endoplasmic reticulum (hsa04141), etc.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>
<bold>(A)</bold>- A chord plot represents pathways and genes&#x2019; relationship. The size of the pathway bubbles increases with the number of connected genes. <bold>(B)</bold>- The left side displays a Sankey plot, illustrating the genes within each pathway. The right side features a dot plot, where the dot sizes represent the number of genes, and the dot colours indicate P values.</p>
</caption>
<graphic xlink:href="fbinf-05-1522401-g006.tif"/>
</fig>
</sec>
<sec id="s3-4">
<title>3.4 Multiphase drug repurposing strategy based on identified biomarkers</title>
<p>The integrated approach of drug repurposing, which utilized connectivity map analysis and drug-gene interaction text-mining, etc, facilitated the creation of a library of potential repurposed drugs. This resulted in the identification of 81 potential candidate drugs. All these candidates have been utilized for molecular docking to analyze further interactions (details of the drugs are given in <xref ref-type="sec" rid="s13">Supplementary Material S4</xref>). Among these 81 drugs, DGIdb (The Drug Gene Interaction Database) based screening identified a total of 19 drugs. The drugs with the highest interaction scores (IS) were Vigabatrin (IS- 11.79601078), Divalproex sodium (IS- 8.847008084), and Pyruvic Acid (IS-5.89800539). Screening using the Drug Repurposing Encyclopedia (DRE) database identified a total of 55 drugs, with Scopolamine (Enrichment score (ES) - 0.973), Amiloride (ES - 0.971), Damnacanthal (ES - 0.967), and Esculin (ES - 0.962) showing the highest enrichment scores. In addition, the search conducted by the Command app revealed a total of 6 drugs, including Glibenclamide, Phenelzine, etc.</p>
</sec>
<sec id="s3-5">
<title>3.5 Drug-target interaction analysis through molecular docking and MM-GBSA calculations</title>
<p>Docking results showed good interaction between ZFP1, C8B, MBPTS1, CENPV, ABCB11, and Diosmin with docking scores of &#x2212;11.6821 kcal/mol, &#x2212;11.134 kcal/mol, 10.4712 kcal/mol, &#x2212;10.2391 kcal/mol, and &#x2212;9.85606 kcal/mol respectively (<xref ref-type="fig" rid="F7">Figures 7A&#x2013;E</xref>). There are three hydrogen bonds between ZFP1 and Diosmin at THR112, ASN116 and GLU133 positions (<xref ref-type="fig" rid="F7">Figure 7A</xref>) and eight hydrogen bonds between C8B and Diosmin at positions&#x2013;ARG82, GLN65, CYS79, THR423, ASP424, LEU259, TYR166 and TYR141 (<xref ref-type="fig" rid="F7">Figure 7B</xref>). The hydrogen bond interactions between MBPTS1 and Diosmin found at VAL54, TRP556, MET353, ARG386, and ASN515 (<xref ref-type="fig" rid="F7">Figure 7C</xref>) and the interactions between CENPV and Diosmin included Hydrogen bonds at HIS210, ARG225, SER254 positions and PI-PI interactions at TYR561 and TRP256 positions (<xref ref-type="fig" rid="F7">Figure 7D</xref>). The details about the interactions between top protein-ligand complexes are provided in <xref ref-type="sec" rid="s13">Supplementary Material S5</xref>. ABAT and Esculin with a docking score of &#x2212;4.30418 kcal/mol (<xref ref-type="fig" rid="F7">Figure 7H</xref>). APOF and Lapatinib with a docking score of &#x2212;6.90333 kcal/mol (<xref ref-type="fig" rid="F7">Figure 7G</xref>), indicating a strong interaction between the protein and ligand. A satisfactory interaction has also been observed between FBX23 and Phenelzine with a docking score of &#x2212;6.30952 kcal/mol (<xref ref-type="fig" rid="F7">Figure 7F</xref>). The MM-GBSA analysis supported the docking results, showing strong binding affinities. CENPV-Diosmin had a binding free energy of &#x2212;100.71 kcal/mol, with Van der Waals and electrostatic contributions of &#x2212;49.50 and &#x2212;46.93 kcal/mol, respectively. Similarly, C8B-Diosmin exhibited a binding free energy of &#x2212;74.58 kcal/mol, with Van der Waals and electrostatic contributions of &#x2212;37.40 and &#x2212;52.24 kcal/mol (The detailed result of MM-GBSA analysis of the top ligand-protein complexes were provided as <xref ref-type="sec" rid="s13">Supplementary Material S6</xref>). These results highlight a significant binding affinity driven by a balanced interplay of electrostatic, van der Waals, and lipophilic interactions.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>
<bold>(A)</bold> ZFP1 with Diosmin, <bold>(B)</bold> C8B with Diosmin, <bold>(C)</bold> MBTPS1 with Diosmin, <bold>(D)</bold> CENPV with Diosmin, <bold>(E)</bold> ABCB11 with Diosmin, <bold>(F)</bold> FBXL3 with Phenelzine, <bold>(G)</bold> APOF with Lapatinib, and <bold>(H)</bold> ABAT with Esculin.</p>
</caption>
<graphic xlink:href="fbinf-05-1522401-g007.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<p>Identification of biomarkers associated with NAFLD-driven HCC through conventional genomics-based methods is not an easy task, as the disease progresses through several stages over a prolonged period of time (<xref ref-type="bibr" rid="B75">Xu et al., 2023</xref>). The present work highlights the implementation of machine learning methods over conventional strategies in prognostic biomarker identification. The selection of an appropriate classifier for disease stage classification was the first step of this study, as different data types respond differently to various classifiers (<xref ref-type="bibr" rid="B20">de Amorim et al., 2023</xref>). It is essential to choose a classifier that is well-suited to the specific characteristics of the data to achieve optimal performance. Factors such as the distribution of the data, the presence of noise, and the complexity of the relationships between features can all influence the effectiveness of a given classifier (<xref ref-type="bibr" rid="B61">Saseendran et al., 2019</xref>). DISCR emerged as the most appropriate classifier in this analysis, exhibiting superior performance and reliability compared to other classifiers. The advantage of DISCR is that it can model intricate decision boundaries by discriminating between classes in a diverse set of data (<xref ref-type="bibr" rid="B46">Meyer-Baese and Schmid, 2014</xref>). Consequently, it was chosen for the subsequent steps of this study. Given that the primary objective of this work is to identify the key genes involved in the stage-wise progression of NAFLD to HCC, the main focus was on finding genes that are consistently expressed across all stages of the disease. While mutual information-based feature selection methods (such as JMI, MIM, and NMIFS) can effectively capture dependencies between variables, they also have potential disadvantages, such as high computational complexity and sensitivity to noise. To overcome these limitations, a variety of alternative feature selection methods were employed, including filter-based methods like ReliefF, Cram&#xe9;r&#x2019;s V, Kendall&#x2019;s Tau, Pearson correlation, and Spearman correlation coefficient, as well as embedded methods like LASSO, Ridge regression, and gradient boosting techniques. These methods help in addressing different aspects of feature selection. For example, filter-based methods assess the individual importance of each feature, without considering how they relate to each other (<xref ref-type="bibr" rid="B4">Bellotti et al., 2014</xref>), while embedded methods integrate feature selection into the model training process, simultaneously selecting the most relevant features, enabling more flexible and refined selection by considering how features interact with each other and influence the target variable (<xref ref-type="bibr" rid="B11">Bouchlaghem et al., 2022</xref>). This ensemble method enhances feature importance, improving model performance and identifying a robust set of ten co-expressed genes linked to disease progression. These genes include <italic>C8B, APOF, FBXL3, ABAT, ZFP1, MBTPS1, CENPV, METTL23, RPL9</italic>, and <italic>ABCB11</italic>. The pathway enrichment analysis results suggest that these genes are primarily involved in various pathways, including metabolism-related pathways such as Alanine, Aspartate, and Glutamate Metabolism, Propanoate Metabolism, Butanoate Metabolism, and Valine, Leucine, and Isoleucine Degradation. They are also involved in protein processing pathways like protein processing in the Endoplasmic Reticulum and Ribosome, as well as inflammation and immunological pathways like Complement and Coagulation Cascades and Systemic Lupus Erythematosus. These results indicate the involvement of both protein and lipid metabolism and inflammation in disease progression. Metabolic events specifically protein and lipid metabolism as well as ER-Mitochondrial dysregulation due to high metabolic stress is a long-suspected event for NAFLD to HCC progression (<xref ref-type="bibr" rid="B79">Zheng et al., 2023</xref>; <xref ref-type="bibr" rid="B40">L&#xe9;veill&#xe9; and Estall, 2019</xref>). Metabolic dysregulation in this disease progression is likely linked to inflammation and oxidative damage, which contribute to the onset of cirrhosis and the eventual development of liver cancer. This significant involvement of metabolism-related pathways also suggests that this evaluated energy production is possibly required for the rapid growth and division of cancer cells, a primary feature of cancer cells (<xref ref-type="bibr" rid="B57">Phan et al., 2014</xref>). Furthermore, out of the ten identified genes, eight genes <italic>ABAT, C8B, FBXL3, ZFP1, ABCB11, MBTPS1, CENPV,</italic> and <italic>APOF</italic> showed a strong association with lower overall survival rates in patients with HCC. The high hazard ratios (HR) for these genes indicate that patients with higher expression levels of these genes tend to have a poorer prognosis, such as a shorter overall survival time. The <italic>ABAT</italic> gene encodes the 4-Aminobutyrate Aminotransferase which is crucial for the catabolism of inhibitory neurotransmitters like GABA-transaminase (<xref ref-type="bibr" rid="B6">Besse et al., 2015</xref>). Altered expression of this gene has been observed in breast cancer (<xref ref-type="bibr" rid="B18">Chen et al., 2019</xref>), and its involvement in tumorigenesis and tumor immunity in HCC is a recent finding (<xref ref-type="bibr" rid="B27">Gao et al., 2022</xref>). The <italic>C8B</italic> gene encodes the beta subunit of complement complex 8 (<xref ref-type="bibr" rid="B78">Zhang Y. et al., 2021</xref>), which has recently been found to have predictive potential in hepatocellular carcinoma (HCC) development (<xref ref-type="bibr" rid="B73">Xiao et al., 2022</xref>). <italic>FBXL3</italic> encodes for an F-box and leucine-rich repeat protein 3, which plays a vital role in regulating circadian rhythm (<xref ref-type="bibr" rid="B25">Fagiani et al., 2022</xref>). It works together with <italic>CRY2</italic> to degrade the C-MYC protein, which helps prevent tumor growth (<xref ref-type="bibr" rid="B31">Huber et al., 2016b</xref>). <italic>FBXL3</italic> has been previously reported as an important cancer marker (<xref ref-type="bibr" rid="B30">Huber et al., 2016a</xref>). The <italic>ZFP1</italic> gene encodes zinc finger motif proteins, which play a crucial role in several transcriptional activation and repression processes (<xref ref-type="bibr" rid="B41">Li X. et al., 2022</xref>). <italic>ABCB11</italic> encodes the primary ABC transporter, which is called the bile salt export pump (BSEP), in hepatic cells (<xref ref-type="bibr" rid="B64">Sohail et al., 2021</xref>). Malfunctioning BSEP is particularly significant in liver malignancies (<xref ref-type="bibr" rid="B39">Lagana et al., 2015</xref>). <italic>MBTPS1</italic>, which encodes the Membrane-Bound Transcription Factor Peptidase protein, has been implicated in the process of cancer cell proliferation (<xref ref-type="bibr" rid="B29">Hartal-Benishay et al., 2022</xref>). <italic>CENPV</italic> encodes Centromere Protein V, a vital component involved in the process of mitosis and exhibiting significant upregulation in several cancer types (<xref ref-type="bibr" rid="B77">Zhang S. et al., 2021</xref>). The <italic>APOF</italic> gene encodes Apolipoprotein F, which plays a role in lipid metabolism by binding to LDL and VLDL (<xref ref-type="bibr" rid="B21">Deprince et al., 2023</xref>). While its exact mechanism in HCC is not completely understood, it has been shown to act as a tumor suppressor and could be a promising target for therapeutic development in HCC. Subsequently, as these genes were identified as key modulators involved in this progression, they were further screened as targets to identify potential drug candidates utilizing DGIdb, Drug Repurposing Encyclopedia, and the COMMAND app. These databases utilize several data sources including gene expression data and approaches, such as expert curation, text-mining, etc to discover possible therapeutic interventions. The resulting combined list of 81 potential drugs was further screened through molecular docking and MM-GBSA analysis. The analysis revealed a strong interaction between the drug Diosmin and targets such as ZFP1, C8B, MBPTS1, CENPV, and ABCB11, characterized by numerous hydrogen bonds and pi-pi interactions. Diosmin is reported to have anti-inflammatory, antioxidative, insulin-sensitizing, antimutagenic, and antiulcer properties, and is widely being used for the improvement of blood-related insufficiencies (<xref ref-type="bibr" rid="B32">Huwait and Mobashir, 2022</xref>). ABAT had a robust interaction with Esculin, whilst APOF and FBX23 demonstrated notable interactions with Lapatinib and Phenelzine, respectively. Esculin is known to have anti-inflammatory properties and is used in multiple disorders like arthritis, ulcerative colitis, etc (<xref ref-type="bibr" rid="B12">Cai and Cai, 2023</xref>). Whereas, Lapatinib is reportedly used in breast cancer treatment (<xref ref-type="bibr" rid="B52">Opdam et al., 2012</xref>) and Phenelzine is a widely used drug in panic disorders, Chronic resistant depression, etc (<xref ref-type="bibr" rid="B9">Blanco et al., 2010</xref>). Since there are no direct medications available for idiopathic HCC, including those induced by NAFLD, and given that the pathogenesis of non-idiopathic HCC differs from idiopathic HCC, drug repurposing to target these mechanisms can significantly reduce the development time and cost of new treatments. This also ensures better patient outcomes using already approved drugs with known safety profiles. These findings indicate promising therapeutic paths that should be further investigated in clinical settings by experts.</p>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>In conclusion, ensemble feature selection framework used in this study improves the discrimination and stability of the final selected features. Using an ensemble feature selection approach, this study successfully identified key biomarkers, including <italic>ABAT, C8B, FBXL3,</italic> and <italic>ZFP1</italic>, providing valuable insights into NAFLD to HCC disease progression. A drug repurposing approach identified therapeutic agents, including Diosmin, Esculin, and Lapatinib, that were found to be effective against these marker genes. These findings offer a strong foundation for future research and therapeutic development in the treatment of NAFLD-mediated HCC. The integration of biomarker prediction with drug repurposing could enhance precision medicine approaches, paving the way for more effective and targeted treatments.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s13">Supplementary Material</xref>, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec sec-type="ethics-statement" id="s7">
<title>Ethics statement</title>
<p>Ethical approval was not required for the study involving humans in accordance with the local legislation and institutional requirements. Written informed consent to participate in this study was not required from the participants or the participants&#x2019; legal guardians/next of kin in accordance with the national legislation and the institutional requirements.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>SG: Conceptualization, Data curation, Formal Analysis, Investigation, Methodology, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review and editing. SM: Supervision, Writing &#x2013; review and editing. ST: Supervision, Writing &#x2013; review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research and/or publication of this article. SG acknowledged the University of North Bengal for research assistant fellowship [University fellowship No- Ref. no.4304/R-2022].</p>
</sec>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s11">
<title>Generative AI statement</title>
<p>The authors declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s12">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s13">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fbinf.2025.1522401/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fbinf.2025.1522401/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table1.docx" id="SM1" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table2.docx" id="SM2" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table5.docx" id="SM3" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table3.docx" id="SM4" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table4.docx" id="SM5" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table6.docx" id="SM6" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Azari</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Nazari</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Mohit</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Asadnia</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Maftooh</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Nassiri</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Machine learning algorithms reveal potential miRNAs biomarkers in gastric cancer</article-title>. <source>Sci. Rep.</source> <volume>13</volume>, <fpage>6147</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-023-32332-x</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barrera</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cesar</surname>
<given-names>R. M.</given-names>
<suffix>Jr</suffix>
</name>
<name>
<surname>Humes</surname>
<given-names>C.</given-names>
<suffix>Jr</suffix>
</name>
<name>
<surname>Martins</surname>
<given-names>D. C.</given-names>
<suffix>Jr</suffix>
</name>
<name>
<surname>Patr&#xe3;o</surname>
<given-names>D. F.</given-names>
</name>
<name>
<surname>Silva</surname>
<given-names>P. J.</given-names>
</name>
<etal/>
</person-group> (<year>2007</year>). <article-title>A feature selection approach for identification of signature genes from SAGE data</article-title>. <source>BMC Bioinforma.</source> <volume>8</volume>, <fpage>169</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-8-169</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barrett</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wilhite</surname>
<given-names>S. E.</given-names>
</name>
<name>
<surname>Ledoux</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Evangelista</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>I. F.</given-names>
</name>
<name>
<surname>Tomashevsky</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>NCBI GEO: archive for functional genomics data sets&#x2014;update</article-title>. <source>Nucleic Acids Res.</source> <volume>41</volume> (<issue>D1</issue>), <fpage>D991</fpage>&#x2013;<lpage>D995</lpage>. <pub-id pub-id-type="doi">10.1093/NAR/GKS1193</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bellotti</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Nouretdinov</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Gammerman</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Feature selection</article-title>. <source>Conformal Predict. Reliab. Mach. Learn. Theory, Adapt. Appl.</source>, <fpage>115</fpage>&#x2013;<lpage>130</lpage>. <pub-id pub-id-type="doi">10.1016/B978-0-12-398537-8.00006-7</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bennasar</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hicks</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Setchi</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Feature selection using joint mutual information maximisation</article-title>. <source>Expert Syst. Appl.</source> <volume>42</volume> (<issue>22</issue>), <fpage>8520</fpage>&#x2013;<lpage>8532</lpage>. <pub-id pub-id-type="doi">10.1016/J.ESWA.2015.07.007</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Besse</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Bruni</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Donti</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Graham</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Craigen</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>The GABA transaminase, ABAT, is essential for mitochondrial nucleoside metabolism</article-title>. <source>Cell Metab.</source> <volume>21</volume> (<issue>3</issue>), <fpage>417</fpage>&#x2013;<lpage>427</lpage>. <pub-id pub-id-type="doi">10.1016/j.cmet.2015.02.008</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bhandari</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Walambe</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Kotecha</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Khare</surname>
<given-names>S. P.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A comprehensive survey on computational learning methods for analysis of gene expression data</article-title>. <source>Front. Mol. Biosci.</source> <volume>9</volume>, <fpage>907150</fpage>. <pub-id pub-id-type="doi">10.3389/fmolb.2022.907150</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="web">
<collab>Bioconductor</collab> (<year>2023</year>). <article-title>Normalizing Affy microarray data</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="http://www.bioconductor.org/">http://www.bioconductor.org/</ext-link>(Accessed August 13, 2023)</comment>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Blanco</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Heimberg</surname>
<given-names>R. G.</given-names>
</name>
<name>
<surname>Schneier</surname>
<given-names>F. R.</given-names>
</name>
<name>
<surname>Fresco</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Turk</surname>
<given-names>C. L.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>A placebo-controlled trial of phenelzine, cognitive behavioral group therapy and their combination for social anxiety disorder</article-title>. <source>Archives general psychiatry</source> <volume>67</volume> (<issue>3</issue>), <fpage>286</fpage>. <pub-id pub-id-type="doi">10.1001/ARCHGENPSYCHIATRY.2010.11</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bol&#xf3;n-Canedo</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Alonso-Betanzos</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Ensembles for feature selection: a review and future trends</article-title>. <source>Inf. Fusion</source> <volume>52</volume>, <fpage>1</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1016/J.INFFUS.2018.11.008</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bouchlaghem</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Akhiat</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Amjad</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Feature selection: a review and comparative study</article-title>. <source>E3S Web Conf.</source> <volume>351</volume>, <fpage>01046</fpage>&#x2013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1051/e3sconf/202235101046</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cai</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Pharmacological activities of esculin and esculetin: a review</article-title>. <source>Med. (United States)</source> <volume>102</volume> (<issue>40</issue>), <fpage>E35306</fpage>. <pub-id pub-id-type="doi">10.1097/MD.0000000000035306</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cannon</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Stevenson</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Stahl</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Basu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Coffman</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kiwala</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>DGIdb 5.0: rebuilding the drug&#x2013;gene interaction database for precision medicine and drug discovery platforms</article-title>. <source>Nucleic Acids Res.</source> <volume>52</volume> (<issue>D1</issue>), <fpage>D1227</fpage>&#x2013;<lpage>D1235</lpage>. <pub-id pub-id-type="doi">10.1093/NAR/GKAD1040</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Castellanos-Garz&#xf3;n</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Ramos</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>L&#xf3;pez-S&#xe1;nchez</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>de Paz</surname>
<given-names>J. F.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>An ensemble approach for gene selection in gene expression data</article-title>. <source>Adv. Intelligent Syst. Comput.</source> <volume>616</volume>, <fpage>237</fpage>&#x2013;<lpage>247</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-319-60816-7_29</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Che</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Bai</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Maximum relevance minimum common redundancy feature selection for nonlinear data</article-title>. <source>Inf. Sci.</source> <volume>409&#x2013;410</volume>, <fpage>68</fpage>&#x2013;<lpage>86</lpage>. <pub-id pub-id-type="doi">10.1016/J.INS.2017.05.013</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Applications of multi-omics analysis in human diseases</article-title>. <source>MedComm</source> <volume>4</volume> (<issue>4</issue>), <fpage>1</fpage>&#x2013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1002/mco2.315</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Mei</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Classification prediction of breast cancer based on machine learning</article-title>. <source>Comput. Intell. Neurosci.</source> <volume>2023</volume>, <fpage>1</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1155/2023/6530719</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Liao</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xun</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Loss of ABAT-mediated GABAergic system promotes basal-like breast cancer progression by activating Ca2&#x2b;-NFAT1 axis</article-title>. <source>Theranostics</source> <volume>9</volume> (<issue>1</issue>), <fpage>34</fpage>&#x2013;<lpage>47</lpage>. <pub-id pub-id-type="doi">10.7150/thno.29407</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Colaprico</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Silva</surname>
<given-names>T. C.</given-names>
</name>
<name>
<surname>Olsen</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Garofano</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Cava</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Garolini</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>TCGAbiolinks: an R/Bioconductor package for integrative analysis of TCGA data</article-title>. <source>Nucleic Acids Res.</source> <volume>44</volume> (<issue>8</issue>), <fpage>e71</fpage>. <pub-id pub-id-type="doi">10.1093/NAR/GKV1507</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>de Amorim</surname>
<given-names>L. B. V.</given-names>
</name>
<name>
<surname>Cavalcanti</surname>
<given-names>G. D. C.</given-names>
</name>
<name>
<surname>Cruz</surname>
<given-names>R. M. O.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>The choice of scaling technique matters for classification performance</article-title>. <source>Appl. Soft Comput.</source> <volume>133</volume>, <fpage>109924</fpage>. <pub-id pub-id-type="doi">10.1016/J.ASOC.2022.109924</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Deprince</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hennuyer</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Kooijman</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Pronk</surname>
<given-names>A. C. M.</given-names>
</name>
<name>
<surname>Baug&#xe9;</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Lienard</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Apolipoprotein F is reduced in humans with steatosis and controls plasma triglyceride-rich lipoprotein metabolism</article-title>. <source>Hepatology</source> <volume>77</volume> (<issue>4</issue>), <fpage>1287</fpage>&#x2013;<lpage>1302</lpage>. <pub-id pub-id-type="doi">10.1002/hep.32631</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Deshpande</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Addess</surname>
<given-names>K. J.</given-names>
</name>
<name>
<surname>Bluhm</surname>
<given-names>W. F.</given-names>
</name>
<name>
<surname>Merino-Ott</surname>
<given-names>J. C.</given-names>
</name>
<name>
<surname>Townsend-Merino</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Q.</given-names>
</name>
<etal/>
</person-group> (<year>2005</year>). <article-title>The RCSB Protein Data Bank: a redesigned query system and relational database based on the mmCIF schema</article-title>. <source>Nucleic Acids Res.</source> <volume>33</volume> (<issue>Suppl. l_1</issue>), <fpage>D233</fpage>&#x2013;<lpage>D237</lpage>. <pub-id pub-id-type="doi">10.1093/NAR/GKI057</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>El-Hashash</surname>
<given-names>E. F.</given-names>
</name>
<name>
<surname>Shiekh</surname>
<given-names>R. H. A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A comparison of the Pearson, spearman rank and Kendall Tau correlation coefficients using quantitative variables</article-title>. <source>Asian J. Probab. Statistics</source> <fpage>36</fpage>&#x2013;<lpage>48</lpage>. <pub-id pub-id-type="doi">10.9734/ajpas/2022/v20i3425</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Est&#xe9;vez</surname>
<given-names>P. A.</given-names>
</name>
<name>
<surname>Tesmer</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Perez</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zurada</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Normalized mutual information feature selection</article-title>. <source>IEEE Trans. Neural Netw.</source> <volume>20</volume> (<issue>2</issue>), <fpage>189</fpage>&#x2013;<lpage>201</lpage>. <pub-id pub-id-type="doi">10.1109/TNN.2008.2005601</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fagiani</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Di Marino</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Romagnoli</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Travelli</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Voltan</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Di Cesare Mannelli</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Molecular regulations of circadian rhythm and implications for physiology and diseases</article-title>. <source>Signal Transduct. Target. Ther.</source> <volume>7</volume> (<issue>1</issue>), <fpage>41</fpage>&#x2013;<lpage>20</lpage>. <pub-id pub-id-type="doi">10.1038/s41392-022-00899-y</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Class-specific mutual information variation for feature selection</article-title>. <source>Pattern Recognit.</source> <volume>79</volume>, <fpage>328</fpage>&#x2013;<lpage>339</lpage>. <pub-id pub-id-type="doi">10.1016/j.patcog.2018.02.020</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Xiang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lei</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Regulation of gamma-aminobutyric acid transaminase expression and its clinical significance in hepatocellular carcinoma</article-title>. <source>Front. Oncol.</source> <volume>12</volume> (<issue>June</issue>), <fpage>1</fpage>&#x2013;<lpage>16</lpage>. <pub-id pub-id-type="doi">10.3389/fonc.2022.879810</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Harder</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Damm</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Maple</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Reboul</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Xiang</surname>
<given-names>J. Y.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>OPLS3: a force field providing broad coverage of drug-like small molecules and proteins</article-title>. <source>J. Chem. Theory Comput.</source> <volume>12</volume> (<issue>1</issue>), <fpage>281</fpage>&#x2013;<lpage>296</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jctc.5b00864</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hartal-Benishay</surname>
<given-names>L. H.</given-names>
</name>
<name>
<surname>Saadi</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Toubiana</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shaked</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Lalzar</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Abu Hatoum</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>MBTPS1 regulates proliferation of colorectal cancer primarily through its action on sterol regulatory element-binding proteins</article-title>. <source>Front. Oncol.</source> <volume>12</volume> <fpage>1</fpage>&#x2013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.3389/fonc.2022.1004014</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huber</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Papp</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>A. B.</given-names>
</name>
<name>
<surname>Henriksson</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Jordan</surname>
<given-names>S. D.</given-names>
</name>
<name>
<surname>Kriebs</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2016a</year>). <article-title>CRY2 and FBXL3 cooperatively degrade c-MYC</article-title>. <source>Mol. cell</source> <volume>64</volume> (<issue>4</issue>), <fpage>774</fpage>&#x2013;<lpage>789</lpage>. <pub-id pub-id-type="doi">10.1016/J.MOLCEL.2016.10.012</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huber</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Papp</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>A. B.</given-names>
</name>
<name>
<surname>Henriksson</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Jordan</surname>
<given-names>S. D.</given-names>
</name>
<name>
<surname>Kriebs</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2016b</year>). <article-title>CRY2 and FBXL3 cooperatively degrade c-MYC</article-title>. <source>Mol. Cell</source> <volume>64</volume> (<issue>4</issue>), <fpage>774</fpage>&#x2013;<lpage>789</lpage>. <pub-id pub-id-type="doi">10.1016/j.molcel.2016.10.012</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huwait</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Mobashir</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Potential and therapeutic roles of Diosmin in human diseases</article-title>. <source>Biomedicines</source> <volume>10</volume> (<issue>5</issue>), <fpage>1076</fpage>. <pub-id pub-id-type="doi">10.3390/biomedicines10051076</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jendele</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Krivak</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Skoda</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Novotny</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hoksza</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>PrankWeb: a web server for ligand binding site prediction and visualization</article-title>. <source>Nucleic Acids Res.</source> <volume>47</volume> (<issue>W1</issue>), <fpage>W345</fpage>&#x2013;<lpage>W349</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkz424</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kearney</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>&#x2018;Cram&#xe9;r&#x2019;s V&#x2019;</article-title>. <source>The SAGE Encyclopedia of Communication Research Methods</source>. <publisher-loc>Thousand Oaks, CA</publisher-loc>: <publisher-name>SAGE Publications, Inc.</publisher-name> <volume>4</volume>, <fpage>290</fpage>. <pub-id pub-id-type="doi">10.4135/9781483381411</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khatun</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Akter</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Islam</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Uddin</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Talukder</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Kamruzzaman</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Cancer classification utilizing voting classifier with ensemble feature selection method and transcriptomic data</article-title>. <source>Genes</source> <volume>14</volume> (<issue>9</issue>), <fpage>1802</fpage>. <pub-id pub-id-type="doi">10.3390/GENES14091802</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kiefer</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Arnold</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kunzli</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bordoli</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Schwede</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>The SWISS-MODEL Repository and associated resources</article-title>. <source>Nucleic Acids Res.</source> <volume>37</volume> (<issue>Suppl. l_1</issue>), <fpage>D387</fpage>&#x2013;<lpage>D392</lpage>. <pub-id pub-id-type="doi">10.1093/NAR/GKN750</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Koul</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Manvi</surname>
<given-names>S. S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Ensemble feature selection from cancer gene expression data using mutual information and recursive feature elimination</article-title>. <fpage>1</fpage>, <lpage>6</lpage>. <pub-id pub-id-type="doi">10.1109/ICAECC50550.2020.9339518</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Krishnamurthy</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Grimshaw</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Axson</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Choe</surname>
<given-names>S. H.</given-names>
</name>
<name>
<surname>Miller</surname>
<given-names>J. E.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Drug repurposing: a systematic review on root causes, barriers and facilitators</article-title>. <source>BMC Health Serv. Res.</source> <volume>22</volume>, <fpage>970</fpage>. <pub-id pub-id-type="doi">10.1186/s12913-022-08272-z</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lagana</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Salomao</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Remotti</surname>
<given-names>H. E.</given-names>
</name>
<name>
<surname>Knisely</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Moreira</surname>
<given-names>R. K.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Bile salt export pump: a sensitive and specific immunohistochemical marker of hepatocellular carcinoma</article-title>. <source>Histopathology</source> <volume>66</volume> (<issue>4</issue>), <fpage>598</fpage>&#x2013;<lpage>602</lpage>. <pub-id pub-id-type="doi">10.1111/his.12601</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>L&#xe9;veill&#xe9;</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Estall</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Mitochondrial dysfunction in the transition from NASH to HCC</article-title>. <source>Metabolites</source> <volume>9</volume> (<issue>10</issue>), <fpage>233</fpage>. <pub-id pub-id-type="doi">10.3390/metabo9100233</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Structures and biological functions of zinc finger proteins and their roles in hepatocellular carcinoma</article-title>. <source>Biomark. Res.</source> <volume>10</volume> (<issue>1</issue>), <fpage>2</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1186/s40364-021-00345-1</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Sanchez-Burgos</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>H&#xfc;hn</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Fernandez-Capetillo</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>The Drug Repurposing Encyclopedia (DRE): a web server for systematic drug repurposing across 20 organisms</article-title>. <source>bioRxiv</source>, <fpage>2023</fpage>. <pub-id pub-id-type="doi">10.1101/2023.03.10.532084</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Mansmann</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hornung</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Benchmark study of feature selection strategies for multi-omics data</article-title>. <source>BMC Bioinforma.</source> <volume>23</volume> (<issue>1</issue>), <fpage>412</fpage>&#x2013;<lpage>418</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-022-04962-x</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Conditional infomax learning: an integrated framework for feature extraction and fusion</article-title>. <source>Lect. Notes Comput. Sci. Incl. Subser. Lect. Notes Artif. Intell. Lect. Notes Bioinforma. 3951 LNCS</source>, <fpage>68</fpage>&#x2013;<lpage>82</lpage>. <pub-id pub-id-type="doi">10.1007/11744023_6</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Love</surname>
<given-names>M. I.</given-names>
</name>
<name>
<surname>Anders</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Huber</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Moderated estimation of fold change and dispersion for RNA-seq data with DESeq2</article-title>. <source>Genome Biol.</source> <volume>15</volume>, <fpage>550</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-014-0550-8</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meyer-Baese</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Schmid</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Statistical and syntactic pattern recognition</article-title>. <fpage>151</fpage>&#x2013;<lpage>196</lpage>. <pub-id pub-id-type="doi">10.1016/B978-0-12-409545-8.00006-6</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mohammed</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mboya</surname>
<given-names>I. B.</given-names>
</name>
<name>
<surname>Mwambi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Elbashir</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Omolo</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Predictors of colorectal cancer survival using cox regression and random survival forests models based on gene expression data</article-title>. <source>PLoS ONE</source> <volume>16</volume> (<issue>12</issue>), <fpage>e0261625</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0261625</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moritz</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bartz-Beielstein</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>imputeTS: time series missing value imputation in R</article-title>. <source>R J.</source> <volume>9</volume> (<issue>1</issue>), <fpage>207</fpage>&#x2013;<lpage>218</lpage>. <pub-id pub-id-type="doi">10.32614/rj-2017-009</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Muthukrishnan</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Rohini</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>LASSO: a feature selection technique in predictive modeling for machine learning</article-title>. <fpage>18</fpage>&#x2013;<lpage>20</lpage>. <pub-id pub-id-type="doi">10.1109/ICACA.2016.7887916</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nasir</surname>
<given-names>I. M.</given-names>
</name>
<name>
<surname>Khan</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Yasmin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Shah</surname>
<given-names>J. H.</given-names>
</name>
<name>
<surname>Gabryel</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Scherer</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Pearson correlation-based feature selection for document classification using balanced training</article-title>. <source>Sensors</source> <volume>20</volume> (<issue>23</issue>), <fpage>6793</fpage>. <pub-id pub-id-type="doi">10.3390/S20236793</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nazir</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Haq</surname>
<given-names>I. u.</given-names>
</name>
<name>
<surname>AlQahtani</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Jadoon</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Dahshan</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Machine learning-based lung cancer detection using multiview image registration and fusion</article-title>. <source>J. Sensors</source> <volume>2023</volume>. <pub-id pub-id-type="doi">10.1155/2023/6683438</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Opdam</surname>
<given-names>F. L.</given-names>
</name>
<name>
<surname>Guchelaar</surname>
<given-names>H. J.</given-names>
</name>
<name>
<surname>Beijnen</surname>
<given-names>J. H.</given-names>
</name>
<name>
<surname>Schellens</surname>
<given-names>J. H.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Lapatinib for advanced or metastatic breast cancer</article-title>. <source>Oncol.</source> <volume>17</volume> (<issue>4</issue>), <fpage>536</fpage>&#x2013;<lpage>542</lpage>. <pub-id pub-id-type="doi">10.1634/theoncologist.2011-0461</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Otchere</surname>
<given-names>D. A.</given-names>
</name>
<name>
<surname>Ganat</surname>
<given-names>T. O. A.</given-names>
</name>
<name>
<surname>Ojero</surname>
<given-names>J. O.</given-names>
</name>
<name>
<surname>Tackie-Otoo</surname>
<given-names>B. N.</given-names>
</name>
<name>
<surname>Taki</surname>
<given-names>M. Y.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Application of gradient boosting regression model for the evaluation of feature selection techniques in improving reservoir characterisation predictions</article-title>. <source>J. Petroleum Sci. Eng.</source> <volume>208</volume>, <fpage>109244</fpage>. <pub-id pub-id-type="doi">10.1016/j.petrol.2021.109244</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Parkinson</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Kapushesky</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Shojatalab</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Abeygunawardena</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Coulson</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Farne</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2007</year>). <article-title>ArrayExpress&#x2014;a public database of microarray experiments and gene expression profiles</article-title>. <source>Nucleic Acids Res.</source> <volume>35</volume> (<issue>Database issue</issue>), <fpage>D747</fpage>&#x2013;<lpage>D750</lpage>. <pub-id pub-id-type="doi">10.1093/NAR/GKL995</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paul</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Drineas</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Feature selection for ridge regression with provable guarantees</article-title>. <source>Neural Comput.</source> <volume>28</volume> (<issue>4</issue>), <fpage>716</fpage>&#x2013;<lpage>742</lpage>. <pub-id pub-id-type="doi">10.1162/NECO_A_00816</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pawar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chowdhury</surname>
<given-names>O. R.</given-names>
</name>
<name>
<surname>Salvi</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A narrative review of survival analysis in oncology using R</article-title>. <source>Cancer Res. Statistics, Treat.</source> <volume>5</volume> (<issue>3</issue>), <fpage>554</fpage>&#x2013;<lpage>561</lpage>. <pub-id pub-id-type="doi">10.4103/CRST.CRST_230_22</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Phan</surname>
<given-names>L. M.</given-names>
</name>
<name>
<surname>Yeung</surname>
<given-names>S. C. J.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>M. H.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Cancer metabolic reprogramming: importance, main features, and potentials for precise targeted anti-cancer therapies</article-title>. <source>Cancer Biol. Med.</source> <volume>11</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>19</lpage>. <pub-id pub-id-type="doi">10.7497/j.issn.2095-3941.2014.01.001</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Posekany</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Felsenstein</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Sykacek</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Biological assessment of robust noise models in microarray data analysis</article-title>. <source>Bioinformatics</source> <volume>27</volume> (<issue>6</issue>), <fpage>807</fpage>&#x2013;<lpage>814</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btr018</pub-id>
</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ritchie</surname>
<given-names>M. E.</given-names>
</name>
<name>
<surname>Phipson</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Law</surname>
<given-names>C. W.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Limma powers differential expression analyses for RNA-sequencing and microarray studies</article-title>. <source>Nucleic Acids Res.</source> <volume>43</volume> (<issue>7</issue>), <fpage>e47</fpage>. <pub-id pub-id-type="doi">10.1093/NAR/GKV007</pub-id>
</citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sarkar</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Saha</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Sarkar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Maulik</surname>
<given-names>U.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Machine learning integrated ensemble of feature selection methods followed by survival analysis for predicting breast cancer subtype specific miRNA biomarkers</article-title>. <source>Comput. Biol. Med.</source> <volume>131</volume>, <fpage>104244</fpage>. <pub-id pub-id-type="doi">10.1016/J.COMPBIOMED.2021.104244</pub-id>
</citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Saseendran</surname>
<given-names>A. T.</given-names>
</name>
<name>
<surname>Setia</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Chhabria</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Chakraborty</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Roy</surname>
<given-names>A. B.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Impact of noise in dataset on machine learning algorithms</article-title>. <source>Mach. Learn. Module CS7CS4/CS4404</source>. <pub-id pub-id-type="doi">10.13140/RG.2.2.25669.91369</pub-id>
</citation>
</ref>
<ref id="B62">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schober</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Schwarte</surname>
<given-names>L. A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Correlation coefficients: appropriate use and interpretation</article-title>. <source>Anesth. Analgesia</source> <volume>126</volume> (<issue>5</issue>), <fpage>1763</fpage>&#x2013;<lpage>1768</lpage>. <pub-id pub-id-type="doi">10.1213/ANE.0000000000002864</pub-id>
</citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sherman</surname>
<given-names>B. T.</given-names>
</name>
<name>
<surname>Hao</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Qiu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jiao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Baseler</surname>
<given-names>M. W.</given-names>
</name>
<name>
<surname>Lane</surname>
<given-names>H. C.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>DAVID: a web server for functional enrichment analysis and functional annotation of gene lists (2021 update)</article-title>. <source>Nucleic Acids Res.</source> <volume>50</volume> (<issue>W1</issue>), <fpage>W216</fpage>&#x2013;<lpage>W221</lpage>. <pub-id pub-id-type="doi">10.1093/NAR/GKAC194</pub-id>
</citation>
</ref>
<ref id="B64">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sohail</surname>
<given-names>M. I.</given-names>
</name>
<name>
<surname>D&#xf6;nmez-Cakil</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sz&#xf6;ll&#x151;si</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Stockner</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chiba</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>The bile salt export pump: molecular structure, study models and small-molecule drugs for the treatment of inherited bsep deficiencies</article-title>. <source>Int. J. Mol. Sci.</source> <volume>22</volume> (<issue>13</issue>), <fpage>7058</fpage>. <pub-id pub-id-type="doi">10.3390/ijms22137058</pub-id>
</citation>
</ref>
<ref id="B65">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stra&#x15b;</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Ma&#x142;kowski</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Tronina</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Hepatocellular carcinoma in patients with non-alcoholic steatohepatitis - epidemiology, risk factors, clinical implications and treatment</article-title>. <source>Clin. Exp. Hepatology</source> <volume>6</volume> (<issue>3</issue>), <fpage>170</fpage>&#x2013;<lpage>175</lpage>. <pub-id pub-id-type="doi">10.5114/ceh.2020.99506</pub-id>
</citation>
</ref>
<ref id="B66">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Talukder</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Islam</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Uddin</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Akhter</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hasan</surname>
<given-names>K. F.</given-names>
</name>
<name>
<surname>Moni</surname>
<given-names>M. A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Machine learning-based lung and colon cancer detection using deep feature extraction and ensemble learning</article-title>. <source>Expert Syst. Appl.</source> <volume>205</volume>, <fpage>117695</fpage>. <pub-id pub-id-type="doi">10.1016/J.ESWA.2022.117695</pub-id>
</citation>
</ref>
<ref id="B67">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>SRplot: a free online platform for data visualization and graphing</article-title>. <source>PLOS ONE</source> <volume>18</volume> (<issue>11</issue>), <fpage>e0294236</fpage>. <pub-id pub-id-type="doi">10.1371/JOURNAL.PONE.0294236</pub-id>
</citation>
</ref>
<ref id="B68">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Teng</surname>
<given-names>M. L. P.</given-names>
</name>
<name>
<surname>Ng</surname>
<given-names>C. H.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>D. Q.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>K. E.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Lim</surname>
<given-names>W. H.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Global incidence and prevalence of nonalcoholic fatty liver disease</article-title>. <source>Clin. Mol. Hepatology</source> <volume>29</volume> (<issue>Suppl. 1</issue>), <fpage>32</fpage>&#x2013;<lpage>42</lpage>. <pub-id pub-id-type="doi">10.3350/CMH.2022.0365</pub-id>
</citation>
</ref>
<ref id="B69">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Therneau</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A package for survival analysis in R</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://cran.r-project.org/package=survival">https://cran.r-project.org/package&#x3d;survival</ext-link>.</comment>
</citation>
</ref>
<ref id="B70">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Urbanowicz</surname>
<given-names>R. J.</given-names>
</name>
<name>
<surname>Meeker</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>La Cava</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Olson</surname>
<given-names>R. S.</given-names>
</name>
<name>
<surname>Moore</surname>
<given-names>J. H.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Relief-based feature selection: introduction and review</article-title>. <source>J. Biomed. Inf.</source> <volume>85</volume>, <fpage>189</fpage>&#x2013;<lpage>203</lpage>. <pub-id pub-id-type="doi">10.1016/j.jbi.2018.07.014</pub-id>
</citation>
</ref>
<ref id="B71">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Valencia</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Lillo</surname>
<given-names>R. E.</given-names>
</name>
<name>
<surname>Romo</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A Kendall correlation coefficient between functional data</article-title>. <source>Adv. Data Analysis Classif.</source> <volume>13</volume> (<issue>4</issue>), <fpage>1083</fpage>&#x2013;<lpage>1103</lpage>. <pub-id pub-id-type="doi">10.1007/s11634-019-00360-z</pub-id>
</citation>
</ref>
<ref id="B72">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Suzek</surname>
<given-names>T. O.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Bryant</surname>
<given-names>S. H.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>PubChem: a public information system for analyzing bioactivities of small molecules</article-title>. <source>Nucleic Acids Res.</source> <volume>37</volume> (<issue>Suppl. l_2</issue>), <fpage>W623</fpage>&#x2013;<lpage>W633</lpage>. <pub-id pub-id-type="doi">10.1093/NAR/GKP456</pub-id>
</citation>
</ref>
<ref id="B73">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xiao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yeung</surname>
<given-names>C. L. S.</given-names>
</name>
<name>
<surname>Yam</surname>
<given-names>J. W. P.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>An update on the role of complement in hepatocellular carcinoma</article-title>. <source>Front. Immunol.</source> <volume>13</volume> (<issue>October 2022</issue>), <fpage>1</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.3389/fimmu.2022.1007382</pub-id>
</citation>
</ref>
<ref id="B74">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xie</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Kropiwnicki</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Wojciechowicz</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>Jagodnik</surname>
<given-names>K. M.</given-names>
</name>
<name>
<surname>Shu</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Bailey</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Getting started with LINCS datasets and tools</article-title>. <source>Curr. Protoc.</source> <volume>2</volume> (<issue>7</issue>), <fpage>1</fpage>&#x2013;<lpage>90</lpage>. <pub-id pub-id-type="doi">10.1002/cpz1.487</pub-id>
</citation>
</ref>
<ref id="B75">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>From NAFLD to HCC: advances in noninvasive diagnosis</article-title>. <source>Biomed. Pharmacother.</source> <volume>165</volume>, <fpage>115028</fpage>. <pub-id pub-id-type="doi">10.1016/j.biopha.2023.115028</pub-id>
</citation>
</ref>
<ref id="B76">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>The emerging factors and treatment options for nafld-related hepatocellular carcinoma</article-title>. <source>Cancers</source> <volume>13</volume>, <fpage>3503</fpage>&#x2013;<lpage>3515</lpage>. <pub-id pub-id-type="doi">10.2147/CMAR.S302917</pub-id>
</citation>
</ref>
<ref id="B77">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Qiu</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>High expression levels of centromere protein A plus upregulation of the phosphatidylinositol 3-kinase/Akt/mammalian target of rapamycin signaling pathway affect chemotherapy response and prognosis in patients with breast cancer</article-title>. <source>Oncol. Lett.</source> <volume>21</volume> (<issue>5</issue>), <fpage>410</fpage>. <pub-id pub-id-type="doi">10.3892/ol.2021.12671</pub-id>
</citation>
</ref>
<ref id="B78">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>C8b in complement and coagulation cascades signaling pathway is a predictor for survival in hbv-related hepatocellular carcinoma patients</article-title>. <source>Cancer Manag. Res.</source> <volume>13</volume>, <fpage>3503</fpage>&#x2013;<lpage>3515</lpage>. <pub-id pub-id-type="doi">10.2147/CMAR.S302917</pub-id>
</citation>
</ref>
<ref id="B79">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Mitochondrial metabolic dysfunction and non-alcoholic fatty liver disease: new insights from pathogenic mechanisms to clinically targeted therapy</article-title>. <source>J. Transl. Med.</source> <volume>21</volume> (<issue>1</issue>), <fpage>510</fpage>&#x2013;<lpage>521</lpage>. <pub-id pub-id-type="doi">10.1186/s12967-023-04367-1</pub-id>
</citation>
</ref>
</ref-list>
<sec id="s14">
<title>Glossary</title>
<def-list>
<def-item>
<term id="G1-fbinf.2025.1522401">
<bold>NAFLD</bold>
</term>
<def>
<p>Non-Alcoholic Fatty Liver Disease</p>
</def>
</def-item>
<def-item>
<term id="G2-fbinf.2025.1522401">
<bold>NASH</bold>
</term>
<def>
<p>Non-Alcoholic Steatohepatitis</p>
</def>
</def-item>
<def-item>
<term id="G3-fbinf.2025.1522401">
<bold>HCC</bold>
</term>
<def>
<p>Hepatocellular Carcinoma</p>
</def>
</def-item>
<def-item>
<term id="G4-fbinf.2025.1522401">
<bold>ML</bold>
</term>
<def>
<p>Machine Learning</p>
</def>
</def-item>
<def-item>
<term id="G5-fbinf.2025.1522401">
<bold>SVM</bold>
</term>
<def>
<p>Support Vector Machine</p>
</def>
</def-item>
<def-item>
<term id="G6-fbinf.2025.1522401">
<bold>ANN</bold>
</term>
<def>
<p>Artificial Neural Network</p>
</def>
</def-item>
<def-item>
<term id="G7-fbinf.2025.1522401">
<bold>KNN</bold>
</term>
<def>
<p>k-Nearest Neighbors</p>
</def>
</def-item>
<def-item>
<term id="G8-fbinf.2025.1522401">
<bold>DISCR</bold>
</term>
<def>
<p>Discriminant Analysis</p>
</def>
</def-item>
<def-item>
<term id="G9-fbinf.2025.1522401">
<bold>RF</bold>
</term>
<def>
<p>Random Forest</p>
</def>
</def-item>
<def-item>
<term id="G10-fbinf.2025.1522401">
<bold>DT</bold>
</term>
<def>
<p>Decision Tree</p>
</def>
</def-item>
<def-item>
<term id="G11-fbinf.2025.1522401">
<bold>NB</bold>
</term>
<def>
<p>Naive Bayes</p>
</def>
</def-item>
<def-item>
<term id="G12-fbinf.2025.1522401">
<bold>CIFE</bold>
</term>
<def>
<p>Conditional Informative Feature Extraction</p>
</def>
</def-item>
<def-item>
<term id="G13-fbinf.2025.1522401">
<bold>JMI</bold>
</term>
<def>
<p>Joint Mutual Information</p>
</def>
</def-item>
<def-item>
<term id="G14-fbinf.2025.1522401">
<bold>MIM</bold>
</term>
<def>
<p>Mutual Information Maximization/Maximum Relevance</p>
</def>
</def-item>
<def-item>
<term id="G15-fbinf.2025.1522401">
<bold>LASSO</bold>
</term>
<def>
<p>Least Absolute Shrinkage and Selection Operator</p>
</def>
</def-item>
<def-item>
<term id="G16-fbinf.2025.1522401">
<bold>GO</bold>
</term>
<def>
<p>Gene Ontology</p>
</def>
</def-item>
<def-item>
<term id="G17-fbinf.2025.1522401">
<bold>KEGG</bold>
</term>
<def>
<p>Kyoto Encyclopedia of Genes and Genomes</p>
</def>
</def-item>
<def-item>
<term id="G18-fbinf.2025.1522401">
<bold>ER</bold>
</term>
<def>
<p>Endoplasmic Reticulum</p>
</def>
</def-item>
<def-item>
<term id="G19-fbinf.2025.1522401">
<bold>GEO</bold>
</term>
<def>
<p>Gene Expression Omnibus</p>
</def>
</def-item>
<def-item>
<term id="G20-fbinf.2025.1522401">
<bold>NCBI</bold>
</term>
<def>
<p>National Center for Biotechnology Information</p>
</def>
</def-item>
<def-item>
<term id="G21-fbinf.2025.1522401">
<bold>RMA</bold>
</term>
<def>
<p>Robust Multichip Averaging</p>
</def>
</def-item>
<def-item>
<term id="G22-fbinf.2025.1522401">
<bold>RMR</bold>
</term>
<def>
<p>Relevant Mutual Redundancy</p>
</def>
</def-item>
<def-item>
<term id="G23-fbinf.2025.1522401">
<bold>DISR</bold>
</term>
<def>
<p>Double Input Symmetrical Relevance</p>
</def>
</def-item>
<def-item>
<term id="G24-fbinf.2025.1522401">
<bold>NMIFS</bold>
</term>
<def>
<p>Normalized Mutual Information Feature Selection</p>
</def>
</def-item>
<def-item>
<term id="G25-fbinf.2025.1522401">
<bold>DAVID</bold>
</term>
<def>
<p>Database for Annotation, Visualization, and Integrated Discovery</p>
</def>
</def-item>
<def-item>
<term id="G26-fbinf.2025.1522401">
<bold>FDR</bold>
</term>
<def>
<p>False Discovery Rate</p>
</def>
</def-item>
<def-item>
<term id="G27-fbinf.2025.1522401">
<bold>TCGA-LIHC</bold>
</term>
<def>
<p>The Cancer Genome AtlasLiver Hepatocellular Carcinoma</p>
</def>
</def-item>
<def-item>
<term id="G28-fbinf.2025.1522401">
<bold>HR</bold>
</term>
<def>
<p>Hazard Ratio</p>
</def>
</def-item>
<def-item>
<term id="G29-fbinf.2025.1522401">
<bold>CI</bold>
</term>
<def>
<p>Confidence Interval</p>
</def>
</def-item>
<def-item>
<term id="G30-fbinf.2025.1522401">
<bold>DGIdb</bold>
</term>
<def>
<p>The Drug Gene Interaction Database</p>
</def>
</def-item>
<def-item>
<term id="G31-fbinf.2025.1522401">
<bold>MSigDB</bold>
</term>
<def>
<p>Molecular Signatures Database</p>
</def>
</def-item>
<def-item>
<term id="G32-fbinf.2025.1522401">
<bold>DREIMT</bold>
</term>
<def>
<p>Drug Response Element Interactions Modelling Tool</p>
</def>
</def-item>
<def-item>
<term id="G33-fbinf.2025.1522401">
<bold>CMap</bold>
</term>
<def>
<p>Connectivity Map</p>
</def>
</def-item>
<def-item>
<term id="G34-fbinf.2025.1522401">
<bold>PDB</bold>
</term>
<def>
<p>Protein Data Bank</p>
</def>
</def-item>
<def-item>
<term id="G35-fbinf.2025.1522401">
<bold>RCSB</bold>
</term>
<def>
<p>Research Collaboratory for Structural Bioinformatics</p>
</def>
</def-item>
<def-item>
<term id="G36-fbinf.2025.1522401">
<bold>BLAST</bold>
</term>
<def>
<p>Basic Local Alignment Search Tool</p>
</def>
</def-item>
<def-item>
<term id="G37-fbinf.2025.1522401">
<bold>OPLS</bold>
</term>
<def>
<p>Optimized Potentials for Liquid Simulations</p>
</def>
</def-item>
</def-list>
</sec>
</back>
</article>