<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Microbiol.</journal-id>
<journal-title>Frontiers in Microbiology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Microbiol.</abbrev-journal-title>
<issn pub-type="epub">1664-302X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmicb.2023.1264941</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Microbiology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>microBiomeGSM: the identification of taxonomic biomarkers from metagenomic data using grouping, scoring and modeling (G-S-M) approach</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Bakir-Gungor</surname>
<given-names>Burcu</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/841729/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Temiz</surname>
<given-names>Mustafa</given-names>
</name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2062218/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Jabeer</surname>
<given-names>Amhar</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1347917/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wu</surname>
<given-names>Di</given-names>
</name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Yousef</surname>
<given-names>Malik</given-names>
</name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/62776/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Computer Engineering, Faculty of Engineering, Abdullah Gul University</institution>, <addr-line>Kayseri</addr-line>, <country>T&#x00FC;rkiye</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Electrical and Computer Engineering, Faculty of Engineering, Abdullah Gul University</institution>, <addr-line>Kayseri</addr-line>, <country>T&#x00FC;rkiye</country></aff>
<aff id="aff3"><sup>3</sup><institution>Department of Biostatistics, University of North Carolina at Chapel Hill</institution>, <addr-line>Chapel Hill, NC</addr-line>, <country>United States</country></aff>
<aff id="aff4"><sup>4</sup><institution>Division of Oral and Craniofacial Health Sciences, Adams School of Dentistry, University of North Carolina at Chapel Hill</institution>, <addr-line>Chapel Hill, NC</addr-line>, <country>United States</country></aff>
<aff id="aff5"><sup>5</sup><institution>Department of Information Systems, Zefat Academic College</institution>, <addr-line>Zefat</addr-line>, <country>Israel</country></aff>
<aff id="aff6"><sup>6</sup><institution>Galilee Digital Health Research Center (GDH), Zefat Academic College</institution>, <addr-line>Zefat</addr-line>, <country>Israel</country></aff>
<author-notes>
<fn id="fn0001" fn-type="edited-by"><p>Edited by: Domenica D&#x2019;Elia, National Research Council (CNR), Italy</p></fn>
<fn id="fn0002" fn-type="edited-by"><p>Reviewed by: Tatjana Loncar-Turukalo, University of Novi Sad Faculty of Technical Sciences, Serbia; Naida Babic Jordamovic, International Centre for Genetic Engineering and Biotechnology, Italy</p></fn>
<corresp id="c001">&#x002A;Correspondence: Mustafa Temiz, <email>mustafa.temiz@agu.edu.tr</email></corresp>
<corresp id="c002">Malik Yousef, <email>malik.yousef@gmail.com</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>22</day>
<month>11</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>14</volume>
<elocation-id>1264941</elocation-id>
<history>
<date date-type="received">
<day>21</day>
<month>07</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>08</day>
<month>11</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2023 Bakir-Gungor, Temiz, Jabeer, Wu and Yousef.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Bakir-Gungor, Temiz, Jabeer, Wu and Yousef</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Numerous biological environments have been characterized with the advent of metagenomic sequencing using next generation sequencing which lays out the relative abundance values of microbial taxa. Modeling the human microbiome using machine learning models has the potential to identify microbial biomarkers and aid in the diagnosis of a variety of diseases such as inflammatory bowel disease, diabetes, colorectal cancer, and many others. The goal of this study is to develop an effective classification model for the analysis of metagenomic datasets associated with different diseases. In this way, we aim to identify taxonomic biomarkers associated with these diseases and facilitate disease diagnosis. The microBiomeGSM tool presented in this work incorporates the pre-existing taxonomy information into a machine learning approach and challenges to solve the classification problem in metagenomics disease-associated datasets. Based on the G-S-M (Grouping-Scoring-Modeling) approach, species level information is used as features and classified by relating their taxonomic features at different levels, including genus, family, and order. Using four different disease associated metagenomics datasets, the performance of microBiomeGSM is comparatively evaluated with other feature selection methods such as Fast Correlation Based Filter (FCBF), Select K Best (SKB), Extreme Gradient Boosting (XGB), Conditional Mutual Information Maximization (CMIM), Maximum Likelihood and Minimum Redundancy (MRMR) and Information Gain (IG), also with other classifiers such as AdaBoost, Decision Tree, LogitBoost and Random Forest. microBiomeGSM achieved the highest results with an Area under the curve (AUC) value of 0.98% at the order taxonomic level for IBDMD dataset. Another significant output of microBiomeGSM is the list of taxonomic groups that are identified as important for the disease under study and the names of the species within these groups. The association between the detected species and the disease under investigation is confirmed by previous studies in the literature. The microBiomeGSM tool and other supplementary files are publicly available at: <ext-link xlink:href="https://github.com/malikyousef/microBiomeGSM" ext-link-type="uri">https://github.com/malikyousef/microBiomeGSM</ext-link>.</p>
</abstract>
<kwd-group>
<kwd>gut microbiome</kwd>
<kwd>metagenomics</kwd>
<kwd>type 2 diabetes</kwd>
<kwd>inflammatory bowel disease</kwd>
<kwd>colorectal cancer</kwd>
<kwd>machine learning</kwd>
<kwd>classification</kwd>
<kwd>feature selection</kwd>
</kwd-group>
<counts>
<fig-count count="3"/>
<table-count count="7"/>
<equation-count count="0"/>
<ref-count count="77"/>
<page-count count="18"/>
<word-count count="14015"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Evolutionary and Genomic Microbiology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<label>1</label>
<title>Introduction</title>
<p>A diverse community of trillions of microorganisms, including bacteria, archaea, viruses, as well as microbial eukaryotes like fungus, protozoa, and helminths, comprise the human microbiome. Human microbiome has an impact on overall human health and on homeostasis by influencing immunological function and by actively contributing to human metabolism (<xref ref-type="bibr" rid="ref37">Marcos-Zambrano et al., 2021</xref>). Several disease-related conditions have been connected to a rupture in the stable interaction between gut epithelial cells and the gut microbiota (<xref ref-type="bibr" rid="ref48">Petersen and Round, 2014</xref>). The number of microbiome-related studies has significantly risen in the last 10&#x2009;years, and large population studies such as the American Gut Project (<xref ref-type="bibr" rid="ref39">McDonald et al., 2018</xref>), the metagenomics of the Human Intestinal Tract (<xref ref-type="bibr" rid="ref51">Qin et al., 2010</xref>), and the Human Microbiome Project (<xref ref-type="bibr" rid="ref61">The Human Microbiome Project Consortium, 2012</xref>) have greatly expanded the amount of information currently accessible on the content and function of the human gut microbiome. The information from these studies is crucial for further research on host-microbiome linkages and how they relate to the commencement and evolution of many complicated diseases.</p>
<p>The community of microbes performs a variety of tasks for the host, including facilitating the uptake of nutrients (<xref ref-type="bibr" rid="ref38">Martin et al., 2019</xref>), preserving homeostasis (<xref ref-type="bibr" rid="ref44">Ohland and Jobin, 2015</xref>), fending off pathogens (<xref ref-type="bibr" rid="ref49">Pickard et al., 2017</xref>), regulating immunological response (<xref ref-type="bibr" rid="ref40">Mendes et al., 2019</xref>), among many others. Understanding these tasks and revealing the dialog between the bacterium and the host may help in developing plans for preserving the health status, treating diseases. In the last few decades, there has been an increased interest in researching microbial communities (and their associations) that live in various habitats, from the gut to the biosphere. Technological advancements lead to lower costs for 16S and metagenomic sequencing, greater sequencing resolution and depth (<xref ref-type="bibr" rid="ref30">Levy and Myers, 2016</xref>). Synchronous development of brand-new techniques for high throughput characterization of different -omic data types, such as lipidomics, metabolomics, metagenomics, metatranscriptomics and metaproteomics (<xref ref-type="bibr" rid="ref41">Muller, 2019</xref>) made this possible. However, it is a difficult task to experimentally detect the inter species microbe host associations due to several other difficulties relating to scale, scope, feasibility, and availability of samples for concurrent -omic readouts (<xref ref-type="bibr" rid="ref18">Fritz et al., 2013</xref>). Computational approaches can circumvent some of these constraints, improving our knowledge of microbial associations (<xref ref-type="bibr" rid="ref13">Dix et al., 2016</xref>).</p>
<p>The interactions between the host and the microbiome are critical factors affecting human health and disease. Therefore, recently there has been an exponential increase in microbiome studies. Many research efforts have been devoted to predicting disease based on taxonomic profiles derived from metagenomic sequencing data. In these studies, machine learning methods are used to predict the microbiome interactions associated with diseases. Beyond simply assessing their predictive capabilities using machine learning, these studies also highlight the importance of specific microbiomes as potential biomarkers for disease. In literature, there are numerous articles investigating microbiomes associated with three specific diseases: Colorectal Cancer (CRC), Type 2 Diabetes (T2D) and Inflammatory Bowel Disease (IBD). In particular, several studies aiming to uncover microbiomes related to T2D are summarized in <xref ref-type="bibr" rid="ref19">Gao et al. (2018)</xref>, <xref ref-type="bibr" rid="ref21">Gurung et al. (2020)</xref>, <xref ref-type="bibr" rid="ref8">Cena et al. (2023)</xref>, and <xref ref-type="bibr" rid="ref32">Li R. et al. (2023)</xref>. Microbiomes associated with CRC are reviewed in <xref ref-type="bibr" rid="ref24">Huybrechts et al. (2020)</xref>, <xref ref-type="bibr" rid="ref60">Tabowei et al. (2022)</xref>, <xref ref-type="bibr" rid="ref42">Negrut et al. (2023)</xref>, and <xref ref-type="bibr" rid="ref77">Zwezerijnen-Jiwa et al. (2023)</xref>. The studies of <xref ref-type="bibr" rid="ref59">Soueidan and Nikolski (2016)</xref>, <xref ref-type="bibr" rid="ref29">LaPierre et al. (2019)</xref>, <xref ref-type="bibr" rid="ref37">Marcos-Zambrano et al. (2021)</xref>, <xref ref-type="bibr" rid="ref33">Lim et al. (2022)</xref>, <xref ref-type="bibr" rid="ref23">Hsu et al. (2023)</xref>, and <xref ref-type="bibr" rid="ref35">Mah et al. (2023)</xref> reviews the microbiomes associated with IBD.</p>
<p>More specifically, <xref ref-type="bibr" rid="ref10">Desch&#x00EA;nes et al. (2023)</xref> employed machine learning techniques to predict diseases by representing microbiomes using gene-based representations and taxonomic profiles. Through the creation of taxonomic profiles from shotgun metagenomic data, they identified significant taxa using their proposed methodology. They conducted experiments for five different diseases, namely type 2 diabetes, obesity, liver cirrhosis, colorectal cancer, and inflammatory bowel disease. For both IBD and CRC disease, the datasets used in <xref ref-type="bibr" rid="ref10">Desch&#x00EA;nes et al. (2023)</xref> are the same datasets used by the proposed approach in this study. In their study, they assessed the performance of nine distinct classifiers, including random forest, decision tree, two support vector machines with a linear kernel, random set coverage machine (rSCM), two logistic regressions, SVM with a radial basis function kernel (SVMrbf), and an ensemble algorithm derived from SCM (set coverage machine). For each dataset, they applied embedded feature selection techniques, such as random forest and ranking features based on resulting models, followed by machine learning model application. They reported improved classification performance for certain diseases by employing taxonomic profiling. The most effective results in taxonomic profiling were achieved using the random forest algorithm for liver cirrhosis, yielding an AUC of 88%. Their study demonstrated the effective use of converting microbiome data into taxonomic representation data for disease prediction. They reported that Lachnospiraceae microbiome is found as associated with T2D and it can be considered as a biomarker for this disease.</p>
<p><xref ref-type="bibr" rid="ref57">Sharma et al. (2020)</xref> predicted disease states using machine learning methods by examining related Operational Taxonomic Units (OTUs) at the same phylum taxonomic level, exploiting the connections among OTUs at this taxonomic rank. Their investigation focused on the relationship between disease and the microbiome, utilizing shotgun datasets for two distinct diseases, T2D and Cirrhosis. The dataset they chose for T2D analysis is the same as the dataset used by our proposed tool. They applied their proposed method, which they called &#x201C;TaxoNN,&#x201D; to a dataset with 174 cases and 170 controls for T2D (<xref ref-type="bibr" rid="ref50">Qin et al., 2012</xref>) and a dataset with 118 cases and 114 controls for cirrhosis (<xref ref-type="bibr" rid="ref52">Qin et al., 2014</xref>). TaxoNN is a Deep Learning based multi-layered approach to group OTU information based on phylum clusters. It trains clusters containing OTUs that share the same phylum separately using Convolutional Neural Networks (CNNs). It combines features from each cluster to enhance prediction accuracy via an ensemble learning technique. Their proposed method was evaluated using six different classifiers, including Random Forest, Gaussian Bayes Classifier, Naive Bayes, Ridge Regression, Lasso Regression, and Support Vector Machines. The TaxoNN method yielded the highest result, achieving an AUC of 92% for cirrhosis and 75% for T2D. Moreover, TaxoNN identified microbiomes at the level of three dominant phyla (Firmicutes, Proteobacteria, and Actinobacteria) for both diseases, highlighting their impact on the diseases.</p>
<p><xref ref-type="bibr" rid="ref20">Giliberti et al. (2022)</xref> investigated the influence of the relative abundance of microbial taxa on host phenotype classification using human metagenomes. They employed machine learning methods to construct species-level taxonomic profiles and accurately detected the presence of microbial taxa. In their evaluation scheme, they encompassed a total of 4,128 samples from 25 shotgun metagenomic datasets. Among the datasets used in their study, T2D dataset is same with the dataset used in this study. They also explored the effect on disease prediction using relative abundance values at three different taxonomic levels: genus, family, and order. Employing the Random Forest classification algorithm on species level dataset, they achieved the best performance for IBD dataset, across other datasets containing seven distinct disease categories (atherosclerotic cardiovascular disease, Alzheimer&#x2019;s disease, Beh&#x00E7;et&#x2019;s disease, colorectal cancer, irritable bowel disease, type 1 diabetes, and type 2 diabetes). They identified statistically significant microbiomes for the diseases they identified. Among these microbiomes for these cases, the most significant result was obtained for Clostridium and this microbiome was followed by <italic>Streptococcus</italic> and <italic>Ruthenibacterium</italic>.</p>
<p><xref ref-type="bibr" rid="ref46">Pasolli et al. (2016)</xref> investigated the utility of microbiomes in disease prediction using metagenomic datasets for five different diseases: liver cirrhosis, CRC, IBD, obesity, and T2D. Among the datasets used in this study, T2D dataset is also utilized within this study. They conducted species-level prediction using microbiome profiles at the species level derived from metagenomic data. Their analysis encompassed a total of 2,424 shotgun metagenomic data samples from eight distinct studies. Employing cross-validation techniques, they compared classification outcomes using two widely employed classifiers in metagenomic data analysis, Random Forest and Support Vector Machine. In addition to these classifiers, they also evaluated the effectiveness of elastic network, neural network, and multiple regression methods. In addition to predicting diseases using microbiome data, they highlighted prominent microbiomes related to these diseases. Notably, they identified the Peptostreptococcus microbiome for colorectal cancer, the Streptococcus microbiome for T2D, and the Lachnospiraceae microbiome for IBD as influential microbiomes in disease prediction. Collectively, these papers advance our understanding for the potential role of the microbiome in these diseases using a variety of approaches and analyzes.</p>
<p>Identifying microbial taxa that may cause disease development and identifying microbial taxa whose impact varies depending on their abundance is one of the major goals of human microbiome studies. Uncovering the influence of taxons can help to the investigation of disease development processes and hence can contribute to the emergence of new approaches for prevention of these diseases (<xref ref-type="bibr" rid="ref75">Zhang W. et al., 2022</xref>). Computational methods dealing with microbial relative abundances face several challenges in drawing meaningful conclusions due to their complex data structures and properties. Traditional computational methods are inadequate to assess microbiome population effects in isolation and to produce effective results without considering the diversity of the human microbiome. Recent research has used machine learning (ML) approaches to evaluate data from the human microbiome, more specifically to identify and understand the diversity of taxonomy and function within microbial communities, and to assess the impact of these factors on human health (<xref ref-type="bibr" rid="ref64">Top&#x00E7;uo&#x011F;lu et al., 2020</xref>). The use of ML in microbiome studies can be summarized as follows:</p>
<list list-type="bullet">
<list-item><p>ML models have been created to promote taxonomic representation and differentiation in microbiology.</p></list-item>
<list-item><p>ML has been used for disease prediction by inferring host phenotypes.</p></list-item>
<list-item><p>ML facilitates the characterization of disease-specific microbial signatures to classify patients based on microbial communities (<xref ref-type="bibr" rid="ref37">Marcos-Zambrano et al., 2021</xref>).</p></list-item>
</list>
<p>In this paper, we present a novel approach, microBiomeGSM, to detect disease-associated taxonomic biomarkers by developing an efficient machine learning model based on the Grouping, Scoring and Modeling (G-S-M) approach. We have analyzed taxonomically transformed microbiome sequencing datasets with our proposed machine learning method. In this way, we aim to reveal the impact of the identified taxonomic biomarkers on specific diseases. To this end, our study contributes to the diagnosis and treatment of the disease under investigation. The proposed approach is applied on metagenomic datasets associated with 4 different datasets; and the taxonomic groups that have an impact on disease under study are identified. In the data preprocessing step, the MetaPhlAn tool developed by <xref ref-type="bibr" rid="ref12">Ditzler et al. (2015)</xref> is used to extract taxonomic data from microbiome sequencing data. In the first component (grouping component) of microBiomeGSM, the species identified in a sample are grouped according to the level of taxa known to be associated with them. In the second component (scoring component) of microBiomeGSM, importance scores are assigned to taxon groups using inherent machine learning techniques. The score is a predictor of how well a sample can be classified based on the abundance values of the species included in that taxon group. In the final (modeling) component of microBiomeGSM, three different outputs are generated. The first output is the performance metrics of the developed machine learning model. The second output is the list of important taxa groups associated with the disease under study, and these taxonomic features can be considered as biomarkers. The third output is the species associated with the taxa groups. Performance evaluation of microBiomeGSM is assessed separately for each disease, and for 3 different taxonomic levels (genus, family, order). Feature selection algorithms are applied to the same dataset in order to comparatively evaluate the performance of microBiomeGSM. The biological relevance of the identified taxon groups at genus, family, order levels for different diseases is discussed with reference to existing knowledge in the literature.</p>
</sec>
<sec sec-type="materials|methods" id="sec2">
<label>2</label>
<title>Materials and methods</title>
<sec id="sec3">
<label>2.1</label>
<title>Dataset</title>
<p>The data used in this study are obtained from the NCBI Sequence Read Archive (SRA045646, SRA050230) provided by <xref ref-type="bibr" rid="ref50">Qin et al. (2012)</xref> for T2D; accession number PRJNA398089 in the SRA for the Integrative Human Microbiome Project for IBDMDB (<xref ref-type="bibr" rid="ref6">Beghini et al., 2021</xref>). IBD dataset is obtained from the MetaHit project (<xref ref-type="bibr" rid="ref36">Marco-Ramell et al., 2018</xref>) (ERA000116). The CRC metagenomic dataset containing 1,262 samples was created by <xref ref-type="bibr" rid="ref6">Beghini et al. (2021)</xref>. Microbiome sequencing data is classified into disease states based on the metadata associated with them. To ensure data quality, we applied quality filtering to meet the standards outlined in the Human Microbiome Project Consortium SOP (2012), as referenced in <xref ref-type="bibr" rid="ref62">Thomas et al. (2019)</xref>. This procedure allowed us to categorize the raw sequencing data according to relevant disease states, enabling our subsequent analyzes. The microbiome samples were associated with the microbial species of origin (taxa) using the MetaPhlAn tool, and the relative abundance composition for each taxon was generated accordingly. These taxa and their relative abundances serve as features or variables in our machine learning approaches. MetaPhlAn first assigns reads to microbial clusters using clade-specific genes for assignment. It then presents the relative abundance of microbial taxa based on these readings. In this study, the assignment to microbial species of origin (taxa) was determined for each DNA sequence using the MetaPhlAn tool. The relative abundance value is normalized by dividing the number of reads for each taxonomic level by the total number of reads for only one sample. In this way, the taxonomic abundance values are expressed as real numbers in the range [0,1] with a sum of 1 for each sample. Samples with less than 1&#x2009;million total reads were not included in our study. For each sample, we determined the diversity of disease-relevant microbiomes, where diversity represents the presence and relative abundance of microorganisms (<xref ref-type="bibr" rid="ref2">Alatawi et al., 2022</xref>).</p>
<p>The four microbiome datasets used to evaluate the microBiomeGSM tool are listed in <xref ref-type="table" rid="tab1">Table 1</xref>. The table presents the number of samples in each dataset and the number of samples that are labeled as positive. Positive samples refer to patients, while negative samples refer to controls. Each dataset contains the abundance values of the species, which we consider as features. We have considered 3 taxonomic levels for creating the groups, i.e., genus, family, and order. For each dataset, the number of extracted groups is listed in the corresponding column, while &#x2018;-&#x2019; denotes missing information.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption><p>The list of datasets used to test the model.</p></caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="middle">#</th>
<th align="left" valign="middle">Dataset</th>
<th align="center" valign="middle"># of Samples</th>
<th align="center" valign="middle"># of positives</th>
<th align="center" valign="middle"># of features (Species)</th>
<th align="center" valign="middle"># of Groups (Genus)</th>
<th align="center" valign="middle"># of Groups (Family)</th>
<th align="center" valign="middle"># of Groups (Order)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">1</td>
<td align="left" valign="top">CRC</td>
<td align="center" valign="top">1,262</td>
<td align="center" valign="top">600</td>
<td align="center" valign="top">912</td>
<td align="center" valign="top">261</td>
<td align="center" valign="top">100</td>
<td align="center" valign="top">49</td>
</tr>
<tr>
<td align="left" valign="top">2</td>
<td align="left" valign="top">IBDMDB</td>
<td align="center" valign="top">1,638</td>
<td align="center" valign="top">1,209</td>
<td align="center" valign="top">579</td>
<td align="center" valign="top">187</td>
<td align="center" valign="top">77</td>
<td align="center" valign="top">43</td>
</tr>
<tr>
<td align="left" valign="top">3</td>
<td align="left" valign="top">IBD</td>
<td align="center" valign="top">382</td>
<td align="center" valign="top">148</td>
<td align="center" valign="top">1,456</td>
<td align="center" valign="top">448</td>
<td align="center" valign="top">177</td>
<td align="center" valign="top">84</td>
</tr>
<tr>
<td align="left" valign="top">4</td>
<td align="left" valign="top">T2D</td>
<td align="center" valign="top">290</td>
<td align="center" valign="top">155</td>
<td align="center" valign="top">1,456</td>
<td align="center" valign="top">448</td>
<td align="center" valign="top">177</td>
<td align="center" valign="top">84</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Number of samples who have positive class label are shown in the second column. The number of features/Species is shown in the third column. Number of groups created at Order, Family and Genus taxonomic levels are listed at the 4-6th columns, respectively.</p>
</table-wrap-foot>
</table-wrap>
<p>Statistical information regarding the numbers of features in each group is given in <xref ref-type="table" rid="tab2">Table 2</xref>. For each data set and for each taxonomic level (genus, family, and order), the average, maximum, and minimum numbers of features within a group are given.</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption><p>Statistical information about the numbers of features within a group, shown separately for each taxonomic level.</p></caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="middle">#</th>
<th align="left" valign="middle">Dataset</th>
<th align="center" valign="middle">Genus (avg/max/min)</th>
<th align="center" valign="middle">Family (avg/max/min)</th>
<th align="center" valign="middle">Order (avg/max/min)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">1</td>
<td align="left" valign="top">CRC</td>
<td align="char" valign="top" char=".">3.51 /52/1</td>
<td align="char" valign="top" char=".">9.16 /76/1</td>
<td align="char" valign="top" char=".">18.71/202/ 1</td>
</tr>
<tr>
<td align="left" valign="top">2</td>
<td align="left" valign="top">IBDMDB</td>
<td align="char" valign="top" char=".">3.09 /34/1</td>
<td align="char" valign="top" char=".">7.50 /64/1</td>
<td align="char" valign="top" char=".">13.44/163/ 1</td>
</tr>
<tr>
<td align="left" valign="top">3</td>
<td align="left" valign="top">IBD</td>
<td align="char" valign="top" char=".">3.24 /61/1</td>
<td align="char" valign="top" char=".">8.22 /65/1</td>
<td align="char" valign="top" char=".">17.32/195/1</td>
</tr>
<tr>
<td align="left" valign="top">4</td>
<td align="left" valign="top">Type 2 diabetes</td>
<td align="char" valign="top" char=".">3.24 /61/1</td>
<td align="char" valign="top" char=".">8.22 /65/1</td>
<td align="char" valign="top" char=".">17.32/195/ 1</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Sen is the sensitivity, Spe is the specificity, AUC is the Area Under the Curve.</p>
</table-wrap-foot>
</table-wrap>
<p><xref rid="SM1" ref-type="supplementary-material">Supplementary Table S1</xref> shows the distribution of the groups based on their sizes for the IBDMDB dataset. The numbers in the table indicate the number of groups that have the specified number of species for that specific taxonomic level. There are 187, 77, and 43 groups for genus, family and order levels, respectively. About 90% of the groups at the order level, about 90% of the groups at the family level, and about 97% of the groups at the genus level contain 20 or fewer species for the IBDMDB dataset.</p>
</sec>
<sec id="sec4">
<label>2.2</label>
<title>microBiomeGSM</title>
<p>Our proposed method, microBiomeGSM, consists of three main components: Grouping, Scoring, and Modeling (G-S-M). The G-S-M approach has been used in other studies that consider the pre-existing biological knowledge (<xref ref-type="bibr" rid="ref67">Yousef et al., 2019</xref>, <xref ref-type="bibr" rid="ref69">2021a</xref>,<xref ref-type="bibr" rid="ref72">c</xref>, <xref ref-type="bibr" rid="ref71">2022a</xref>; <xref ref-type="bibr" rid="ref53">Qumsiyeh et al., 2022</xref>; <xref ref-type="bibr" rid="ref73">Yousef and Voskergian, 2022</xref>; <xref ref-type="bibr" rid="ref15">Ersoz et al., 2023</xref>; <xref ref-type="bibr" rid="ref26">Jabeer et al., 2023</xref>). Additionally it was modified to integrate two-omics datasets such as the miRcorrNet and miRModuleNet tools (<xref ref-type="bibr" rid="ref69">Yousef et al., 2021a</xref>, <xref ref-type="bibr" rid="ref68">2022b</xref>); and even to integrate 3 omics datasets such as 3Mint tool (<xref ref-type="bibr" rid="ref65">Unlu Yazici et al., 2023</xref>). Interested readers can find further details about those approaches in our recent reviews (<xref ref-type="bibr" rid="ref70">Yousef et al., 2021b</xref>; <xref ref-type="bibr" rid="ref28">Kuzudisli et al., 2023</xref>).</p>
<p>Utilizing the G-S-M approach, microBiomeGSM performs a search to identify the most important taxonomic groups in disease-associated metagenomic datasets. The relative abundance values of the species within the group can be checked for each sample; and the generated model decides whether the sample has the disease or not. By focusing on a specific taxonomic level, we can use the G component to find the most significant group for the disease under study. This approach provides the advantage of focusing on either the macroscopic or microscopic view of the most important group to distinguish between healthy samples and patient samples. An overview of the steps performed in microBiomeGSM is presented in <xref ref-type="fig" rid="fig1">Figure 1</xref>.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption><p>G-S-M approach in microBiomeGSM. MCCV denotes Monte Carlo Cross-Validation.</p></caption>
<graphic xlink:href="fmicb-14-1264941-g001.tif"/>
</fig>
<p>Let X be the two-class dataset consisting of the species in the columns, and samples in the rows including the class labels (1 denoting the disease state and 0 denoting the healthy state). To understand the approach in detail, let us assume that the taxonomic level is selected as &#x201C;genus&#x201D; for the &#x201C;Select taxa rank&#x201D; step in <xref ref-type="fig" rid="fig1">Figure 1</xref>. The input X<sub>abd</sub> (abundance matrix) is first split into a training set (X<sub>train</sub>) and a test set (X<sub>test</sub>) with a ratio of 80:20 based on the class labels. Denote by S the feature space of all species in X<sub>abd</sub> and by U<sub>genus</sub> all unique genera for S. Grp{} denotes the selection function of each U<sub>genus</sub> in S, grouping all species on the basis of similar genuses. Grp{U<sub>genus</sub><sup>i</sup> for S} represents each genus in S, with all the species grouped by genus. For example, if we take Alistipes as one of the genus in U<sub>genus</sub>, we get the following when we apply the Grp function.</p>
<p>Grp{U<sub>genus</sub><sup>i</sup>}, where i&#x2009;=&#x2009;Alistipes and &#x2208; S.</p>
<p>Grp{Alistipes}&#x2009;=&#x2009;{alistipes_finegoldi, alistipes_indistinctus, alistipes_inops, alistipes_shahii}.</p>
<p>Similarly, this approach is applied to all genuses that are present in X<sub>abd</sub>, and a list of genus groups is created, as shown in <xref ref-type="fig" rid="fig1">Figure 1</xref> after the select taxa rank step. This is repeated for the three taxonomic levels identified.</p>
<p>When <xref ref-type="fig" rid="fig1">Figure 1</xref> is examined, firstly, in the grouping component G, for all the groups of genus, we partition X<sub>train</sub> into sub data denoted as sub_d<sub>x</sub>. Following the earlier example of Alistipes, this group yields sub_d<sub>alistipes</sub> which is created from X<sub>train</sub>. The sub_d<sub>alistipes</sub> contains the labels of the samples, but the feature space is restricted only to species within the Alistipes genus. This is applied to all different genera created in the prior step, so we have multiple subsets of data with a feature space specified by genus. Secondly, in the scoring step S, the generated sub_d is trained on a Random Forest classifier with 5-fold cross-validation with randomized stratified shuffling. Each sub_d is given a score equal to the mean of the accuracy over all foldings based on the prediction of the labels. Each sub_d is scored and then sorted based on the score. The top k groups with the highest score are used for the subsequent step. The value chosen for k is 10, but other values for k have been tested. Following the example of selecting genus as the taxonomic level, the top 10 genus groups that show strong discriminative ability are used to build the classification model. Thirdly, in the modeling component, the species from the top 10 genus groups are used to train a Random Forest model with 100-fold Monte Carlo Cross-Validation (MCCV). The top ranking set of species corresponding to the top ranked group is trained on X<sub>train</sub> and then tested on X<sub>test</sub>. Then, the second set of species corresponding to the second highest scoring group is aggregated with the top scoring set of species; and then used to train and test the model. This process is repeated until all species in the top 10 ranked genus groups are aggregated; and used to train and test the classifier. This whole process is repeated 100 times, stratifying the initial X<sub>abd</sub> and randomly splitting it into X<sub>train</sub> and X<sub>test</sub> without replacement. The classification performance metrics are determined as the average of the metrics obtained in 100 folds. Similarly, the top ranked groups and the top ranked species are retained for each run.</p>
</sec>
<sec id="sec5">
<label>2.3</label>
<title>Implementation of microBiomeGSM</title>
<p>The microBiomeGSM tool utilizes the pre-existing biological knowledge of the assignment of the species into different taxonomic levels, such as genus, family, and order. Experiments with the microBiomeGSM tool were conducted on the open-source KNIME platform (<xref ref-type="bibr" rid="ref7">Berthold et al., 2009</xref>). This platform can handle a wide range of data types and operations. The user can configure the number of iterations, the rank function, and the number of iterations for MCCV. All rows with missing values are removed within the workflow.</p>
</sec>
<sec id="sec6">
<label>2.4</label>
<title>Application of feature selection and classifiers using metagenomic data</title>
<p>In metagenomics research, it is observed that in studies using taxonomic features, the number of observations used for training data is higher than the number of observations used for testing data. This situation is undesirable if studies are to produce more effective results, and researchers are proposing various methods of resolution, particularly feature selection methods. Although the process of feature selection in disease prediction problems based on metagenome data has not been well studied, the literature suggests that this process may be as important as the choice of a classification method (<xref ref-type="bibr" rid="ref29">LaPierre et al., 2019</xref>). The process of feature selection in metagenome-based disease prediction could help us learn more about disease development mechanisms. Therefore, further research in this direction is warranted. In metagenomics studies, in order to reduce the number of taxa, i.e., to select informative species (features), min Redundancy Max Relevance (mRMR) (<xref ref-type="bibr" rid="ref11">Ding and Peng, 2005</xref>), Lasso (<xref ref-type="bibr" rid="ref63">Tibshirani, 1996</xref>), Elastic Net (<xref ref-type="bibr" rid="ref76">Zou and Hastie, 2005</xref>), and the iterative sure select algorithm (<xref ref-type="bibr" rid="ref14">Duvallet et al., 2017</xref>) have been used extensively. Another feature selection method, called Fizzy, addresses the challenge of using classification techniques to identify important functional elements for downstream analysis (<xref ref-type="bibr" rid="ref12">Ditzler et al., 2015</xref>). Oudah and Henschel presented an alternative taxonomy-based method for feature selection (<xref ref-type="bibr" rid="ref45">Oudah and Henschel, 2018</xref>). <xref ref-type="bibr" rid="ref4">Bakir-Gungor et al. (2021)</xref> applied CMIM (<xref ref-type="bibr" rid="ref16">Fleuret and Ch, 2004</xref>), FCBF (<xref ref-type="bibr" rid="ref56">Senliol et al., 2008</xref>), mRMR (<xref ref-type="bibr" rid="ref11">Ding and Peng, 2005</xref>), and Select K best (SKB) (<xref ref-type="bibr" rid="ref47">Pedregosa et al., 2011</xref>) to type 2 diabetes-associated metagenomics datasets and obtained powerful performance metrics (<xref ref-type="bibr" rid="ref4">Bakir-Gungor et al., 2021</xref>). Jabeer et al. also proposed a robust classification method for evaluating colorectal cancer associated metagenomic datasets using a combination of feature selection methods and machine learning methods (<xref ref-type="bibr" rid="ref25">Jabeer et al., 2022</xref>). <xref ref-type="bibr" rid="ref5">Bakir-Gungor et al. (2022)</xref> also proposed a powerful method for IBD classification with fewer features by combining feature selection methods and machine learning methods (<xref ref-type="bibr" rid="ref5">Bakir-Gungor et al., 2022</xref>). While these feature selection approaches have produced effective results in a variety of fields, they have only recently been applied to microbiome-based disease prediction problems.</p>
<p>In this study, we have comparatively evaluated microBiomeGSM with different classifiers and with different feature selection methods. As the feature selection methods, we have utilized Select K best (SKB), Fast Correlation Based Filter (FCBF), Extreme Gradient Boosting (XGBoost), Min Redundancy Max Relevance (mRMR), Information Gain (IG), and Conditional Mutual Information Maximization (CMIM). <xref ref-type="bibr" rid="ref66">Wang and Liu (2020)</xref> compare the performance of classifiers with traditional methods and ensemble methods for disease prediction based on human microbiome data. They use Elastic Network and SVM as traditional methods and Random Forest and Extreme Gradient Boosting (XGBoost) as ensemble methods. In their study, they find that the XGBoost algorithm shows superior performance compared to other algorithms (<xref ref-type="bibr" rid="ref66">Wang and Liu, 2020</xref>). In another study, <xref ref-type="bibr" rid="ref37">Marcos-Zambrano et al. (2021)</xref> conducted an important review paper to reveal the links between the microbiome and diseases. In this study, which included information on the performance of machine learning methods, they found that the Support Vector Machines (SVM), Random Forest (RF), k-Nearest Neighbors (k-NN), and Logical Regression (LR) algorithms were widely used. They concluded that when selecting a machine learning algorithm, several factors should be considered such as the set of observations, the set of features, the type of data, and the quality of the data. They suggest using several different methods, comparing them, and choosing the one that provides the best performance value (<xref ref-type="bibr" rid="ref37">Marcos-Zambrano et al., 2021</xref>).</p>
</sec>
<sec id="sec7">
<label>2.5</label>
<title>microBiomeGSM model performance evaluation</title>
<p>Accuracy, F1 score, sensitivity, specificity, and AUC were used to evaluate the predictive performance of the proposed models. AUC score is a common measure for performance evaluation and a reliable metric for evaluating balanced datasets. Other metrics such as F1 score, sensitivity, specificity, and accuracy, were used to evaluate the performance of the created models because the dataset for this study has an uneven distribution of classes. When a balance between precision and recall is desired and there is an uneven distribution of classes, the F1 score is a good option among the performance metrics (many true negatives). Several classifiers report the probability values for their predictions, which can also be considered as confidence values for the prediction. The AUC often uses this information to figure out how often incorrect predictions occur at different confidence levels. In real life, test results from positive and negative examples overlap. AUC illustrates how the threshold or cut-off value for identifying positive examples affects the relationship between recall and precision. In this study, all of the above-mentioned metrics were calculated as the mean of 100 times MCCV. After each iteration, we obtain lists of significant taxonomic groups and species associated with these taxa groups for a given disease. To assign scores to the entities in the taxonomic groups list and in the species lists, a prioritization approach is used. For this purpose, we integrated the RobustRankAggreg algorithm (<xref ref-type="bibr" rid="ref27">Kolde et al., 2012</xref>) and microBiomeGSM. RobustRankAggreg algorithm is available as an R package. Each entity (taxonomic group or species) in the lists is given a value of p by the RobustRankAggreg technique, indicating how highly ranked that entity. Using the RobustRankAggreg tool, microBiomeGSM outputs a list of species to which it has assigned a significance value (value of <italic>p</italic>) for a specific taxonomic group. Each taxa group is assigned a significance value and the species associated with that group are assigned the same value.</p>
</sec>
</sec>
<sec sec-type="results" id="sec8">
<label>3</label>
<title>Results</title>
<p>The main objective of this study is to identify the microbial communities that are associated with specific diseases. In order to facilitate disease diagnosis, using metagenomic data we develop an efficient classification model based on taxonomic levels. In this section we present our findings for four different datasets. Here we also present comparative evaluation results against other existing methods.</p>
<sec id="sec9">
<label>3.1</label>
<title>Comparing varying group size for microBiomeGSM</title>
<p>One approach to evaluate model performance in the context of microBiomeGSM is to compare model performance between different values of the parameter k. k represents the number of groups (taxa) used in microBiomeGSM models. This approach can help researchers determine the optimal value of k that balances model complexity and predictive power, ultimately leading to more effective and interpretable models in microbiome-related research. It provides insight into how the inclusion or exclusion of specific taxa affects the overall performance of microBiomeGSM models.</p>
<p><xref rid="SM1" ref-type="supplementary-material">Supplementary Table S2</xref> shows the performance metrics obtained with 100-fold MCCV for the aggregated top 10 groups for four different datasets compared at three different taxonomic levels (genus, family, order) for grouping. For the IBDMDB dataset, microBiomeGSM achieved an AUC of 93% using the top 1 group at the family level. Performance metrics are shown for the top 2 groups via combining species from the first and second highest scoring groups. We obtained an AUC of 97% when the top 2 groups are combined at the family taxonomic level for the IBDMDB dataset. In this way, microBiomeGSM provides cumulative performance results for the top 10 highest scoring groups. For the IBDMDB dataset, the highest performance metric (an AUC of 98%) is obtained using the species from the top 10 groups at the order taxonomic level. For the IBD dataset, the highest performance metric (an AUC of 93%) is obtained using the species from the top 9 groups at the order taxonomic level. For the T2D dataset, the highest performance metric (an AUC of %74) is obtained using the species from the top 9 groups at the order taxonomic level. For the CRC dataset, the highest performance metric (an AUC of %83) is obtained using the species from the top 10 groups at the family taxonomic level. While examining other performance metrics (such as accuracy, sensitivity, specificity in <xref rid="SM1" ref-type="supplementary-material">Supplementary Table S2</xref>), it is noteworthy that satisfactory results are obtained with microBiomeGSM for each taxonomic level, especially for the IBDMDB dataset. The high sensitivity values that are reported for the CRC, IBDMDB, and IBD datasets display the success of the microBiomeGSM tool in terms of detecting the patient samples. In the CRC, IBDMDB, and IBD datasets, the strikingly high specificity values indicate that the microBiomeGSM tool correctly identifies the negative samples (i.e., individuals who do not have the disease). However, in the T2D dataset, the specificity rate appears to be relatively low compared to the other datasets. Nevertheless, the ability to detect negative samples remains at a reasonable level.</p>
<p>In addition, <xref ref-type="fig" rid="fig2">Figures 2</xref>, <xref ref-type="fig" rid="fig3">3</xref> show the sensitivity and specificity values obtained with the microBiomeGSM tool for all datasets. <xref ref-type="fig" rid="fig2">Figure 2</xref> shows the sensitivity values obtained using the microBiomeGSM tool across all datasets. One can notice from <xref ref-type="fig" rid="fig2">Figure 2A</xref> that for the CRC data set the highest sensitivity value (73%) is obtained for the order taxon level using 10 cumulative groups. In particular, the sensitivity values calculated for the IBDMDB dataset were quite impressive, especially in group 1 and group 6, both at the family taxon level, reaching 99% sensitivity value, as shown in <xref ref-type="fig" rid="fig2">Figure 2B</xref>. <xref ref-type="fig" rid="fig2">Figure 2C</xref> shows another impressive set of results for the IBD data set. In <xref ref-type="fig" rid="fig2">Figure 2C</xref>, we observe high values for sensitivity, in particular 87% sensitivity at the taxon level in group 1. As shown in <xref ref-type="fig" rid="fig2">Figure 2D</xref>, the highest sensitivity value for the T2D data set is 69%. This result is obtained for the genus taxon level using 10 cumulative groups. A sensitivity value of 69% is also obtained for the family taxon level using 4 cumulative groups.</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption><p>Sensitivity values obtained at the family, order, and genus taxon levels for the top 10 significant groups across all 4 datasets. <bold>(A&#x2013;D)</bold> Represents the results obtained in CRC, IBDMDB, IBD, T2D datasets, respectively.</p></caption>
<graphic xlink:href="fmicb-14-1264941-g002.tif"/>
</fig>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption><p>Specificity values at the order, genus, and family taxon level for the top 10 significant groups for all 4 disease datasets. <bold>(A&#x2013;D)</bold> Represents the results obtained in CRC, IBDMDB, IBD, T2D datasets, respectively.</p></caption>
<graphic xlink:href="fmicb-14-1264941-g003.tif"/>
</fig>
<p><xref ref-type="fig" rid="fig3">Figure 3</xref> shows the specificity values obtained using the microBiomeGSM tool for all datasets. As shown in <xref ref-type="fig" rid="fig3">Figure 3A</xref>, the specificity value obtained for the CRC dataset is remarkable, reaching an impressive specificity value of 94% at the family taxon level for 1 group. <xref ref-type="fig" rid="fig3">Figure 3B</xref> depicts that the highest specificity value obtained for the IBDMDB dataset is 93% for 1 group at the order taxon level. As displayed in <xref ref-type="fig" rid="fig3">Figure 3C</xref>, the highest specificity value obtained for the IBD dataset is 85% for the 4 cumulative groups at the order taxon level. The same result is also obtained at the order taxon level for the 5 cumulative groups. One can notice in <xref ref-type="fig" rid="fig3">Figure 3D</xref> that the highest specificity value that is obtained for the T2D dataset is 71% for the 6 cumulative groups at the order taxon level.</p>
<p>The number of significant groups used to train the model could affect the performance of microBiomeGSM. <xref ref-type="table" rid="tab3">Table 3</xref> shows the influence of the number of groups and the number of species at family, genus and order levels on four datasets. <xref ref-type="table" rid="tab3">Table 3</xref> presents the performance of the top 10 cumulative groups and top 1 group for each taxonomic level on different tested datasets. For the IBDMDB dataset, for the family taxonomic level, one can observe that the AUC increases by 5% when we consider the top 10 significant groups cumulatively, while we increase the number of species from 34 to 205. On the same dataset, an increase of 8% in AUC score is observed at the Genus taxonomic level via increasing the number of species from 34 to 119. For the same dataset, a decrease of 1% is observed at the Order taxonomic level. Order taxonomic level using the top group that includes 98 species achieves the highest AUC success rate of 98% for the IBDMDB dataset. Similarly, family taxonomic level using the top 10 combined groups achieves 97% AUC on the IBDMDB dataset, but these 10 combined groups include a much higher number of species (205 species). For the IBD dataset, the highest AUC value of 91% was obtained using the microBiomeGSM tool. This value at the family taxonomic level was obtained by cumulatively combining 10 groups, using an average of 260.4 species. For the T2D dataset, the highest AUC value of 72% was obtained using the microBiomeGSM tool. This value, obtained at the order taxanomic level, was obtained by combining 10 groups cumulatively. For 1 group, an average of 138.28 species are used at the taxonomic level, while for 10 groups, an average of 596.99 species are used. For the CRC dataset, the highest AUC value of 87% was obtained using the microBiomeGSM tool. This value at the order taxanomic level was obtained by cumulatively combining 10 groups, using an average of 604 species.</p>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption><p>The effect of the number of groups that are generated at different taxonomic levels on performance metrics for all dataset.</p></caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="center" valign="top" colspan="9">CRC</th>
</tr>
<tr>
<th align="left" valign="middle">Taxonomic hierarchy</th>
<th align="center" valign="middle"># of groups</th>
<th align="center" valign="middle">Average # of species</th>
<th align="center" valign="middle">Accuracy</th>
<th align="center" valign="middle">Sen</th>
<th align="center" valign="middle">Spe</th>
<th align="center" valign="middle">F measure</th>
<th align="center" valign="middle">AUC</th>
<th align="center" valign="middle">Precision</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="bottom">Family</td>
<td align="char" valign="bottom" char=".">10</td>
<td align="char" valign="bottom" char=".">239.11</td>
<td align="char" valign="bottom" char=".">0.78</td>
<td align="char" valign="bottom" char=".">0.84</td>
<td align="char" valign="bottom" char=".">0.72</td>
<td align="char" valign="bottom" char=".">0.79</td>
<td align="char" valign="bottom" char=".">0.84</td>
<td align="char" valign="bottom" char=".">0.76</td>
</tr>
<tr>
<td align="left" valign="bottom">Family</td>
<td align="char" valign="bottom" char=".">1</td>
<td align="char" valign="bottom" char=".">16.17</td>
<td align="char" valign="bottom" char=".">0.67</td>
<td align="char" valign="bottom" char=".">0.91</td>
<td align="char" valign="bottom" char=".">0.43</td>
<td align="char" valign="bottom" char=".">0.73</td>
<td align="char" valign="bottom" char=".">0.69</td>
<td align="char" valign="bottom" char=".">0.62</td>
</tr>
<tr>
<td align="left" valign="bottom">Genus</td>
<td align="char" valign="bottom" char=".">10</td>
<td align="char" valign="bottom" char=".">102.06</td>
<td align="char" valign="bottom" char=".">0.77</td>
<td align="char" valign="bottom" char=".">0.82</td>
<td align="char" valign="bottom" char=".">0.71</td>
<td align="char" valign="bottom" char=".">0.78</td>
<td align="char" valign="bottom" char=".">0.84</td>
<td align="char" valign="bottom" char=".">0.76</td>
</tr>
<tr>
<td align="left" valign="bottom">Genus</td>
<td align="char" valign="bottom" char=".">1</td>
<td align="char" valign="bottom" char=".">7.16</td>
<td align="char" valign="bottom" char=".">0.66</td>
<td align="char" valign="bottom" char=".">0.88</td>
<td align="char" valign="bottom" char=".">0.44</td>
<td align="char" valign="bottom" char=".">0.72</td>
<td align="char" valign="bottom" char=".">0.71</td>
<td align="char" valign="bottom" char=".">0.63</td>
</tr>
<tr>
<td align="left" valign="bottom">Order</td>
<td align="char" valign="bottom" char=".">10</td>
<td align="char" valign="bottom" char=".">604</td>
<td align="char" valign="bottom" char=".">0.82</td>
<td align="char" valign="bottom" char=".">0.86</td>
<td align="char" valign="bottom" char=".">0.78</td>
<td align="char" valign="bottom" char=".">0.82</td>
<td align="char" valign="bottom" char="."><bold>0.87</bold></td>
<td align="char" valign="bottom" char=".">0.81</td>
</tr>
<tr>
<td align="left" valign="bottom">Order</td>
<td align="char" valign="bottom" char=".">1</td>
<td align="char" valign="bottom" char=".">154.25</td>
<td align="char" valign="bottom" char=".">0.76</td>
<td align="char" valign="bottom" char=".">0.82</td>
<td align="char" valign="bottom" char=".">0.71</td>
<td align="char" valign="bottom" char=".">0.78</td>
<td align="char" valign="bottom" char=".">0.81</td>
<td align="char" valign="bottom" char=".">0.76</td>
</tr>
</tbody>
</table>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="center" valign="middle" colspan="9">IBDMDB</th>
</tr>
<tr>
<th align="left" valign="middle">Taxonomic hierarchy</th>
<th align="center" valign="middle"># of Groups</th>
<th align="center" valign="middle">Average # of species</th>
<th align="center" valign="middle">Accuracy</th>
<th align="center" valign="middle">Sen</th>
<th align="center" valign="middle">Spe</th>
<th align="center" valign="middle">F measure</th>
<th align="center" valign="middle">AUC</th>
<th align="center" valign="middle">Precision</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="bottom">Family</td>
<td align="char" valign="middle" char=".">10</td>
<td align="char" valign="bottom" char=".">205.76</td>
<td align="char" valign="bottom" char=".">0.95</td>
<td align="char" valign="bottom" char=".">0.98</td>
<td align="char" valign="bottom" char=".">0.87</td>
<td align="char" valign="bottom" char=".">0.95</td>
<td align="char" valign="bottom" char=".">0.97</td>
<td align="char" valign="bottom" char=".">0.93</td>
</tr>
<tr>
<td align="left" valign="bottom">Family</td>
<td align="char" valign="middle" char=".">1</td>
<td align="char" valign="bottom" char=".">34</td>
<td align="char" valign="bottom" char=".">0.93</td>
<td align="char" valign="bottom" char=".">0.98</td>
<td align="char" valign="bottom" char=".">0.81</td>
<td align="char" valign="bottom" char=".">0.94</td>
<td align="char" valign="bottom" char=".">0.93</td>
<td align="char" valign="bottom" char=".">0.9</td>
</tr>
<tr>
<td align="left" valign="bottom">Genus</td>
<td align="char" valign="middle" char=".">10</td>
<td align="char" valign="bottom" char=".">119.4</td>
<td align="char" valign="bottom" char=".">0.92</td>
<td align="char" valign="bottom" char=".">0.98</td>
<td align="char" valign="bottom" char=".">0.82</td>
<td align="char" valign="bottom" char=".">0.95</td>
<td align="char" valign="bottom" char=".">0.97</td>
<td align="char" valign="bottom" char=".">0.93</td>
</tr>
<tr>
<td align="left" valign="bottom">Genus</td>
<td align="char" valign="middle" char=".">1</td>
<td align="char" valign="bottom" char=".">34</td>
<td align="char" valign="bottom" char=".">0.92</td>
<td align="char" valign="bottom" char=".">0.98</td>
<td align="char" valign="bottom" char=".">0.80</td>
<td align="char" valign="bottom" char=".">0.94</td>
<td align="char" valign="bottom" char=".">0.91</td>
<td align="char" valign="bottom" char=".">0.91</td>
</tr>
<tr>
<td align="left" valign="bottom">Order</td>
<td align="char" valign="middle" char=".">10</td>
<td align="char" valign="bottom" char=".">341.22</td>
<td align="char" valign="bottom" char=".">0.93</td>
<td align="char" valign="bottom" char=".">0.97</td>
<td align="char" valign="bottom" char=".">0.86</td>
<td align="char" valign="bottom" char=".">0.95</td>
<td align="char" valign="bottom" char="."><bold>0.98</bold></td>
<td align="char" valign="bottom" char=".">0.93</td>
</tr>
<tr>
<td align="left" valign="bottom">Order</td>
<td align="char" valign="middle" char=".">1</td>
<td align="char" valign="bottom" char=".">98</td>
<td align="char" valign="bottom" char=".">0.96</td>
<td align="char" valign="bottom" char=".">0.98</td>
<td align="char" valign="bottom" char=".">0.93</td>
<td align="char" valign="bottom" char=".">0.97</td>
<td align="char" valign="bottom" char=".">0.98</td>
<td align="char" valign="bottom" char=".">0.95</td>
</tr>
</tbody>
</table>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="center" valign="bottom" colspan="9">IBD</th>
</tr>
<tr>
<th align="left" valign="middle">Taxonomic hierarchy</th>
<th align="center" valign="middle"># of Groups</th>
<th align="center" valign="middle">Average # of species</th>
<th align="center" valign="middle">Accuracy</th>
<th align="center" valign="middle">Sen</th>
<th align="center" valign="middle">Spe</th>
<th align="center" valign="middle">F measure</th>
<th align="center" valign="middle">AUC</th>
<th align="center" valign="middle">Precision</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="bottom">Family</td>
<td align="char" valign="bottom" char=".">10</td>
<td align="char" valign="bottom" char=".">260.24</td>
<td align="char" valign="bottom" char=".">0.82</td>
<td align="char" valign="bottom" char=".">0.85</td>
<td align="char" valign="bottom" char=".">0.79</td>
<td align="char" valign="bottom" char=".">0.82</td>
<td align="char" valign="bottom" char="."><bold>0.91</bold></td>
<td align="char" valign="bottom" char=".">0.81</td>
</tr>
<tr>
<td align="left" valign="bottom">Family</td>
<td align="char" valign="bottom" char=".">1</td>
<td align="char" valign="bottom" char=".">51.59</td>
<td align="char" valign="bottom" char=".">0.78</td>
<td align="char" valign="bottom" char=".">0.78</td>
<td align="char" valign="bottom" char=".">0.78</td>
<td align="char" valign="bottom" char=".">0.78</td>
<td align="char" valign="bottom" char=".">0.86</td>
<td align="char" valign="bottom" char=".">0.78</td>
</tr>
<tr>
<td align="left" valign="bottom">Genus</td>
<td align="char" valign="bottom" char=".">10</td>
<td align="char" valign="bottom" char=".">121.78</td>
<td align="char" valign="bottom" char=".">0.81</td>
<td align="char" valign="bottom" char=".">0.83</td>
<td align="char" valign="bottom" char=".">0.79</td>
<td align="char" valign="bottom" char=".">0.81</td>
<td align="char" valign="bottom" char=".">0.88</td>
<td align="char" valign="bottom" char=".">0.8</td>
</tr>
<tr>
<td align="left" valign="bottom">Genus</td>
<td align="char" valign="bottom" char=".">1</td>
<td align="char" valign="bottom" char=".">12.26</td>
<td align="char" valign="bottom" char=".">0.7</td>
<td align="char" valign="bottom" char=".">0.67</td>
<td align="char" valign="bottom" char=".">0.74</td>
<td align="char" valign="bottom" char=".">0.69</td>
<td align="char" valign="bottom" char=".">0.78</td>
<td align="char" valign="bottom" char=".">0.73</td>
</tr>
<tr>
<td align="left" valign="bottom">Order</td>
<td align="char" valign="bottom" char=".">10</td>
<td align="char" valign="bottom" char=".">608.27</td>
<td align="char" valign="bottom" char=".">0.82</td>
<td align="char" valign="bottom" char=".">0.82</td>
<td align="char" valign="bottom" char=".">0.81</td>
<td align="char" valign="bottom" char=".">0.82</td>
<td align="char" valign="bottom" char=".">0.9</td>
<td align="char" valign="bottom" char=".">0.82</td>
</tr>
<tr>
<td align="left" valign="bottom">Order</td>
<td align="char" valign="bottom" char=".">1</td>
<td align="char" valign="bottom" char=".">174.86</td>
<td align="char" valign="bottom" char=".">0.81</td>
<td align="char" valign="bottom" char=".">0.82</td>
<td align="char" valign="bottom" char=".">0.8</td>
<td align="char" valign="bottom" char=".">0.81</td>
<td align="char" valign="bottom" char=".">0.9</td>
<td align="char" valign="bottom" char=".">0.81</td>
</tr>
</tbody>
</table>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="center" valign="bottom" colspan="9">T2D</th>
</tr>
<tr>
<th align="left" valign="middle">Taxonomic hierarchy</th>
<th align="center" valign="middle"># of groups</th>
<th align="center" valign="middle">Average # of species</th>
<th align="center" valign="middle">Accuracy</th>
<th align="center" valign="middle">Sen</th>
<th align="center" valign="middle">Spe</th>
<th align="center" valign="middle">F measure</th>
<th align="center" valign="middle">AUC</th>
<th align="center" valign="middle">Precision</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="bottom">Family</td>
<td align="char" valign="bottom" char=".">10</td>
<td align="char" valign="bottom" char=".">321.16</td>
<td align="char" valign="bottom" char=".">0.65</td>
<td align="char" valign="bottom" char=".">0.68</td>
<td align="char" valign="bottom" char=".">0.63</td>
<td align="char" valign="bottom" char=".">0.66</td>
<td align="char" valign="bottom" char=".">0.71</td>
<td align="char" valign="bottom" char=".">0.65</td>
</tr>
<tr>
<td align="left" valign="bottom">Family</td>
<td align="char" valign="bottom" char=".">1</td>
<td align="char" valign="bottom" char=".">39.86</td>
<td align="char" valign="bottom" char=".">0.59</td>
<td align="char" valign="bottom" char=".">0.71</td>
<td align="char" valign="bottom" char=".">0.47</td>
<td align="char" valign="bottom" char=".">0.63</td>
<td align="char" valign="bottom" char=".">0.63</td>
<td align="char" valign="bottom" char=".">0.58</td>
</tr>
<tr>
<td align="left" valign="bottom">Genus</td>
<td align="char" valign="bottom" char=".">10</td>
<td align="char" valign="bottom" char=".">129.8</td>
<td align="char" valign="bottom" char=".">0.64</td>
<td align="char" valign="bottom" char=".">0.64</td>
<td align="char" valign="bottom" char=".">0.64</td>
<td align="char" valign="bottom" char=".">0.64</td>
<td align="char" valign="bottom" char=".">0.69</td>
<td align="char" valign="bottom" char=".">0.65</td>
</tr>
<tr>
<td align="left" valign="bottom">Genus</td>
<td align="char" valign="bottom" char=".">1</td>
<td align="char" valign="bottom" char=".">15.94</td>
<td align="char" valign="bottom" char=".">0.56</td>
<td align="char" valign="bottom" char=".">0.62</td>
<td align="char" valign="bottom" char=".">0.49</td>
<td align="char" valign="bottom" char=".">0.58</td>
<td align="char" valign="bottom" char=".">0.58</td>
<td align="char" valign="bottom" char=".">0.55</td>
</tr>
<tr>
<td align="left" valign="bottom">Order</td>
<td align="char" valign="bottom" char=".">10</td>
<td align="char" valign="bottom" char=".">596.99</td>
<td align="char" valign="bottom" char=".">0.65</td>
<td align="char" valign="bottom" char=".">0.65</td>
<td align="char" valign="bottom" char=".">0.64</td>
<td align="char" valign="bottom" char=".">0.64</td>
<td align="char" valign="bottom" char="."><bold>0.72</bold></td>
<td align="char" valign="bottom" char=".">0.65</td>
</tr>
<tr>
<td align="left" valign="bottom">Order</td>
<td align="char" valign="bottom" char=".">1</td>
<td align="char" valign="bottom" char=".">138.28</td>
<td align="char" valign="bottom" char=".">0.59</td>
<td align="char" valign="bottom" char=".">0.67</td>
<td align="char" valign="bottom" char=".">0.52</td>
<td align="char" valign="bottom" char=".">0.62</td>
<td align="char" valign="bottom" char=".">0.64</td>
<td align="char" valign="bottom" char=".">0.59</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>The results in bold in table represent the best AUC results.</p>
</table-wrap-foot>
</table-wrap>
<p>microBiomeGSM reports important groups of features that are detected at different taxonomic levels for the disease under study. <xref ref-type="table" rid="tab4">Table 4</xref> lists the top 10 important groups that are identified by microBiomeGSM for three different taxonomic levels on four different datasets. The identified features are ranked by their importance scores from high to low. The feature with the highest importance value is the strongest candidate to be announced as potential taxonomic biomarker for the disease under investigation.</p>
<table-wrap position="float" id="tab4">
<label>Table 4</label>
<caption><p>Top 10 groups identified by microBiomeGSM for different taxonomic levels, applied on all microbiome datasets.</p></caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="center" valign="top" colspan="4">CRC</th>
</tr>
<tr>
<th align="left" valign="top">#</th>
<th align="center" valign="top" colspan="3">Taxonomic levels</th>
</tr>
<tr>
<th align="left" valign="top">Rank</th>
<th align="left" valign="top">Family</th>
<th align="left" valign="top">Order</th>
<th align="left" valign="top">Genus</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">1</td>
<td align="left" valign="middle">PEPTOSTREPTOCOCCACEAE</td>
<td align="left" valign="middle">CLOSTRIDIALES</td>
<td align="left" valign="middle">PARVIMONAS</td>
</tr>
<tr>
<td align="left" valign="middle">2</td>
<td align="left" valign="bottom">PEPTONIPHILACEAE</td>
<td align="left" valign="bottom">TISSIERELLALES</td>
<td align="left" valign="bottom">PEPTOSTREPTOCOCCUS</td>
</tr>
<tr>
<td align="left" valign="middle">3</td>
<td align="left" valign="bottom">FUSOBACTERIACEAE</td>
<td align="left" valign="bottom">BACTEROIDALES</td>
<td align="left" valign="bottom">FUSOBACTERIUM</td>
</tr>
<tr>
<td align="left" valign="middle">4</td>
<td align="left" valign="bottom">BACILLALES_UNCLASSIFIED</td>
<td align="left" valign="bottom">FUSOBACTERIALES</td>
<td align="left" valign="bottom">GEMELLA</td>
</tr>
<tr>
<td align="left" valign="middle">5</td>
<td align="left" valign="bottom">VEILLONELLACEAE</td>
<td align="left" valign="bottom">BACILLALES</td>
<td align="left" valign="bottom">DIALISTER</td>
</tr>
<tr>
<td align="left" valign="middle">6</td>
<td align="left" valign="bottom">LACHNOSPIRACEAE</td>
<td align="left" valign="bottom">VEILLONELLALES</td>
<td align="left" valign="bottom">LACHNOCLOSTRIDIUM</td>
</tr>
<tr>
<td align="left" valign="middle">7</td>
<td align="left" valign="bottom">ERYSIPELOTRICHACEAE</td>
<td align="left" valign="bottom">ERYSIPELOTRICHALES</td>
<td align="left" valign="bottom">PREVOTELLA</td>
</tr>
<tr>
<td align="left" valign="middle">8</td>
<td align="left" valign="bottom">RUMINOCOCCACEAE</td>
<td align="left" valign="bottom">LACTOBACILLALES</td>
<td align="left" valign="bottom">STREPTOCOCCUS</td>
</tr>
<tr>
<td align="left" valign="middle">9</td>
<td align="left" valign="bottom">PREVOTELLACEAE</td>
<td align="left" valign="bottom">ACTINOMYCETALES</td>
<td align="left" valign="bottom">PORPHYROMONAS</td>
</tr>
<tr>
<td align="left" valign="middle">10</td>
<td align="left" valign="bottom">STREPTOCOCCACEAE</td>
<td align="left" valign="bottom">DESULFOVIBRIONALES</td>
<td align="left" valign="bottom">SOLOBACTERIUM</td>
</tr>
</tbody>
</table>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="center" valign="top" colspan="4">IBDMDB</th>
</tr>
<tr>
<th align="left" valign="top">#</th>
<th align="center" valign="middle" colspan="3">Taxonomic levels</th>
</tr>
<tr>
<th align="left" valign="top">Rank</th>
<th align="left" valign="middle">Family</th>
<th align="left" valign="middle">Order</th>
<th align="left" valign="middle">Genus</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">1</td>
<td align="left" valign="middle">BACTEROIDACEAE</td>
<td align="left" valign="middle">BACTEROIDALES</td>
<td align="left" valign="middle">BACTEROIDES</td>
</tr>
<tr>
<td align="left" valign="middle">2</td>
<td align="left" valign="middle">LACHNOSPIRACEAE</td>
<td align="left" valign="middle">CLOSTRIDIALES</td>
<td align="left" valign="middle">ALISTIPES</td>
</tr>
<tr>
<td align="left" valign="middle">3</td>
<td align="left" valign="middle">RUMINOCOCCACEAE</td>
<td align="left" valign="middle">FIRMICUTES_UNCLASSIFIED</td>
<td align="left" valign="middle">EUBACTERIUM</td>
</tr>
<tr>
<td align="left" valign="middle">4</td>
<td align="left" valign="middle">RIKENELLACEAE</td>
<td align="left" valign="middle">VEILLONELLALES</td>
<td align="left" valign="middle">ROSEBURIA</td>
</tr>
<tr>
<td align="left" valign="middle">5</td>
<td align="left" valign="middle">FIRMICUTES_UNCLASSIFIED</td>
<td align="left" valign="middle">BURKHOLDERIALES</td>
<td align="left" valign="middle">FIRMICUTES_UNCLASSIFIED</td>
</tr>
<tr>
<td align="left" valign="middle">6</td>
<td align="left" valign="middle">TANNERELLACEAE</td>
<td align="left" valign="middle">METHANOMASSILIICOCCALES</td>
<td align="left" valign="middle">PARABACTEROIDES</td>
</tr>
<tr>
<td align="left" valign="middle">7</td>
<td align="left" valign="middle">EUBACTERIACEAE</td>
<td align="left" valign="middle">DESULFOVIBRIONALES</td>
<td align="left" valign="middle">RUMINOCOCCUS</td>
</tr>
<tr>
<td align="left" valign="middle">8</td>
<td align="left" valign="middle">CLOSTRIDIACEAE</td>
<td align="left" valign="middle">ERYSIPELOTRICHALES</td>
<td align="left" valign="middle">COPROCOCCUS</td>
</tr>
<tr>
<td align="left" valign="middle">9</td>
<td align="left" valign="middle">VEILLONELLACEAE</td>
<td align="left" valign="middle">BIFIDOBACTERIALES</td>
<td align="left" valign="middle">BLAUTIA</td>
</tr>
<tr>
<td align="left" valign="middle">10</td>
<td align="left" valign="middle">ODORIBACTERACEAE</td>
<td align="left" valign="middle">EGGERTHELLALES</td>
<td align="left" valign="middle">CLOSTRIDIUM</td>
</tr>
</tbody>
</table>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="center" valign="top" colspan="4">IBD</th>
</tr>
<tr>
<th align="left" valign="top">#</th>
<th align="center" valign="middle" colspan="3">Taxonomic levels</th>
</tr>
<tr>
<th align="left" valign="top">Rank</th>
<th align="left" valign="middle">Family</th>
<th align="left" valign="middle">Order</th>
<th align="left" valign="middle">Genus</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">1</td>
<td align="left" valign="bottom">LACHNOSPIRACEAE</td>
<td align="left" valign="bottom">CLOSTRIDIALES</td>
<td align="left" valign="bottom">BLAUTIA</td>
</tr>
<tr>
<td align="left" valign="middle">2</td>
<td align="left" valign="bottom">BIFIDOBACTERIACEAE</td>
<td align="left" valign="bottom">CORIOBACTERIALES</td>
<td align="left" valign="bottom">BIFIDOBACTERIUM</td>
</tr>
<tr>
<td align="left" valign="middle">3</td>
<td align="left" valign="bottom">CORIOBACTERIACEAE</td>
<td align="left" valign="bottom">BIFIDOBACTERIALES</td>
<td align="left" valign="bottom">EUBACTERIUM</td>
</tr>
<tr>
<td align="left" valign="middle">4</td>
<td align="left" valign="bottom">RUMINOCOCCACEAE</td>
<td align="left" valign="bottom">ERYSIPELOTRICHALES</td>
<td align="left" valign="bottom">DOREA</td>
</tr>
<tr>
<td align="left" valign="middle">5</td>
<td align="left" valign="bottom">ERYSIPELOTRICHACEAE</td>
<td align="left" valign="bottom">BACTEROIDALES</td>
<td align="left" valign="bottom">COLLINSELLA</td>
</tr>
<tr>
<td align="left" valign="middle">6</td>
<td align="left" valign="bottom">CLOSTRIDIALES_FAMILY_XIII_INCERTAE_SEDIS</td>
<td align="left" valign="bottom">LACTOBACILLALES</td>
<td align="left" valign="bottom">PEPTOSTREPTOCOCCUS</td>
</tr>
<tr>
<td align="left" valign="middle">7</td>
<td align="left" valign="bottom">EUBACTERIACEAE</td>
<td align="left" valign="bottom">SELENOMONADALES</td>
<td align="left" valign="bottom">COPROCOCCUS</td>
</tr>
<tr>
<td align="left" valign="middle">8</td>
<td align="left" valign="bottom">PEPTOSTREPTOCOCCACEAE</td>
<td align="left" valign="bottom">VERRUCOMICROBIALES</td>
<td align="left" valign="bottom">ERYSIPELOTRICHACEAE_NONAME</td>
</tr>
<tr>
<td align="left" valign="middle">9</td>
<td align="left" valign="bottom">CARNOBACTERIACEAE</td>
<td align="left" valign="bottom">CANDIDATUS_SACCHARIBACTERIA_NONAME</td>
<td align="left" valign="bottom">LACHNOSPIRACEAE_NONAME</td>
</tr>
<tr>
<td align="left" valign="middle">10</td>
<td align="left" valign="bottom">CLOSTRIDIACEAE</td>
<td align="left" valign="bottom">BACILLALES</td>
<td align="left" valign="bottom">BACTEROIDES</td>
</tr>
</tbody>
</table>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="center" valign="top" colspan="4">T2D</th>
</tr>
<tr>
<th align="left" valign="top">#</th>
<th align="center" valign="middle" colspan="3">Taxonomic levels</th>
</tr>
<tr>
<th align="left" valign="top">Rank</th>
<th align="left" valign="middle">Family</th>
<th align="left" valign="middle">Order</th>
<th align="left" valign="middle">Genus</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">1</td>
<td align="left" valign="bottom">LACHNOSPIRACEAE</td>
<td align="left" valign="bottom">CLOSTRIDIALES</td>
<td align="left" valign="bottom">EUBACTERIUM</td>
</tr>
<tr>
<td align="left" valign="middle">2</td>
<td align="left" valign="bottom">BIFIDOBACTERIACEAE</td>
<td align="left" valign="bottom">BIFIDOBACTERIALES</td>
<td align="left" valign="bottom">BIFIDOBACTERIUM</td>
</tr>
<tr>
<td align="left" valign="middle">3</td>
<td align="left" valign="bottom">RUMINOCOCCACEAE</td>
<td align="left" valign="bottom">CORIOBACTERIALES</td>
<td align="left" valign="bottom">BLAUTIA</td>
</tr>
<tr>
<td align="left" valign="middle">4</td>
<td align="left" valign="bottom">EUBACTERIACEAE</td>
<td align="left" valign="bottom">BACTEROIDALES</td>
<td align="left" valign="bottom">DOREA</td>
</tr>
<tr>
<td align="left" valign="middle">5</td>
<td align="left" valign="bottom">CORIOBACTERIACEAE</td>
<td align="left" valign="bottom">LACTOBACILLALES</td>
<td align="left" valign="bottom">LACHNOSPIRACEAE_NONAME</td>
</tr>
<tr>
<td align="left" valign="middle">6</td>
<td align="left" valign="bottom">CLOSTRIDIALES_FAMILY_XIII_INCERTAE_SEDIS</td>
<td align="left" valign="bottom">ERYSIPELOTRICHALES</td>
<td align="left" valign="bottom">RUMINOCOCCUS</td>
</tr>
<tr>
<td align="left" valign="middle">7</td>
<td align="left" valign="bottom">ERYSIPELOTRICHACEAE</td>
<td align="left" valign="bottom">SELENOMONADALES</td>
<td align="left" valign="bottom">COPROCOCCUS</td>
</tr>
<tr>
<td align="left" valign="middle">8</td>
<td align="left" valign="bottom">PEPTOSTREPTOCOCCACEAE</td>
<td align="left" valign="bottom">VERRUCOMICROBIALES</td>
<td align="left" valign="bottom">PEPTOSTREPTOCOCCUS</td>
</tr>
<tr>
<td align="left" valign="middle">9</td>
<td align="left" valign="bottom">CARNOBACTERIACEAE</td>
<td align="left" valign="bottom">METHANOBACTERIALES</td>
<td align="left" valign="bottom">ERYSIPELOTRICHACEAE_NONAME</td>
</tr>
<tr>
<td align="left" valign="middle">10</td>
<td align="left" valign="bottom">BACTEROIDACEAE</td>
<td align="left" valign="bottom">BACILLALES</td>
<td align="left" valign="bottom">GRANULICATELLA</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The microBiomeGSM tool lists a number of associated species for each identified group. The species included in the top 5 significant groups are listed in <xref rid="SM1" ref-type="supplementary-material">Supplementary Tables S3&#x2013;S5</xref> for family, order, and genus taxonomic levels, respectively for four different datasets. All species for the family, order, and genus taxonomic levels for the T2D, IBDMDB and CRC datasets can be found in <xref rid="SM1" ref-type="supplementary-material">Supplementary Tables S6&#x2013;S14</xref>, respectively.</p>
<p>For the IBDMDB dataset, the changes in the AUC score when the number of groups is increased from 1 to 10 are shown in <xref rid="SM1" ref-type="supplementary-material">Supplementary Figure S1</xref>. For the IBDMDB dataset, a high AUC score is obtained at the order taxonomic level. When the number of groups was increased, the AUC score decreased relatively, and no significant change was observed after 5 groups. At the genus and family taxonomic levels, there is a significant increase in the AUC score until 5 groups are combined and no significant change after 5 groups.</p>
</sec>
<sec id="sec10">
<label>3.2</label>
<title>Comparing against traditional machine learning methods</title>
<p>Our Grouping-Scoring-Modeling (G-S-M) approach emerges as a paradigm shift from traditional feature selection methods. Instead of pinpointing individual informative features, the GSM methodology groups these features. These groups are then scored, and a classification model is built using these top-ranking feature conglomerates. The versatility of the GSM method, as detailed in our prior work (<xref ref-type="bibr" rid="ref70">Yousef et al., 2021b</xref>), lies in its adaptability. Groups can be created either by computational/statistical methods or by using domain-specific knowledge. In order to use the GSM strategy for a given dataset, a deep domain expertise is required to skillfully define these groups, which makes each application different. The modifications required to tailor the G-S-M approach to the unique needs of microbiome research highlight the adaptability of the G-S-M method and the novelty of our current study.</p>
<p>We have comparatively evaluated the performance of microBiomeGSM against 4 different classifiers and 6 different feature selection methods using the same datasets. All algorithms are run with default parameters. The developed approach and feature selection methods were executed multiple times, and the results were averaged and shared. <xref ref-type="table" rid="tab5">Table 5</xref> shows the performance of the different feature selection algorithms and different classifiers on the same disease associated microbiome datasets. In these experiments, the number of features was set to 100. The best result for the IBDMDB dataset is obtained by using the XGBoost feature selection algorithm in combination with the Random Forest classification algorithm with 98% AUC. For the CRC dataset, the best result is obtained by using the XGBoost feature selection algorithm in combination with the Random Forest classification algorithm with an AUC of 85%. For the IBD dataset, the best result is obtained using the Random Forest classification algorithm with 92% AUC and the SKB feature selection algorithm. For the T2D dataset, the best result is obtained by using the XGBoost feature selection algorithm in combination with the Random Forest classification algorithm with 70% AUC.</p>
<table-wrap position="float" id="tab5">
<label>Table 5</label>
<caption><p>Area under the curve (AUC) results obtained using 100 features for different feature selection methods and classifiers for all dataset.</p></caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="center" valign="top" colspan="7">CRC</th>
</tr>
<tr>
<th align="left" valign="top">Model</th>
<th align="center" valign="top">SKB</th>
<th align="center" valign="top">IG</th>
<th align="center" valign="top">XGB</th>
<th align="center" valign="top">FCBF</th>
<th align="center" valign="top">MRMR</th>
<th align="center" valign="top">CMIM</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Adaboost</td>
<td align="char" valign="bottom" char="&#x00B1;">0.75 &#x00B1; 0.02</td>
<td align="char" valign="bottom" char="&#x00B1;">0.71 &#x00B1; 0.05</td>
<td align="char" valign="bottom" char="&#x00B1;">0.78 &#x00B1; 0.04</td>
<td align="char" valign="bottom" char="&#x00B1;">0.71 &#x00B1; 0.05</td>
<td align="char" valign="bottom" char="&#x00B1;">0.63 &#x00B1; 0.06</td>
<td align="char" valign="bottom" char="&#x00B1;">0.77 &#x00B1; 0.04</td>
</tr>
<tr>
<td align="left" valign="top">DT</td>
<td align="char" valign="bottom" char="&#x00B1;">0.67 &#x00B1; 0.04</td>
<td align="char" valign="bottom" char="&#x00B1;">0.64 &#x00B1; 0.04</td>
<td align="char" valign="bottom" char="&#x00B1;">0.69 &#x00B1; 0.04</td>
<td align="char" valign="bottom" char="&#x00B1;">0.63 &#x00B1; 0.06</td>
<td align="char" valign="bottom" char="&#x00B1;">0.61 &#x00B1; 0.04</td>
<td align="char" valign="bottom" char="&#x00B1;">0.65 &#x00B1; 0.05</td>
</tr>
<tr>
<td align="left" valign="top">Logitboost</td>
<td align="char" valign="bottom" char="&#x00B1;">0.76 &#x00B1; 0.04</td>
<td align="char" valign="bottom" char="&#x00B1;">0.72 &#x00B1; 0.05</td>
<td align="char" valign="bottom" char="&#x00B1;">0.78 &#x00B1; 0.06</td>
<td align="char" valign="bottom" char="&#x00B1;">0.70 &#x00B1; 0.04</td>
<td align="char" valign="bottom" char="&#x00B1;">0.64 &#x00B1; 0.06</td>
<td align="char" valign="bottom" char="&#x00B1;">0.76 &#x00B1; 0.05</td>
</tr>
<tr>
<td align="left" valign="top">RF</td>
<td align="char" valign="bottom" char="&#x00B1;">0.82 &#x00B1; 0.03</td>
<td align="char" valign="bottom" char="&#x00B1;">0.79 &#x00B1; 0.04</td>
<td align="char" valign="bottom" char="&#x00B1;"><bold>0.85 &#x00B1; 0.03</bold></td>
<td align="char" valign="bottom" char="&#x00B1;">0.77 &#x00B1; 0.05</td>
<td align="char" valign="bottom" char="&#x00B1;">0.74 &#x00B1; 0.04</td>
<td align="char" valign="bottom" char="&#x00B1;">0.80 &#x00B1; 0.03</td>
</tr>
</tbody>
</table>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="center" valign="middle" colspan="7">IBDMDB</th>
</tr>
<tr>
<th align="left" valign="top">Model</th>
<th align="center" valign="top">SKB</th>
<th align="center" valign="top">IG</th>
<th align="center" valign="top">XGB</th>
<th align="center" valign="top">FCBF</th>
<th align="center" valign="top">MRMR</th>
<th align="center" valign="top">CMIM</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Adaboost</td>
<td align="char" valign="top" char="&#x00B1;">0.89 &#x00B1; 0.04</td>
<td align="char" valign="top" char="&#x00B1;">0.90 &#x00B1; 0.03</td>
<td align="char" valign="top" char="&#x00B1;">0.89 &#x00B1; 0.06</td>
<td align="char" valign="top" char="&#x00B1;">0.49 &#x00B1; 0.08</td>
<td align="char" valign="top" char="&#x00B1;">0.51 &#x00B1; 0.08</td>
<td align="char" valign="top" char="&#x00B1;">0.51 &#x00B1; 0.08</td>
</tr>
<tr>
<td align="left" valign="top">DT</td>
<td align="char" valign="top" char="&#x00B1;">0.83 &#x00B1; 0.03</td>
<td align="char" valign="top" char="&#x00B1;">0.82 &#x00B1; 0.04</td>
<td align="char" valign="top" char="&#x00B1;">0.84 &#x00B1; 0.03</td>
<td align="char" valign="top" char="&#x00B1;">0.46 &#x00B1; 0.07</td>
<td align="char" valign="top" char="&#x00B1;">0.50 &#x00B1; 0.07</td>
<td align="char" valign="top" char="&#x00B1;">0.50 &#x00B1; 0.06</td>
</tr>
<tr>
<td align="left" valign="top">Logitboost</td>
<td align="char" valign="top" char="&#x00B1;">0.89 &#x00B1; 0.04</td>
<td align="char" valign="top" char="&#x00B1;">0.91 &#x00B1; 0.03</td>
<td align="char" valign="top" char="&#x00B1;">0.86 &#x00B1; 0.06</td>
<td align="char" valign="top" char="&#x00B1;">0.50 &#x00B1; 0.06</td>
<td align="char" valign="top" char="&#x00B1;">0.51 &#x00B1; 0.08</td>
<td align="char" valign="top" char="&#x00B1;">0.49 &#x00B1; 0.08</td>
</tr>
<tr>
<td align="left" valign="top">RF</td>
<td align="char" valign="top" char="&#x00B1;">0.96 &#x00B1; 0.01</td>
<td align="char" valign="top" char="&#x00B1;">0.96 &#x00B1; 0.01</td>
<td align="char" valign="top" char="&#x00B1;"><bold>0.98 &#x00B1; 0.01</bold></td>
<td align="char" valign="top" char="&#x00B1;">0.46 &#x00B1; 0.1</td>
<td align="char" valign="top" char="&#x00B1;">0.54 &#x00B1; 0.08</td>
<td align="char" valign="top" char="&#x00B1;">0.52 &#x00B1; 0.07</td>
</tr>
</tbody>
</table>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="center" valign="middle" colspan="7">IBD</th>
</tr>
<tr>
<th align="left" valign="top">Model</th>
<th align="center" valign="top">SKB</th>
<th align="center" valign="top">IG</th>
<th align="center" valign="top">XGB</th>
<th align="center" valign="top">FCBF</th>
<th align="center" valign="top">MRMR</th>
<th align="center" valign="top">CMIM</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Adaboost</td>
<td align="char" valign="bottom" char="&#x00B1;">0,90 &#x00B1; 0.07</td>
<td align="char" valign="bottom" char="&#x00B1;">0,89 &#x00B1; 0.03</td>
<td align="char" valign="bottom" char="&#x00B1;">0,91 &#x00B1; 0.03</td>
<td align="char" valign="bottom" char="&#x00B1;">0,51 &#x00B1; 0.06</td>
<td align="char" valign="bottom" char="&#x00B1;">0,51 &#x00B1; 0.03</td>
<td align="char" valign="bottom" char="&#x00B1;">0,66 &#x00B1; 0.08</td>
</tr>
<tr>
<td align="left" valign="top">DT</td>
<td align="char" valign="bottom" char="&#x00B1;">0,78 &#x00B1; 0.08</td>
<td align="char" valign="bottom" char="&#x00B1;">0,70 &#x00B1; 0.08</td>
<td align="char" valign="bottom" char="&#x00B1;">0,73 &#x00B1; 0.07</td>
<td align="char" valign="bottom" char="&#x00B1;">0,53 &#x00B1; 0.08</td>
<td align="char" valign="bottom" char="&#x00B1;">0,51 &#x00B1; 0.04</td>
<td align="char" valign="bottom" char="&#x00B1;">0,56 &#x00B1; 0.09</td>
</tr>
<tr>
<td align="left" valign="top">Logitboost</td>
<td align="char" valign="bottom" char="&#x00B1;">0,90 &#x00B1; 0.04</td>
<td align="char" valign="bottom" char="&#x00B1;">0,90 &#x00B1; 0.05</td>
<td align="char" valign="bottom" char="&#x00B1;">0,92 &#x00B1; 0.05</td>
<td align="char" valign="bottom" char="&#x00B1;">0,55 &#x00B1; 0.1</td>
<td align="char" valign="bottom" char="&#x00B1;">0,53 &#x00B1; 0.05</td>
<td align="char" valign="bottom" char="&#x00B1;">0,59 &#x00B1; 0.1</td>
</tr>
<tr>
<td align="left" valign="top">RF</td>
<td align="char" valign="bottom" char="&#x00B1;"><bold>0,92 &#x00B1; 0.03</bold></td>
<td align="char" valign="bottom" char="&#x00B1;">0,88 &#x00B1; 0.06</td>
<td align="char" valign="bottom" char="&#x00B1;">0,91 &#x00B1; 0.04</td>
<td align="char" valign="bottom" char="&#x00B1;">0,53 &#x00B1; 0.09</td>
<td align="char" valign="bottom" char="&#x00B1;">0,55 &#x00B1; 0.07</td>
<td align="char" valign="bottom" char="&#x00B1;">0,63 &#x00B1; 0.11</td>
</tr>
</tbody>
</table>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="center" valign="middle" colspan="7">T2D</th>
</tr>
<tr>
<th align="left" valign="top">Model</th>
<th align="center" valign="top">SKB</th>
<th align="center" valign="top">IG</th>
<th align="center" valign="top">XGB</th>
<th align="center" valign="top">FCBF</th>
<th align="center" valign="top">MRMR</th>
<th align="center" valign="top">CMIM</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Adaboost</td>
<td align="char" valign="bottom" char="&#x00B1;">0,56 &#x00B1; 0.12</td>
<td align="char" valign="bottom" char="&#x00B1;">0,60 &#x00B1; 0.05</td>
<td align="char" valign="bottom" char="&#x00B1;">0,64 &#x00B1; 0.07</td>
<td align="char" valign="bottom" char="&#x00B1;">0,50 &#x00B1; 0.10</td>
<td align="char" valign="bottom" char="&#x00B1;">0,5 &#x00B1; 0.01</td>
<td align="char" valign="bottom" char="&#x00B1;">0,50 &#x00B1; 0.12</td>
</tr>
<tr>
<td align="left" valign="top">DT</td>
<td align="char" valign="bottom" char="&#x00B1;">0,52 &#x00B1; 0.08</td>
<td align="char" valign="bottom" char="&#x00B1;">0,52 &#x00B1; 0.08</td>
<td align="char" valign="bottom" char="&#x00B1;">0,53 &#x00B1; 0.05</td>
<td align="char" valign="bottom" char="&#x00B1;">0,41 &#x00B1; 0.10</td>
<td align="char" valign="bottom" char="&#x00B1;">0,51 &#x00B1; 0.02</td>
<td align="char" valign="bottom" char="&#x00B1;">0,49 &#x00B1; 0.10</td>
</tr>
<tr>
<td align="left" valign="top">Logitboost</td>
<td align="char" valign="bottom" char="&#x00B1;">0,55 &#x00B1; 0.10</td>
<td align="char" valign="bottom" char="&#x00B1;">0,58 &#x00B1; 0.09</td>
<td align="char" valign="bottom" char="&#x00B1;">0,62 &#x00B1; 0.10</td>
<td align="char" valign="bottom" char="&#x00B1;">0,48 &#x00B1; 0.08</td>
<td align="char" valign="bottom" char="&#x00B1;">0,50 &#x00B1; 0.01</td>
<td align="char" valign="bottom" char="&#x00B1;">0,51 &#x00B1; 0.11</td>
</tr>
<tr>
<td align="left" valign="top">RF</td>
<td align="char" valign="bottom" char="&#x00B1;">0,62 &#x00B1; 0.11</td>
<td align="char" valign="bottom" char="&#x00B1;">0,62 &#x00B1; 0.07</td>
<td align="char" valign="bottom" char="&#x00B1;"><bold>0,70 &#x00B1; 0.06</bold></td>
<td align="char" valign="bottom" char="&#x00B1;">0,49 &#x00B1; 0.08</td>
<td align="char" valign="bottom" char="&#x00B1;">0,51 &#x00B1; 0.03</td>
<td align="char" valign="bottom" char="&#x00B1;">0,54 &#x00B1; 0.10</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>The results in bold in table represent the best AUC results for the respective disease (CRC, IBDMDB, IBD, T2D).</p>
</table-wrap-foot>
</table-wrap>
<p>We would like to note that the primary objective of microBiomeGSM is not to compete with other feature selection methods (FS). Even if microBiomeGSM&#x2019;s performance is on par with or slightly less favorable than other FS methods, its fundamental contribution lies in identifying the most informative microbiomes. These microbiomes play a pivotal role in aiding researchers in gaining a deeper understanding of the biological underpinnings of the disease under investigation. In essence, microBiomeGSM&#x2019;s value lies in its ability to contribute to the advancement of biological knowledge, rather than merely outperforming other feature selection techniques.</p>
<p><xref ref-type="table" rid="tab6">Table 6</xref> shows the performance metrics of microBiomeGSM for each taxonomic level for four different datasets. The # of species column shows the number of species (features/variables) used to train and test the model. Since the number of species changes in each iteration of MCCV, we also report the standard deviation. Performance metrics are reported as the average of 100 iterations with the corresponding standard deviation. For the CRC dataset, among different classifiers the RF algorithm has the highest performance for all calculated metrics including the accuracy, sensitivity, specificity, precision, and AUC metric. The AdaBoost, LogitBoost and DT models show lower performance compared to the RF model. The performance metrics of these three algorithms are similar but not as high as RF model. At the order taxonomic level, the mean values of the performance metrics are stable and the standard deviations are low. This indicates that the order level is a more appropriate choice for CRC classification. Comparing the RF model and the microBiomeGSM model, similar performance metrics are obtained for the CRC dataset, but it is worth mentioning that the number of features used in the proposed tool is lower. In other words, for the CRC dataset the microBiomeGSM model can accurately classify using fewer taxonomic features. For the IBDMDB dataset, among different classifiers the RF algorithm has the highest accuracy, sensitivity, specificity, precision, and AUC values. In particular, RF model achieved very high sensitivity and AUC values. For the IBDMDB dataset, the microBiomeGSM tool achieves an AUC of 98% for the order taxon level, the same performance metrics as obtained by the RF classification algorithm. However, the microBiomeGSM tool uses 341 features for the order taxon level, while the RF model uses 579 features. For IBD dataset, the RF algorithm generates the highest performance on several metrics, including accuracy, sensitivity, specificity, precision, and AUC. It performs particularly well on sensitivity and AUC. In our analysis, microBiomeGSM achieved an impressive AUC value of 91% at the family taxon level. Equally remarkable is the similar performance of the RF classification algorithm (an AUC of 92%) for the same task. However, it is important to highlight an important difference between these two approaches. For IBD dataset the RF classification algorithm achieved an AUC of 92% by using a much larger set of features (1,456 features) for the classification task. For the same dataset, the microBiomeGSM tool also showed remarkable performance (an AUC value of 91%). In stark contrast, microBiomeGSM achieved nearly equivalent AUC performance while using a much smaller set of features, only 260 features. This divergence in feature usage highlights the effectiveness and potential advantages of the microBiomeGSM tool in extracting meaningful information from microbiome data while optimizing computational resources. For T2D dataset, the RF classification algorithm outperforms other classification algorithms on several performance metrics including accuracy, sensitivity, specificity, precision and AUC. microBiomeGSM achieved an AUC value of 72% at the order taxon level. Interestingly, a similar level of performance is observed using the RF classification algorithm, which achieves an AUC value of 75%. However, it is important to note that the underlying mechanisms of these two methods are very different. The RF classification algorithm achieves this AUC value by incorporating a much larger set of features, 1,456 features, into its classification process. In contrast, the microBiomeGSM tool achieves comparable AUC metric by using a leaner set of 596 features. This difference in feature usage is worth highlighting as it shows that the microBiomeGSM tool is able to deliver competitive results with a lower computational load, making it an efficient and resource-efficient choice for the classification task at hand. These results highlight the nuanced trade-offs in selecting the appropriate tool or algorithm for the specific data analysis requirements.</p>
<table-wrap position="float" id="tab6">
<label>Table 6</label>
<caption><p>Evaluation metrics obtained with microBiomeGSM on four datasets for different taxonomic levels, compared with traditional classifiers using all features.</p></caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="center" valign="top" colspan="7">CRC</th>
</tr>
<tr>
<th align="left" valign="top">Model</th>
<th align="center" valign="top"># of Species</th>
<th align="center" valign="top">Accuracy</th>
<th align="center" valign="top">Sensitivity</th>
<th align="center" valign="top">Specificity</th>
<th align="center" valign="top">Precision</th>
<th align="center" valign="top">AUC</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="bottom">AdaBoost</td>
<td align="char" valign="middle" char="&#x00B1;">912</td>
<td align="char" valign="middle" char="&#x00B1;">0.72 &#x00B1; 0.06</td>
<td align="char" valign="middle" char="&#x00B1;">0.79 &#x00B1; 0.09</td>
<td align="char" valign="middle" char="&#x00B1;">0.66 &#x00B1; 0.17</td>
<td align="char" valign="middle" char="&#x00B1;">0.7 &#x00B1; 0.09</td>
<td align="char" valign="middle" char="&#x00B1;">0.78 &#x00B1; 0.04</td>
</tr>
<tr>
<td align="left" valign="bottom">DT</td>
<td align="char" valign="middle" char="&#x00B1;">912</td>
<td align="char" valign="middle" char="&#x00B1;">0.68 &#x00B1; 0.09</td>
<td align="char" valign="middle" char="&#x00B1;">0.75 &#x00B1; 0.12</td>
<td align="char" valign="middle" char="&#x00B1;">0.62 &#x00B1; 0.26</td>
<td align="char" valign="middle" char="&#x00B1;">0.66 &#x00B1; 0.09</td>
<td align="char" valign="middle" char="&#x00B1;">0.7 &#x00B1; 0.04</td>
</tr>
<tr>
<td align="left" valign="bottom">LogitBoost</td>
<td align="char" valign="middle" char="&#x00B1;">912</td>
<td align="char" valign="middle" char="&#x00B1;">0.73 &#x00B1; 0.06</td>
<td align="char" valign="middle" char="&#x00B1;">0.78 &#x00B1; 0.09</td>
<td align="char" valign="middle" char="&#x00B1;">0.68 &#x00B1; 0.18</td>
<td align="char" valign="middle" char="&#x00B1;">0.71 &#x00B1; 0.09</td>
<td align="char" valign="middle" char="&#x00B1;">0.78 &#x00B1; 0.04</td>
</tr>
<tr>
<td align="left" valign="bottom">RF</td>
<td align="char" valign="middle" char="&#x00B1;">912</td>
<td align="char" valign="middle" char="&#x00B1;">0.78 &#x00B1; 0.05</td>
<td align="char" valign="middle" char="&#x00B1;">0.82 &#x00B1; 0.08</td>
<td align="char" valign="middle" char="&#x00B1;">0.75 &#x00B1; 0.14</td>
<td align="char" valign="middle" char="&#x00B1;">0.76 &#x00B1; 0.09</td>
<td align="char" valign="middle" char="&#x00B1;">0.86 &#x00B1; 0.03</td>
</tr>
<tr>
<td align="left" valign="bottom">microBiomeGSM: family</td>
<td align="char" valign="middle" char="&#x00B1;">292.88 &#x00B1; 16.09</td>
<td align="char" valign="middle" char="&#x00B1;">0.74 &#x00B1; 0.65</td>
<td align="char" valign="middle" char="&#x00B1;">0.7 &#x00B1; 0.39</td>
<td align="char" valign="middle" char="&#x00B1;">0.77 &#x00B1; 0.91</td>
<td align="char" valign="middle" char="&#x00B1;">0.75 &#x00B1; 0.83</td>
<td align="char" valign="middle" char="&#x00B1;">0.81 &#x00B1; 0.67</td>
</tr>
<tr>
<td align="left" valign="bottom">microBiomeGSM: genus</td>
<td align="char" valign="middle" char="&#x00B1;">161.21 &#x00B1; 5.17</td>
<td align="char" valign="middle" char="&#x00B1;">0.74 &#x00B1; 0.67</td>
<td align="char" valign="middle" char="&#x00B1;">0.69 &#x00B1; 0.41</td>
<td align="char" valign="middle" char="&#x00B1;">0.79 &#x00B1; 0.92</td>
<td align="char" valign="middle" char="&#x00B1;">0.76 &#x00B1; 0.84</td>
<td align="char" valign="middle" char="&#x00B1;">0.8 &#x00B1; 0.68</td>
</tr>
<tr>
<td align="left" valign="bottom">microBiomeGSM: order</td>
<td align="char" valign="middle" char="&#x00B1;">607.5 &#x00B1; 188.32</td>
<td align="char" valign="middle" char="&#x00B1;">0.73 &#x00B1; 0.69</td>
<td align="char" valign="middle" char="&#x00B1;">0.72 &#x00B1; 0.66</td>
<td align="char" valign="middle" char="&#x00B1;">0.75 &#x00B1; 0.73</td>
<td align="char" valign="middle" char="&#x00B1;">0.74 &#x00B1; 0.71</td>
<td align="char" valign="middle" char="&#x00B1;">0.81 &#x00B1; 0.77</td>
</tr>
</tbody>
</table>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="center" valign="middle" colspan="7">IBDMDB</th>
</tr>
<tr>
<th align="left" valign="bottom">Model</th>
<th align="center" valign="middle"># of Species</th>
<th align="center" valign="middle">Accuracy</th>
<th align="center" valign="middle">Sensitivity</th>
<th align="center" valign="middle">Specificity</th>
<th align="center" valign="middle">Precision</th>
<th align="center" valign="middle">AUC</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="bottom">AdaBoost</td>
<td align="char" valign="middle" char="&#x00B1;">579</td>
<td align="char" valign="middle" char="&#x00B1;">0.92 &#x00B1; 0.02</td>
<td align="char" valign="middle" char="&#x00B1;">0.97 &#x00B1; 0.02</td>
<td align="char" valign="middle" char="&#x00B1;">0.79 &#x00B1; 0.1</td>
<td align="char" valign="middle" char="&#x00B1;">0.93 &#x00B1; 0.03</td>
<td align="char" valign="middle" char="&#x00B1;">0.94 &#x00B1; 0.01</td>
</tr>
<tr>
<td align="left" valign="bottom">DT</td>
<td align="char" valign="middle" char="&#x00B1;">579</td>
<td align="char" valign="middle" char="&#x00B1;">0.91 &#x00B1; 0.02</td>
<td align="char" valign="middle" char="&#x00B1;">0.94 &#x00B1; 0.01</td>
<td align="char" valign="middle" char="&#x00B1;">0.84 &#x00B1; 0.05</td>
<td align="char" valign="middle" char="&#x00B1;">0.94 &#x00B1; 0.02</td>
<td align="char" valign="middle" char="&#x00B1;">0.89 &#x00B1; 0.02</td>
</tr>
<tr>
<td align="left" valign="bottom">LogitBoost</td>
<td align="char" valign="middle" char="&#x00B1;">579</td>
<td align="char" valign="middle" char="&#x00B1;">0.92 &#x00B1; 0.01</td>
<td align="char" valign="middle" char="&#x00B1;">0.98 &#x00B1; 0.01</td>
<td align="char" valign="middle" char="&#x00B1;">0.76 &#x00B1; 0.07</td>
<td align="char" valign="middle" char="&#x00B1;">0.92 &#x00B1; 0.02</td>
<td align="char" valign="middle" char="&#x00B1;">0.91 &#x00B1; 0.04</td>
</tr>
<tr>
<td align="left" valign="bottom">RF</td>
<td align="char" valign="middle" char="&#x00B1;">579</td>
<td align="char" valign="middle" char="&#x00B1;">0.98 &#x00B1; 0.01</td>
<td align="char" valign="middle" char="&#x00B1;">1 &#x00B1; 0</td>
<td align="char" valign="middle" char="&#x00B1;">0.93 &#x00B1; 0.06</td>
<td align="char" valign="middle" char="&#x00B1;">0.98 &#x00B1; 0.02</td>
<td align="char" valign="middle" char="&#x00B1;">0.98 &#x00B1; 0.01</td>
</tr>
<tr>
<td align="left" valign="bottom">microBiomeGSM: Family</td>
<td align="char" valign="middle" char="&#x00B1;">205.76 &#x00B1; 16.23</td>
<td align="char" valign="middle" char="&#x00B1;">0.94 &#x00B1; 0.02</td>
<td align="char" valign="middle" char="&#x00B1;">0.98 &#x00B1; 0.01</td>
<td align="char" valign="middle" char="&#x00B1;">0.86 &#x00B1; 0.05</td>
<td align="char" valign="middle" char="&#x00B1;">0.93 &#x00B1; 0.05</td>
<td align="char" valign="middle" char="&#x00B1;">0.97 &#x00B1; 0.02</td>
</tr>
<tr>
<td align="left" valign="bottom">microBiomeGSM: Genus</td>
<td align="char" valign="middle" char="&#x00B1;">119.4 &#x00B1; 15.87</td>
<td align="char" valign="middle" char="&#x00B1;">0.93 &#x00B1; 0.02</td>
<td align="char" valign="middle" char="&#x00B1;">0.98 &#x00B1; 0.01</td>
<td align="char" valign="middle" char="&#x00B1;">0.85 &#x00B1; 0.05</td>
<td align="char" valign="middle" char="&#x00B1;">0.93 &#x00B1; 0.05</td>
<td align="char" valign="middle" char="&#x00B1;">0.97 &#x00B1; 0.02</td>
</tr>
<tr>
<td align="left" valign="bottom">microBiomeGSM: Order</td>
<td align="char" valign="middle" char="&#x00B1;">341.22 &#x00B1; 15.6</td>
<td align="char" valign="middle" char="&#x00B1;">0.93 &#x00B1; 0.02</td>
<td align="char" valign="middle" char="&#x00B1;">0.97 &#x00B1; 0.02</td>
<td align="char" valign="middle" char="&#x00B1;">0.86 &#x00B1; 0.06</td>
<td align="char" valign="middle" char="&#x00B1;">0.93 &#x00B1; 0.06</td>
<td align="char" valign="middle" char="&#x00B1;">0.98 &#x00B1; 0.03</td>
</tr>
</tbody>
</table>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="center" valign="middle" colspan="7">IBD</th>
</tr>
<tr>
<th align="left" valign="bottom">Model</th>
<th align="center" valign="middle"># of Species</th>
<th align="center" valign="middle">Accuracy</th>
<th align="center" valign="middle">Sensitivity</th>
<th align="center" valign="middle">Specificity</th>
<th align="center" valign="middle">Precision</th>
<th align="center" valign="middle">AUC</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="bottom">AdaBoost</td>
<td align="char" valign="middle" char="&#x00B1;">1,456</td>
<td align="char" valign="middle" char="&#x00B1;">0.88 &#x00B1; 0.04</td>
<td align="char" valign="middle" char="&#x00B1;">0.85 &#x00B1; 0.12</td>
<td align="char" valign="middle" char="&#x00B1;">0.89 &#x00B1; 0.05</td>
<td align="char" valign="middle" char="&#x00B1;">0.84 &#x00B1; 0.05</td>
<td align="char" valign="middle" char="&#x00B1;">0.9 &#x00B1; 0.04</td>
</tr>
<tr>
<td align="left" valign="bottom">DT</td>
<td align="char" valign="middle" char="&#x00B1;">1,456</td>
<td align="char" valign="middle" char="&#x00B1;">0.75 &#x00B1; 0.05</td>
<td align="char" valign="middle" char="&#x00B1;">0.72 &#x00B1; 0.09</td>
<td align="char" valign="middle" char="&#x00B1;">0.78 &#x00B1; 0.06</td>
<td align="char" valign="middle" char="&#x00B1;">0.67 &#x00B1; 0.08</td>
<td align="char" valign="middle" char="&#x00B1;">0.75 &#x00B1; 0.06</td>
</tr>
<tr>
<td align="left" valign="bottom">LogitBoost</td>
<td align="char" valign="middle" char="&#x00B1;">1,456</td>
<td align="char" valign="middle" char="&#x00B1;">0.85 &#x00B1; 0.04</td>
<td align="char" valign="middle" char="&#x00B1;">0.81 &#x00B1; 0.1</td>
<td align="char" valign="middle" char="&#x00B1;">0.87 &#x00B1; 0.07</td>
<td align="char" valign="middle" char="&#x00B1;">0.8 &#x00B1; 0.09</td>
<td align="char" valign="middle" char="&#x00B1;">0.88 &#x00B1; 0.04</td>
</tr>
<tr>
<td align="left" valign="bottom">RF</td>
<td align="char" valign="middle" char="&#x00B1;">1,456</td>
<td align="char" valign="middle" char="&#x00B1;">0.87 &#x00B1; 0.05</td>
<td align="char" valign="middle" char="&#x00B1;">0.91 &#x00B1; 0.1</td>
<td align="char" valign="middle" char="&#x00B1;">0.84 &#x00B1; 0.05</td>
<td align="char" valign="middle" char="&#x00B1;">0.78 &#x00B1; 0.06</td>
<td align="char" valign="middle" char="&#x00B1;">0.92 &#x00B1; 0.05</td>
</tr>
<tr>
<td align="left" valign="bottom">microBiomeGSM: Family</td>
<td align="char" valign="middle" char="&#x00B1;">260.24 &#x00B1; 26.92</td>
<td align="char" valign="middle" char="&#x00B1;">0.82 &#x00B1; 0.06</td>
<td align="char" valign="middle" char="&#x00B1;">0.85 &#x00B1; 0.07</td>
<td align="char" valign="middle" char="&#x00B1;">0.79 &#x00B1; 0.1</td>
<td align="char" valign="middle" char="&#x00B1;">0.81 &#x00B1; 0.13</td>
<td align="char" valign="middle" char="&#x00B1;">0.91 &#x00B1; 0.07</td>
</tr>
<tr>
<td align="left" valign="bottom">microBiomeGSM: Genus</td>
<td align="char" valign="middle" char="&#x00B1;">121.78 &#x00B1; 27.83</td>
<td align="char" valign="middle" char="&#x00B1;">0.81 &#x00B1; 0.06</td>
<td align="char" valign="middle" char="&#x00B1;">0.83 &#x00B1; 0.06</td>
<td align="char" valign="middle" char="&#x00B1;">0.79 &#x00B1; 0.1</td>
<td align="char" valign="middle" char="&#x00B1;">0.8 &#x00B1; 0.12</td>
<td align="char" valign="middle" char="&#x00B1;">0.88 &#x00B1; 0.08</td>
</tr>
<tr>
<td align="left" valign="bottom">microBiomeGSM: Order</td>
<td align="char" valign="middle" char="&#x00B1;">608.27 &#x00B1; 24.22</td>
<td align="char" valign="middle" char="&#x00B1;">0.82 &#x00B1; 0.07</td>
<td align="char" valign="middle" char="&#x00B1;">0.82 &#x00B1; 0.08</td>
<td align="char" valign="middle" char="&#x00B1;">0.81 &#x00B1; 0.09</td>
<td align="char" valign="middle" char="&#x00B1;">0.82 &#x00B1; 0.15</td>
<td align="char" valign="middle" char="&#x00B1;">0.9 &#x00B1; 0.08</td>
</tr>
</tbody>
</table>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="center" valign="middle" colspan="7">T2D</th>
</tr>
<tr>
<th align="left" valign="bottom">Model</th>
<th align="center" valign="middle"># of Species</th>
<th align="center" valign="middle">Accuracy</th>
<th align="center" valign="middle">Sensitivity</th>
<th align="center" valign="middle">Specificity</th>
<th align="center" valign="middle">Precision</th>
<th align="center" valign="middle">AUC</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="bottom">AdaBoost</td>
<td align="char" valign="middle" char="&#x00B1;">1,456</td>
<td align="char" valign="middle" char="&#x00B1;">0.68 &#x00B1; 0.08</td>
<td align="char" valign="middle" char="&#x00B1;">0.91 &#x00B1; 0.08</td>
<td align="char" valign="middle" char="&#x00B1;">0.39 &#x00B1; 0.26</td>
<td align="char" valign="middle" char="&#x00B1;">0.67 &#x00B1; 0.09</td>
<td align="char" valign="middle" char="&#x00B1;">0.66 &#x00B1; 0.1</td>
</tr>
<tr>
<td align="left" valign="bottom">DT</td>
<td align="char" valign="middle" char="&#x00B1;">1,456</td>
<td align="char" valign="middle" char="&#x00B1;">0.57 &#x00B1; 0.05</td>
<td align="char" valign="middle" char="&#x00B1;">0.98 &#x00B1; 0.06</td>
<td align="char" valign="middle" char="&#x00B1;">0.06 &#x00B1; 0.19</td>
<td align="char" valign="middle" char="&#x00B1;">0.57 &#x00B1; 0.06</td>
<td align="char" valign="middle" char="&#x00B1;">0.57 &#x00B1; 0.09</td>
</tr>
<tr>
<td align="left" valign="bottom">LogitBoost</td>
<td align="char" valign="middle" char="&#x00B1;">1,456</td>
<td align="char" valign="middle" char="&#x00B1;">0.67 &#x00B1; 0.08</td>
<td align="char" valign="middle" char="&#x00B1;">0.93 &#x00B1; 0.08</td>
<td align="char" valign="middle" char="&#x00B1;">0.36 &#x00B1; 0.24</td>
<td align="char" valign="middle" char="&#x00B1;">0.65 &#x00B1; 0.08</td>
<td align="char" valign="middle" char="&#x00B1;">0.65 &#x00B1; 0.1</td>
</tr>
<tr>
<td align="left" valign="bottom">RF</td>
<td align="char" valign="middle" char="&#x00B1;">1,456</td>
<td align="char" valign="middle" char="&#x00B1;">0.72 &#x00B1; 0.09</td>
<td align="char" valign="middle" char="&#x00B1;">0.91 &#x00B1; 0.09</td>
<td align="char" valign="middle" char="&#x00B1;">0.48 &#x00B1; 0.29</td>
<td align="char" valign="middle" char="&#x00B1;">0.71 &#x00B1; 0.12</td>
<td align="char" valign="middle" char="&#x00B1;">0.75 &#x00B1; 0.1</td>
</tr>
<tr>
<td align="left" valign="bottom">microBiomeGSM: family</td>
<td align="char" valign="middle" char="&#x00B1;">321.16 &#x00B1; 36.31</td>
<td align="char" valign="middle" char="&#x00B1;">0.65 &#x00B1; 0.08</td>
<td align="char" valign="middle" char="&#x00B1;">0.68 &#x00B1; 0.09</td>
<td align="char" valign="middle" char="&#x00B1;">0.63 &#x00B1; 0.11</td>
<td align="char" valign="middle" char="&#x00B1;">0.65 &#x00B1; 0.15</td>
<td align="char" valign="middle" char="&#x00B1;">0.71 &#x00B1; 0.08</td>
</tr>
<tr>
<td align="left" valign="bottom">microBiomeGSM: genus</td>
<td align="char" valign="middle" char="&#x00B1;">129.8 &#x00B1; 35.03</td>
<td align="char" valign="middle" char="&#x00B1;">0.64 &#x00B1; 0.09</td>
<td align="char" valign="middle" char="&#x00B1;">0.64 &#x00B1; 0.1</td>
<td align="char" valign="middle" char="&#x00B1;">0.64 &#x00B1; 0.13</td>
<td align="char" valign="middle" char="&#x00B1;">0.65 &#x00B1; 0.18</td>
<td align="char" valign="middle" char="&#x00B1;">0.69 &#x00B1; 0.09</td>
</tr>
<tr>
<td align="left" valign="bottom">microBiomeGSM: order</td>
<td align="char" valign="middle" char="&#x00B1;">596.99 &#x00B1; 35.14</td>
<td align="char" valign="middle" char="&#x00B1;">0.65 &#x00B1; 0.08</td>
<td align="char" valign="middle" char="&#x00B1;">0.65 &#x00B1; 0.09</td>
<td align="char" valign="middle" char="&#x00B1;">0.64 &#x00B1; 0.12</td>
<td align="char" valign="middle" char="&#x00B1;">0.65 &#x00B1; 0.17</td>
<td align="char" valign="middle" char="&#x00B1;">0.72 &#x00B1; 0.09</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>As shown in <xref ref-type="table" rid="tab7">Table 7</xref>, the performance of our proposed method varies depending on the taxonomic level considered. For the order taxonomic level, for all tested datasets, the proposed method outperforms other models in terms of the AUC score, except for the RF classifier. Similarly, for all datasets, at the family and genus taxonomic levels, the AUC values are also highly competitive, outperforming those of the other four machine learning algorithms used in this study, with the sole exception of the RF classifier. These results highlight the robust performance of our method across different taxonomic levels. A remarkable performance of our proposed method was observed when it is applied on the IBDMDB dataset. Here, we obtained an exceptionally high AUC value of 0.98&#x2009;&#x00B1;&#x2009;0.03 at the order taxonomic level using a 100-fold MCCV approach. This remarkable result demonstrates the exceptional performance and the potential of the microBiomeGSM tool.</p>
<table-wrap position="float" id="tab7">
<label>Table 7</label>
<caption><p>Comparative performance evaluation of microBiomeGSM and other machine learning approaches for different microbiome datasets.</p></caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="middle">Dataset</th>
<th/>
<th align="center" valign="middle">AdaBoost</th>
<th align="center" valign="middle">DT</th>
<th align="center" valign="middle">LogitBoost</th>
<th align="center" valign="middle">RF</th>
<th align="center" valign="middle">microBiomeGSM: family</th>
<th align="center" valign="middle">microBiomeGSM: genus</th>
<th align="center" valign="middle">microBiomeGSM: order</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle" rowspan="2">CRC</td>
<td align="char" valign="middle" char="&#x00B1;">AUC</td>
<td align="char" valign="middle" char="&#x00B1;">0.78 &#x00B1; 0.14</td>
<td align="char" valign="middle" char="&#x00B1;">0.70 &#x00B1; 0.04</td>
<td align="char" valign="middle" char="&#x00B1;">0.78 &#x00B1; 0.04</td>
<td align="char" valign="middle" char="&#x00B1;"><bold>0.86</bold> &#x00B1; 0.03</td>
<td align="char" valign="middle" char="&#x00B1;">0.81 &#x00B1; 0.67</td>
<td align="char" valign="middle" char="&#x00B1;">0.80 &#x00B1; 0.68</td>
<td align="char" valign="middle" char="&#x00B1;"><bold>0.81</bold> &#x00B1; 0.77</td>
</tr>
<tr>
<td align="char" valign="middle" char="&#x00B1;"># of Species</td>
<td align="char" valign="middle" char="&#x00B1;">912</td>
<td align="char" valign="middle" char="&#x00B1;">912</td>
<td align="char" valign="middle" char="&#x00B1;">912</td>
<td align="char" valign="middle" char="&#x00B1;">912</td>
<td align="char" valign="middle" char="&#x00B1;">292.88 &#x00B1; 16.09</td>
<td align="char" valign="middle" char="&#x00B1;">161.21 &#x00B1; 5.17</td>
<td align="char" valign="middle" char="&#x00B1;">607.5 &#x00B1; 188.32</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="2">IBDMDB</td>
<td align="char" valign="middle" char="&#x00B1;">AUC</td>
<td align="char" valign="middle" char="&#x00B1;">0.94 &#x00B1; 0.01</td>
<td align="char" valign="middle" char="&#x00B1;">0.89 &#x00B1; 0.02</td>
<td align="char" valign="middle" char="&#x00B1;">0.91 &#x00B1; 0.04</td>
<td align="char" valign="middle" char="&#x00B1;"><bold>0.98</bold> &#x00B1; 0.01</td>
<td align="char" valign="middle" char="&#x00B1;">0.97 &#x00B1; 0.02</td>
<td align="char" valign="middle" char="&#x00B1;">0.97 &#x00B1; 0.03</td>
<td align="char" valign="middle" char="&#x00B1;"><bold>0.98 &#x00B1; 0.03</bold></td>
</tr>
<tr>
<td align="char" valign="middle" char="&#x00B1;"># of Species</td>
<td align="char" valign="middle" char="&#x00B1;">579</td>
<td align="char" valign="middle" char="&#x00B1;">579</td>
<td align="char" valign="middle" char="&#x00B1;">579</td>
<td align="char" valign="middle" char="&#x00B1;">579</td>
<td align="char" valign="middle" char="&#x00B1;">205.76 &#x00B1; 16.23</td>
<td align="char" valign="middle" char="&#x00B1;">119.4 &#x00B1; 15.87</td>
<td align="char" valign="middle" char="&#x00B1;">341.22 &#x00B1; 15.6</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="2">IBD</td>
<td align="char" valign="middle" char="&#x00B1;">AUC</td>
<td align="char" valign="middle" char="&#x00B1;">0.9 &#x00B1; 0.04</td>
<td align="char" valign="middle" char="&#x00B1;">0.75 &#x00B1; 0.06</td>
<td align="char" valign="middle" char="&#x00B1;">0.88 &#x00B1; 0.04</td>
<td align="char" valign="middle" char="&#x00B1;"><bold>0.92</bold> &#x00B1; 0.05</td>
<td align="char" valign="middle" char="&#x00B1;"><bold>0.91</bold> &#x00B1; 0.07</td>
<td align="char" valign="middle" char="&#x00B1;">0.88 &#x00B1; 0.08</td>
<td align="char" valign="middle" char="&#x00B1;">0.9 &#x00B1; 0.08</td>
</tr>
<tr>
<td align="char" valign="middle" char="&#x00B1;"># of Species</td>
<td align="char" valign="middle" char="&#x00B1;">1,456</td>
<td align="char" valign="middle" char="&#x00B1;">1,456</td>
<td align="char" valign="middle" char="&#x00B1;">1,456</td>
<td align="char" valign="middle" char="&#x00B1;">1,456</td>
<td align="char" valign="middle" char="&#x00B1;">260.24 &#x00B1; 26.92</td>
<td align="char" valign="middle" char="&#x00B1;">121.78 &#x00B1; 27.83</td>
<td align="char" valign="middle" char="&#x00B1;">608.27 &#x00B1; 24.22</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="2">T2D</td>
<td align="char" valign="middle" char="&#x00B1;">AUC</td>
<td align="char" valign="middle" char="&#x00B1;">0.66 &#x00B1; 0.1</td>
<td align="char" valign="middle" char="&#x00B1;">0.57 &#x00B1; 0.09</td>
<td align="char" valign="middle" char="&#x00B1;">0.65 &#x00B1; 0.1</td>
<td align="char" valign="middle" char="&#x00B1;"><bold>0.75</bold> &#x00B1; 0.1</td>
<td align="char" valign="middle" char="&#x00B1;">0.71 &#x00B1; 0.08</td>
<td align="char" valign="middle" char="&#x00B1;">0.69 &#x00B1; 0.09</td>
<td align="char" valign="middle" char="&#x00B1;"><bold>0.72</bold> &#x00B1; 0.09</td>
</tr>
<tr>
<td align="char" valign="middle" char="&#x00B1;"># of Species</td>
<td align="char" valign="middle" char="&#x00B1;">1,456</td>
<td align="char" valign="middle" char="&#x00B1;">1,456</td>
<td align="char" valign="middle" char="&#x00B1;">1,456</td>
<td align="char" valign="middle" char="&#x00B1;">1,456</td>
<td align="char" valign="middle" char="&#x00B1;">321.16 &#x00B1; 36.31</td>
<td align="char" valign="middle" char="&#x00B1;">129.8 &#x00B1; 35.03</td>
<td align="char" valign="middle" char="&#x00B1;">596.99 &#x00B1; 35.14</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>The results in bold in table represent the best AUC result among the 4 different traditional classifiers (AdaBoost, DT, LogitBoost and RF) and the best AUC result across taxon levels using microBiomeGSM.</p>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec sec-type="discussion" id="sec11">
<label>4</label>
<title>Discussion</title>
<p>The microbiome is considered as a crucial component of the human body and it is increasingly associated with numerous aspects of development and health. There is growing evidence that the microbiota is essential for understanding, diagnosing, and treating human diseases. In particular, alterations in the gut microbiome community have been linked to a variety of diseases, including CRC (<xref ref-type="bibr" rid="ref58">Song et al., 2020</xref>), T2D (<xref ref-type="bibr" rid="ref54">Salamon et al., 2018</xref>) and IBD (<xref ref-type="bibr" rid="ref1">Alam et al., 2020</xref>). Several research efforts relied on sample-level feature abundance data to identify predictive microbiome biomarkers using machine learning. In this study, we proposed to perform more effective disease classification and prediction with fewer features. To this end, we developed microBiomeGSM to solve this problem compared to tools that perform predictions with a large amount of data. The success of microBiomeGSM can be explained with the following features of the G-S-M approach:</p>
<list list-type="bullet">
<list-item><p>For the grouping component of microBiomeGSM, only the features at the similar taxonomic levels are considered.</p></list-item>
<list-item><p>microBiomeGSM uses efficient classifiers for the scoring component to identify the key groups for each taxonomic level;</p></list-item>
<list-item><p>For the modeling component, significant taxonomic groups are considered cumulatively using effective classifiers.</p></list-item>
</list>
<p>Via analyzing metagenomic data, this study aims to solve the problem of disease diagnosis using existing taxonomic knowledge; and finally introduces a tool called microBiomeGSM. The proposed tool is based on the G-S-M (Grouping-Scoring-Modeling) approach and uses species-level information by grouping taxonomic features at different taxonomic levels such as genus, family, and order. The performance of microBiomeGSM on four different disease-associated metagenomic datasets was evaluated in comparison to other feature selection methods such as Fast Correlation Based Filter (FCBF), Select Best K (SKB), Extreme Gradient Boosting (XGB), Conditional Mutual Information Maximization (CMIM), Maximum Likelihood and Minimum Redundancy (MRMR), and Information Gain (IG).</p>
<p>The presented microBiomeGSM approach offers several advantages in the field of disease diagnosis via analyzing metagenomic datasets. One significant benefit is its ability to efficiently identify disease-associated taxonomic biomarkers through a robust machine learning model based on the Grouping, Scoring, and Modeling (G-S-M) methodology. Differently from existing approaches, microBiomeGSM identifies groups of important taxons and detects important species within that taxon for the disease under study. Hence, this innovative approach enables the extraction of valuable insights from microbiome data, shedding light on the influence of specific taxonomic biomarkers on the disease under investigation. Furthermore, the performance evaluation across different diseases, different taxonomic levels (genus, family, order); and the comparative assessment with different feature selection algorithms exhibits the reliability of microBiomeGSM. Finally, the discussions on the biological relevance of the findings of the proposed approach, via drawing evidence from the existing literature, provide valuable context for the identified taxon groups for the disease under study, making microBiomeGSM an informative tool in disease research. Our tool&#x2019;s significance transcends its mere application; it holds the potential for pioneering discoveries. It is geared to discern not isolated microbial entities but entire assemblages of species, paving the way for profound biological interpretations. By spotlighting groups of bacteria and viruses in lieu of singular entities, our tool offers a holistic view, potentially identifying microbial communities implicated in specific diseases.</p>
<p>With this study, we would also like to motivate biologists and the microbiome community to redesign their grouping methods instead of using individual feature selection approaches. We envision that in the future, various biological datasets, including multi-omics, will be used to redefine the groupings. Such innovative grouping strategies, complemented by modeling, promise to provide profound insights into the molecular mechanisms of diseases and the role of microorganisms in disease development.</p>
<sec id="sec12">
<label>4.1</label>
<title>Biological interpretations of microBiomeGSM&#x2019;s findings</title>
<p>This section discusses the biological relevance of the features discovered by microBiomeGSM at different taxonomic levels for all tested datasets. T2D is a metabolic disease characterized by high glucose levels in blood and caused primarily by cellular resistance to the activity of insulin (<xref ref-type="bibr" rid="ref55">Sedighi et al., 2017</xref>). There are several studies in the literature that have demonstrated the relation of different microorganisms at the genus, family, and order levels with T2D development. For the T2D dataset, the top 10 microbiomes identified by our method at the genus, family, order levels and the relevant literature can be summarized in <xref rid="SM1" ref-type="supplementary-material">Supplementary Table S15</xref>. On the other hand, inflammatory bowel diseases (IBDs), which include primarily ulcerative colitis and Crohn&#x2019;s disease, but also non-infectious inflammation of the bowel, have puzzled gastroenterologists and immunologists alike since their first modern descriptions around some 75&#x2013;100&#x2009;years ago (<xref ref-type="bibr" rid="ref43">Ni et al., 2018</xref>; <xref ref-type="bibr" rid="ref5">Bakir-Gungor et al., 2022</xref>). For the IBDMDB dataset, the top 10 microbiomes identified by our method at the genus, family, and order levels and the relevant literature can be summarized in <xref rid="SM1" ref-type="supplementary-material">Supplementary Table S15</xref>. CRC is a prevalent malignancy affecting the colon and rectum. It constitutes approximately 10% of all newly diagnosed cancer cases worldwide (<xref ref-type="bibr" rid="ref31">Li X. et al., 2023</xref>). For the CRC dataset, the top 10 microbiomes identified by our method at the genus, family, and order levels and the relevant literature can be summarized in <xref rid="SM1" ref-type="supplementary-material">Supplementary Table S15</xref>.</p>
<p>Numerous studies have investigated the relationship between microbiomes and diseases like T2D, CRC, and IBD using similar datasets as used within this study. Upon examination of these studies, it becomes evident that while their experimental designs may vary, they consistently yield comparable results when it comes to identifying microbiomes linked to these diseases. These findings align with the important microbiomes identified by microBiomeGSM for T2D, CRC, and IBD, showcasing the tool&#x2019;s effectiveness in accurately identifying relevant microbiomes associated with these diseases. These congruent findings reinforce the reliability and validity of the microbiome associations detected by the microBiomeGSM tool. It also underscores the tool&#x2019;s capacity to identify microbiomes that are consistently linked to specific diseases, providing valuable insights for disease characterization and prediction. <xref ref-type="bibr" rid="ref22">Hassouneh et al. (2021)</xref> conducted a series of experiments aimed at uncovering microbiomes associated with IBD. In their analysis using the same dataset as used by the microBiomeGSM tool, they observed differences in Clostridium microbiota among IBD patients. Additionally, another microbiome identified for IBD in their study is <italic>Ruminococcus</italic>. Remarkably, these microbiomes align with the important microbiomes detected for the IBD disease by the microBiomeGSM tool. This correspondence in findings highlights the capacity of microBiomeGSM in identifying relevant microbiomes linked to IBD. <xref ref-type="bibr" rid="ref74">Zhang Y. et al. (2022)</xref> conducted a study with the goal of identifying disease-associated microbiome species for Inflammatory Bowel Disease Microbiome Database (IBDMDB), employing the same dataset (PRJNA289734) as used in microBiomeGSM. In their research, they highlighted the significance of the Bacteroides microbiome. Interestingly, the Bacteroides microbiome is also identified as one of the important microbiomes by the microBiomeGSM tool proposed in our study. This alignment in findings underscores the effectiveness of microBiomeGSM in recognizing key microbiomes associated with diseases like IBD. <xref ref-type="bibr" rid="ref3">Bai et al. (2022)</xref> conducted a series of experiments aimed at identifying microbiomes associated with T2D. In their research, they utilized the SRA4565 data for T2D and highlighted the significance of the methanobacteriales microbiome. Notably, methanobacteriales is among the top 10 microbiomes identified by the proposed microBiomeGSM tool. This convergence of findings underscores the effectiveness and utility of the proposed tool in uncovering microbiome associations with diseases like T2D. <xref ref-type="bibr" rid="ref17">Forslund et al. (2015)</xref> conducted experiments utilizing the same T2D dataset employed by microBiomeGSM to investigate microbiomes associated with T2D. Upon close examination of their experiments, they underscored the significance of the Clostridiales microbiome in relation to T2D disease. Interestingly, Clostridiales also emerges as one of the important microbiomes identified by microBiomeGSM. This convergence in findings highlights the relevance and effectiveness of microBiomeGSM in identifying crucial microbiomes associated with T2D. <xref ref-type="bibr" rid="ref34">Ma et al. (2021)</xref> conducted a study that investigated the microbiomes associated with CRC using the same dataset as in our study. Among the various microbiomes they examined, the Prevotella microbiome stood out as strongly linked to CRC. This association aligns with the findings of microBiomeGSM, underscoring the significance of the Prevotella microbiome in the context of characterizing CRC. <xref ref-type="bibr" rid="ref9">Chen et al. (2023)</xref> conducted research using the same dataset to investigate microbiomes in the context of colorectal cancer, akin to the proposed microBiomeGSM tool. Similar to the findings of microBiomeGSM, their study also identified Peptostreptococcus, Fusobacterium, and Porphyromonas microbiomes as valuable and effective biomarkers for CRC. This convergence in results underscores the potential significance of these specific microbiomes in CRC characterization and their importance as potential biomarkers for the disease.</p>
<p>In summary, via analyzing the raw microbiome data of specific diseases, this study aims to identify taxonomic biomarkers that may have a role in the associated diseases. Three different taxon levels (genus, family, and order) are studied and disease prediction is performed by building effective machine learning models using the G-S-M approach. Four different datasets are analyzed and the identified microorganisms at genus, family and order levels are compared with the existing literature.</p>
</sec>
<sec id="sec13">
<label>4.2</label>
<title>Limitation of the study</title>
<p>The quality and the scope of our study have been significantly influenced by several primary limiting factors. These factors encompass the nature of the data set, the tools employed for data preprocessing, the specific taxon groups considered, and the overall volume of data under examination. First and foremost, the data set itself plays a pivotal role in shaping the outcomes and conclusions of our study. Its size, diversity, and representativeness directly impact the generalizability of our findings. Furthermore, the quality of data, its sources, and any potential biases within the dataset significantly affect the reliability of our results. Equally significant is the role of the tools employed for data preprocessing. The choices made in data cleaning, feature selection, and data transformation can introduce variability and influence the robustness of our analytical pipeline. It is paramount to acknowledge how these preprocessing steps can shape the study&#x2019;s outcomes. Additionally, our study&#x2019;s focus on specific taxon groups within the dataset should be considered. The selection of these taxonomic levels and the criteria used for their inclusion or exclusion has bearing on the granularity and relevance of our findings. Finally, the number of data points utilized in our analysis is another crucial factor. A larger dataset provides a broader and potentially more representative sample, which can enhance the reliability and statistical power of our results. Conversely, a smaller dataset may limit the generalizability of our conclusions. A comprehensive understanding of these limiting factors is essential for contextualizing our study&#x2019;s outcomes and conclusions.</p>
</sec>
</sec>
<sec sec-type="conclusions" id="sec14">
<label>5</label>
<title>Conclusion</title>
<p>Over the past two decades, the number of microbiome studies has increased rapidly thanks to the advances in next generation sequencing (NGS) technologies. Lower costs and increasing computational power have enabled us to obtain enormous amounts of data on the diversity and function of a host or habitat&#x2019;s microbiome. Identifying and accounting for effective taxons in microbiome and disease classification can accelerate disease diagnosis, prognosis, and treatment. Here, we use an efficient machine learning model to identify taxonomic biomarkers that can diagnose diseases. The microBiomeGSM enables researchers to explore the diversity of contributions to disease development by examining metagenomic data at different taxonomic levels. While analyzing microbiome datasets, the microBiomeGSM tool that we present in this study exploits the existing biological knowledge about the taxonomic hierarchy of the species at different levels, such as genus, family, and order. Our results showed that via analyzing different microbiome datasets associated with different diseases, microBiomeGSM builds effective machine learning models to facilitate the diagnosis of diseases. It is anticipated that this study will be a guide for future studies and will guide and improve the studies to be conducted on this topic. With this study, we hope to highlight the importance of taxonomic groups in microbiome-based disease prediction and to facilitate the diagnosis of disease using these taxonomic groups.</p>
</sec>
<sec sec-type="data-availability" id="sec15">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="sec19">Supplementary material</xref>, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec sec-type="author-contributions" id="sec16">
<title>Author contributions</title>
<p>BB-G: Methodology, Software, Writing &#x2013; review &#x0026; editing, Project administration, Supervision. MT: Methodology, Software, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing, Investigation, Visualization. AJ: Data curation, Formal analysis, Methodology, Software, Writing &#x2013; original draft. DW: Investigation, Methodology, Supervision, Writing &#x2013; original draft, Project administration, Writing &#x2013; review &#x0026; editing. MY: Formal analysis, Funding acquisition, Methodology, Project administration, Software, Supervision, Writing &#x2013; review &#x0026; editing.</p>
</sec>
</body>
<back>
<sec sec-type="funding-information" id="sec17">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. The work of BB-G has been supported by the L&#x2019;Or&#x00E9;al-UNESCO Young Women Scientist Program and by the Abdullah Gul University Support Foundation (AGUV). The work of MY has been supported by the Zefat Academic College. This article is based upon work from COST Action ML4Microbiome (CA18131), supported by COST (European Cooperation in Science and Technology), <ext-link xlink:href="https://www.cost.eu/" ext-link-type="uri">www.cost.eu</ext-link>, which has played a pivotal role in advancing microbiome research and facilitating the expansion of these research endeavours.</p>
</sec>
<ack>
<p>We extend our gratitude to COST ML4Microbiome Action for the funding, which has played a pivotal role in advancing microbiome research and facilitating the expansion of these research endeavors. This research was made possible by the generous support of the L&#x2019;Or&#x00E9;al-UNESCO Young Women Scientist Program. BB-G would like to express her gratitude for the L&#x2019;Or&#x00E9;al-UNESCO Young Women Scientist Award, received in 2022.</p>
</ack>
<sec sec-type="COI-statement" id="sec18">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="sec100" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="sec19">
<title>Supplementary material</title>
<p>The Supplementary material for this article can be found online at: <ext-link xlink:href="https://www.frontiersin.org/articles/10.3389/fmicb.2023.1264941/full#supplementary-material" ext-link-type="uri">https://www.frontiersin.org/articles/10.3389/fmicb.2023.1264941/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alam</surname> <given-names>M. T.</given-names></name> <name><surname>Amos</surname> <given-names>G. C. A.</given-names></name> <name><surname>Murphy</surname> <given-names>A. R. J.</given-names></name> <name><surname>Murch</surname> <given-names>S.</given-names></name> <name><surname>Wellington</surname> <given-names>E. M. H.</given-names></name> <name><surname>Arasaradnam</surname> <given-names>R. P.</given-names></name></person-group> (<year>2020</year>). <article-title>Microbial imbalance in inflammatory bowel disease patients at different taxonomic levels</article-title>. <source>Gut Pathog.</source> <volume>12</volume>:<fpage>1</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s13099-019-0341-6</pub-id>, PMID: <pub-id pub-id-type="pmid">31911822</pub-id></citation></ref>
<ref id="ref2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alatawi</surname> <given-names>H.</given-names></name> <name><surname>Mosli</surname> <given-names>M.</given-names></name> <name><surname>Saadah</surname> <given-names>O. I.</given-names></name> <name><surname>Annese</surname> <given-names>V.</given-names></name> <name><surname>al-Hindi</surname> <given-names>R.</given-names></name> <name><surname>Alatawy</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Attributes of intestinal microbiota composition and their correlation with clinical primary non-response to anti-TNF-&#x03B1; agents in inflammatory bowel disease patients</article-title>. <source>Biomol. Biomed.</source> <volume>22</volume>, <fpage>412</fpage>&#x2013;<lpage>426</lpage>. doi: <pub-id pub-id-type="doi">10.17305/bjbms.2021.6436</pub-id>, PMID: <pub-id pub-id-type="pmid">34761733</pub-id></citation></ref>
<ref id="ref3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bai</surname> <given-names>X.</given-names></name> <name><surname>Sun</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>M.</given-names></name> <name><surname>Cao</surname> <given-names>Z.</given-names></name> <name><surname>Huang</surname> <given-names>Z.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Landscape of the gut archaeome in association with geography, ethnicity, urbanization, and diet in the Chinese population</article-title>. <source>Microbiome</source> <volume>10</volume>:<fpage>147</fpage>. <comment>Available at:</comment>. doi: <pub-id pub-id-type="doi">10.1186/s40168-022-01335-7</pub-id>, PMID: <pub-id pub-id-type="pmid">36100953</pub-id></citation></ref>
<ref id="ref4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bakir-Gungor</surname> <given-names>B.</given-names></name> <name><surname>Bulut</surname> <given-names>O.</given-names></name> <name><surname>Jabeer</surname> <given-names>A.</given-names></name> <name><surname>Nalbantoglu</surname> <given-names>O. U.</given-names></name> <name><surname>Yousef</surname> <given-names>M.</given-names></name></person-group> (<year>2021</year>). <article-title>Discovering potential taxonomic biomarkers of type 2 diabetes from human gut microbiota via different feature selection methods</article-title>. <source>Front. Microbiol.</source> <volume>12</volume>:<fpage>426</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fmicb.2021.628426</pub-id>, PMID: <pub-id pub-id-type="pmid">34512559</pub-id></citation></ref>
<ref id="ref5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bakir-Gungor</surname> <given-names>B.</given-names></name> <name><surname>Hac&#x0131;lar</surname> <given-names>H.</given-names></name> <name><surname>Jabeer</surname> <given-names>A.</given-names></name> <name><surname>Nalbantoglu</surname> <given-names>O. U.</given-names></name> <name><surname>Aran</surname> <given-names>O.</given-names></name> <name><surname>Yousef</surname> <given-names>M.</given-names></name></person-group> (<year>2022</year>). <article-title>Inflammatory bowel disease biomarkers of human gut microbiota selected via different feature selection methods</article-title>. <source>PeerJ</source> <volume>10</volume>:<fpage>e13205</fpage>. doi: <pub-id pub-id-type="doi">10.7717/peerj.13205</pub-id>, PMID: <pub-id pub-id-type="pmid">35497193</pub-id></citation></ref>
<ref id="ref6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Beghini</surname> <given-names>F.</given-names></name> <name><surname>McIver</surname> <given-names>L.</given-names></name> <name><surname>Blanco-M&#x00ED;guez</surname> <given-names>A.</given-names></name> <name><surname>Dubois</surname> <given-names>L.</given-names></name> <name><surname>Asnicar</surname> <given-names>F.</given-names></name> <name><surname>Maharjan</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Integrating taxonomic, functional, and strain-level profiling of diverse microbial communities with bioBakery 3</article-title>. <source>Elife</source> <volume>10</volume>:<fpage>e65088</fpage>. doi: <pub-id pub-id-type="doi">10.7554/eLife.65088</pub-id>, PMID: <pub-id pub-id-type="pmid">33944776</pub-id></citation></ref>
<ref id="ref7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Berthold</surname> <given-names>M. R.</given-names></name> <name><surname>Cebron</surname> <given-names>N.</given-names></name> <name><surname>Dill</surname> <given-names>F.</given-names></name> <name><surname>Gabriel</surname> <given-names>T. R.</given-names></name> <name><surname>K&#x00F6;tter</surname> <given-names>T.</given-names></name> <name><surname>Meinl</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2009</year>). <article-title>KNIME&#x2013;the Konstanz information miner: version 2.0 and beyond</article-title>. <source>ACM SIGKDD Explor. Newsl.</source> <volume>11</volume>, <fpage>26</fpage>&#x2013;<lpage>31</lpage>. doi: <pub-id pub-id-type="doi">10.1145/1656274.1656280</pub-id></citation></ref>
<ref id="ref8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cena</surname> <given-names>J. A.</given-names></name> <name><surname>Reis</surname> <given-names>L. G.</given-names></name> <name><surname>de Lima</surname> <given-names>A. K. A.</given-names></name> <name><surname>Vieira Lima</surname> <given-names>C. P.</given-names></name> <name><surname>Stefani</surname> <given-names>C. M.</given-names></name> <name><surname>Dame-Teixeira</surname> <given-names>N.</given-names></name></person-group> (<year>2023</year>). <article-title>Enrichment of acid-associated microbiota in the saliva of type 2 diabetes mellitus adults: a systematic review</article-title>. <source>Pathogens</source> <volume>12</volume>:<fpage>404</fpage>. <comment>Available at:</comment>. doi: <pub-id pub-id-type="doi">10.3390/pathogens12030404</pub-id>, PMID: <pub-id pub-id-type="pmid">36986326</pub-id></citation></ref>
<ref id="ref9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>F.</given-names></name> <name><surname>Li</surname> <given-names>S.</given-names></name> <name><surname>Guo</surname> <given-names>R.</given-names></name> <name><surname>Song</surname> <given-names>F.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Meta-analysis of fecal viromes demonstrates high diagnostic potential of the gut viral signatures for colorectal cancer and adenoma risk assessment</article-title>. <source>J. Adv. Res.</source> <volume>49</volume>, <fpage>103</fpage>&#x2013;<lpage>114</lpage>. <comment>Available at:</comment>. doi: <pub-id pub-id-type="doi">10.1016/j.jare.2022.09.012</pub-id>, PMID: <pub-id pub-id-type="pmid">36198381</pub-id></citation></ref>
<ref id="ref10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Desch&#x00EA;nes</surname> <given-names>T.</given-names></name> <name><surname>Tohoundjona</surname> <given-names>F. W. E.</given-names></name> <name><surname>Plante</surname> <given-names>P. L.</given-names></name> <name><surname>di Marzo</surname> <given-names>V.</given-names></name> <name><surname>Raymond</surname> <given-names>F.</given-names></name></person-group> (<year>2023</year>). <article-title>Gene-based microbiome representation enhances host phenotype classification</article-title>. <source>mSystems</source> <volume>8</volume>:<fpage>e0053123</fpage>. doi: <pub-id pub-id-type="doi">10.1128/msystems.00531-23</pub-id>, PMID: <pub-id pub-id-type="pmid">37404032</pub-id></citation></ref>
<ref id="ref11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ding</surname> <given-names>C.</given-names></name> <name><surname>Peng</surname> <given-names>H.</given-names></name></person-group> (<year>2005</year>). <article-title>Minimum redundancy feature selection from microarray gene expression data</article-title>. <source>J. Bioinforma. Comput. Biol.</source> <volume>3</volume>, <fpage>185</fpage>&#x2013;<lpage>205</lpage>. doi: <pub-id pub-id-type="doi">10.1142/S0219720005001004</pub-id></citation></ref>
<ref id="ref12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ditzler</surname> <given-names>G.</given-names></name> <name><surname>Polikar</surname> <given-names>R.</given-names></name> <name><surname>Rosen</surname> <given-names>G.</given-names></name></person-group> (<year>2015</year>). <article-title>Multi-layer and recursive neural networks for metagenomic classification</article-title>. <source>IEEE Trans. Nanobioscience</source> <volume>14</volume>, <fpage>608</fpage>&#x2013;<lpage>616</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TNB.2015.2461219</pub-id>, PMID: <pub-id pub-id-type="pmid">26316190</pub-id></citation></ref>
<ref id="ref13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dix</surname> <given-names>A.</given-names></name> <name><surname>Vlaic</surname> <given-names>S.</given-names></name> <name><surname>Guthke</surname> <given-names>R.</given-names></name> <name><surname>Linde</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). <article-title>Use of systems biology to decipher host&#x2013;pathogen interaction networks and predict biomarkers</article-title>. <source>Clin. Microbiol. Infect.</source> <volume>22</volume>, <fpage>600</fpage>&#x2013;<lpage>606</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.cmi.2016.04.014</pub-id>, PMID: <pub-id pub-id-type="pmid">27113568</pub-id></citation></ref>
<ref id="ref14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Duvallet</surname> <given-names>C.</given-names></name> <name><surname>Gibbons</surname> <given-names>S. M.</given-names></name> <name><surname>Gurry</surname> <given-names>T.</given-names></name> <name><surname>Irizarry</surname> <given-names>R. A.</given-names></name> <name><surname>Alm</surname> <given-names>E. J.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x2018;Meta-analysis of gut microbiome studies identifies disease-specific and shared responses&#x2019;, nature</article-title>. <source>Communications</source> <volume>8</volume>:<fpage>1784</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41467-017-01973-8</pub-id>, PMID: <pub-id pub-id-type="pmid">29209090</pub-id></citation></ref>
<ref id="ref15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ersoz</surname> <given-names>N. S.</given-names></name> <name><surname>Bakir-Gungor</surname> <given-names>B.</given-names></name> <name><surname>Yousef</surname> <given-names>M.</given-names></name></person-group> (<year>2023</year>). <article-title>GeNetOntology: identifying affected gene ontology groups via grouping, scoring and modelling from gene expression data utilizing biological knowledge based machine learning</article-title>. <source>Front. Genet.</source> <volume>14</volume>:<fpage>82</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fgene.2023.1139082</pub-id></citation></ref>
<ref id="ref16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fleuret</surname> <given-names>F.</given-names></name> <name><surname>Ch</surname> <given-names>E.</given-names></name></person-group> (<year>2004</year>). <article-title>Fast binary feature selection with conditional mutual information</article-title>. <source>J. Mach. Learn. Res.</source> <volume>5</volume>, <fpage>1531</fpage>&#x2013;<lpage>1555</lpage>.</citation></ref>
<ref id="ref17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Forslund</surname> <given-names>K.</given-names></name> <name><surname>Hildebrand</surname> <given-names>F.</given-names></name> <name><surname>Nielsen</surname> <given-names>T.</given-names></name> <name><surname>Falony</surname> <given-names>G.</given-names></name> <name><surname>Le Chatelier</surname> <given-names>E.</given-names></name> <name><surname>Sunagawa</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Disentangling type 2 diabetes and metformin treatment signatures in the human gut microbiota</article-title>. <source>Nature</source> <volume>528</volume>, <fpage>262</fpage>&#x2013;<lpage>266</lpage>. doi: <pub-id pub-id-type="doi">10.1038/nature15766</pub-id>, PMID: <pub-id pub-id-type="pmid">26633628</pub-id></citation></ref>
<ref id="ref18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fritz</surname> <given-names>J. V.</given-names></name> <name><surname>Desai</surname> <given-names>M. S.</given-names></name> <name><surname>Shah</surname> <given-names>P.</given-names></name> <name><surname>Schneider</surname> <given-names>J. G.</given-names></name> <name><surname>Wilmes</surname> <given-names>P.</given-names></name></person-group> (<year>2013</year>). <article-title>From meta-omics to causality: experimental models for human microbiome research</article-title>. <source>Microbiome</source> <volume>1</volume>:<fpage>14</fpage>. doi: <pub-id pub-id-type="doi">10.1186/2049-2618-1-14</pub-id>, PMID: <pub-id pub-id-type="pmid">24450613</pub-id></citation></ref>
<ref id="ref19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>R.</given-names></name> <name><surname>Zhu</surname> <given-names>C.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Yin</surname> <given-names>M.</given-names></name> <name><surname>Pan</surname> <given-names>C.</given-names></name> <name><surname>Huang</surname> <given-names>L.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Dysbiosis signatures of gut microbiota along the sequence from healthy, young patients to those with overweight and obesity</article-title>. <source>Obesity</source> <volume>26</volume>, <fpage>351</fpage>&#x2013;<lpage>361</lpage>. doi: <pub-id pub-id-type="doi">10.1002/oby.22088</pub-id></citation></ref>
<ref id="ref20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Giliberti</surname> <given-names>R.</given-names></name> <name><surname>Cavaliere</surname> <given-names>S.</given-names></name> <name><surname>Mauriello</surname> <given-names>I. E.</given-names></name> <name><surname>Ercolini</surname> <given-names>D.</given-names></name> <name><surname>Pasolli</surname> <given-names>E.</given-names></name></person-group> (<year>2022</year>). <article-title>Host phenotype classification from human microbiome data is mainly driven by the presence of microbial taxa</article-title>. <source>PLoS Comput. Biol.</source> <volume>18</volume>:<fpage>e1010066</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pcbi.1010066</pub-id></citation></ref>
<ref id="ref21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gurung</surname> <given-names>M.</given-names></name> <name><surname>Li</surname> <given-names>Z.</given-names></name> <name><surname>You</surname> <given-names>H.</given-names></name> <name><surname>Rodrigues</surname> <given-names>R.</given-names></name> <name><surname>Jump</surname> <given-names>D. B.</given-names></name> <name><surname>Morgun</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Role of gut microbiota in type 2 diabetes pathophysiology</article-title>. <source>EBioMedicine</source> <volume>51</volume>:<fpage>51</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.ebiom.2019.11.051</pub-id>, PMID: <pub-id pub-id-type="pmid">31901868</pub-id></citation></ref>
<ref id="ref22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hassouneh</surname> <given-names>S. A.-D.</given-names></name> <name><surname>Loftus</surname> <given-names>M.</given-names></name> <name><surname>Yooseph</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>Linking inflammatory bowel disease symptoms to changes in the gut microbiome structure and function</article-title>. <source>Front. Microbiol.</source> <volume>12</volume>:<fpage>632</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fmicb.2021.673632</pub-id>, PMID: <pub-id pub-id-type="pmid">34349736</pub-id></citation></ref>
<ref id="ref23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hsu</surname> <given-names>M.</given-names></name> <name><surname>Tun</surname> <given-names>K. M.</given-names></name> <name><surname>Batra</surname> <given-names>K.</given-names></name> <name><surname>Haque</surname> <given-names>L.</given-names></name> <name><surname>Vongsavath</surname> <given-names>T.</given-names></name> <name><surname>Hong</surname> <given-names>A. S.</given-names></name></person-group> (<year>2023</year>). <article-title>Safety and efficacy of fecal microbiota transplantation in treatment of inflammatory bowel disease in the pediatric population: a systematic review and Meta-analysis</article-title>. <source>Microorganisms</source> <volume>11</volume>:<fpage>1272</fpage>. doi: <pub-id pub-id-type="doi">10.3390/microorganisms11051272</pub-id>, PMID: <pub-id pub-id-type="pmid">37317246</pub-id></citation></ref>
<ref id="ref24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huybrechts</surname> <given-names>I.</given-names></name> <name><surname>Zouiouich</surname> <given-names>S.</given-names></name> <name><surname>Loobuyck</surname> <given-names>A.</given-names></name> <name><surname>Vandenbulcke</surname> <given-names>Z.</given-names></name> <name><surname>Vogtmann</surname> <given-names>E.</given-names></name> <name><surname>Pisanu</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>The human microbiome in relation to Cancer risk: a systematic review of epidemiologic studies</article-title>. <source>Cancer Epidemiol. Biomark. Prev.</source> <volume>29</volume>, <fpage>1856</fpage>&#x2013;<lpage>1868</lpage>. doi: <pub-id pub-id-type="doi">10.1158/1055-9965.EPI-20-0288</pub-id>, PMID: <pub-id pub-id-type="pmid">32727720</pub-id></citation></ref>
<ref id="ref25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jabeer</surname> <given-names>A.</given-names></name> <name><surname>Ko&#x00C7;ak</surname> <given-names>A.</given-names></name> <name><surname>Akka&#x015F;</surname> <given-names>H.</given-names></name> <name><surname>Yenisert</surname> <given-names>F.</given-names></name> <name><surname>Nalbanto&#x011F;lu</surname> <given-names>O. U.</given-names></name> <name><surname>Yousef</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Identifying Taxonomic Biomarkers of Colorectal Cancer in Human Intestinal Microbiota Using Multiple Feature Selection Methods&#x2019;, in 2022 Innovations in Intelligent Systems and Applications Conference (ASYU)</article-title>. <source>IEEE</source> <volume>2022</volume>, <fpage>1</fpage>&#x2013;<lpage>6</lpage>. doi: <pub-id pub-id-type="doi">10.1109/ASYU56188.2022.9925551</pub-id></citation></ref>
<ref id="ref26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jabeer</surname> <given-names>A.</given-names></name> <name><surname>Temiz</surname> <given-names>M.</given-names></name> <name><surname>Bakir-Gungor</surname> <given-names>B.</given-names></name> <name><surname>Yousef</surname> <given-names>M.</given-names></name></person-group> (<year>2023</year>). <article-title>miRdisNET: discovering microRNA biomarkers that are associated with diseases utilizing biological knowledge-based machine learning</article-title>. <source>Front. Genet.</source> <volume>13</volume>:<fpage>1076554</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fgene.2022.1076554</pub-id>, PMID: <pub-id pub-id-type="pmid">36712859</pub-id></citation></ref>
<ref id="ref27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kolde</surname> <given-names>R.</given-names></name> <name><surname>Laur</surname> <given-names>S.</given-names></name> <name><surname>Adler</surname> <given-names>P.</given-names></name> <name><surname>Vilo</surname> <given-names>J.</given-names></name></person-group> (<year>2012</year>). <article-title>Robust rank aggregation for gene list integration and meta-analysis</article-title>. <source>Bioinformatics</source> <volume>28</volume>, <fpage>573</fpage>&#x2013;<lpage>580</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btr709</pub-id></citation></ref>
<ref id="ref28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kuzudisli</surname> <given-names>C.</given-names></name> <name><surname>Bakir-Gungor</surname> <given-names>B.</given-names></name> <name><surname>Bulut</surname> <given-names>N.</given-names></name> <name><surname>Qaqish</surname> <given-names>B.</given-names></name> <name><surname>Yousef</surname> <given-names>M.</given-names></name></person-group> (<year>2023</year>). <article-title>Review of feature selection approaches based on grouping of features</article-title>. <source>PeerJ</source> <volume>11</volume>:<fpage>e15666</fpage>. doi: <pub-id pub-id-type="doi">10.7717/peerj.15666</pub-id></citation></ref>
<ref id="ref29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>LaPierre</surname> <given-names>N.</given-names></name> <name><surname>Ju</surname> <given-names>C. J. T.</given-names></name> <name><surname>Zhou</surname> <given-names>G.</given-names></name> <name><surname>Wang</surname> <given-names>W.</given-names></name></person-group> (<year>2019</year>). <article-title>MetaPheno: a critical evaluation of deep learning and machine learning in metagenome-based disease prediction</article-title>. <source>Methods</source> <volume>166</volume>, <fpage>74</fpage>&#x2013;<lpage>82</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.ymeth.2019.03.003</pub-id>, PMID: <pub-id pub-id-type="pmid">30885720</pub-id></citation></ref>
<ref id="ref30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Levy</surname> <given-names>S. E.</given-names></name> <name><surname>Myers</surname> <given-names>R. M.</given-names></name></person-group> (<year>2016</year>). <article-title>Advancements in next-generation sequencing</article-title>. <source>Annu. Rev. Genomics Hum. Genet.</source> <volume>17</volume>, <fpage>95</fpage>&#x2013;<lpage>115</lpage>. doi: <pub-id pub-id-type="doi">10.1146/annurev-genom-083115-022413</pub-id></citation></ref>
<ref id="ref31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Feng</surname> <given-names>J.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Liu</surname> <given-names>G.</given-names></name> <name><surname>Wang</surname> <given-names>F.</given-names></name></person-group> (<year>2023</year>). <article-title>Features of combined gut bacteria and fungi from a Chinese cohort of colorectal cancer, colorectal adenoma, and post-operative patients</article-title>. <source>Front. Microbiol.</source> <volume>14</volume>:<fpage>583</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fmicb.2023.1236583</pub-id>, PMID: <pub-id pub-id-type="pmid">37614602</pub-id></citation></ref>
<ref id="ref32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>R.</given-names></name> <name><surname>Shokri</surname> <given-names>F.</given-names></name> <name><surname>Rincon</surname> <given-names>A. L.</given-names></name> <name><surname>Rivadeneira</surname> <given-names>F.</given-names></name> <name><surname>Medina-Gomez</surname> <given-names>C.</given-names></name> <name><surname>Ahmadizar</surname> <given-names>F.</given-names></name></person-group> (<year>2023</year>). <article-title>Bi-directional interactions between glucose-lowering medications and gut microbiome in patients with type 2 diabetes mellitus: a systematic review</article-title>. <source>Genes</source> <volume>14</volume>:<fpage>1572</fpage>. doi: <pub-id pub-id-type="doi">10.3390/genes14081572</pub-id>, PMID: <pub-id pub-id-type="pmid">37628624</pub-id></citation></ref>
<ref id="ref33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lim</surname> <given-names>H.</given-names></name> <name><surname>Cankara</surname> <given-names>F.</given-names></name> <name><surname>Tsai</surname> <given-names>C. J.</given-names></name> <name><surname>Keskin</surname> <given-names>O.</given-names></name> <name><surname>Nussinov</surname> <given-names>R.</given-names></name> <name><surname>Gursoy</surname> <given-names>A.</given-names></name></person-group> (<year>2022</year>). <article-title>Artificial intelligence approaches to human-microbiome protein&#x2013;protein interactions</article-title>. <source>Curr. Opin. Struct. Biol.</source> <volume>73</volume>:<fpage>102328</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.sbi.2022.102328</pub-id>, PMID: <pub-id pub-id-type="pmid">35152186</pub-id></citation></ref>
<ref id="ref34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ma</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Xiang</surname> <given-names>J.</given-names></name> <name><surname>Xiang</surname> <given-names>S.</given-names></name> <name><surname>Zhao</surname> <given-names>Y.</given-names></name> <name><surname>Xiao</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Metagenome analysis of intestinal Bacteria in healthy people, patients with inflammatory bowel disease and colorectal Cancer</article-title>. <source>Front. Cell. Infect. Microbiol.</source> <volume>11</volume>:<fpage>734</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fcimb.2021.599734</pub-id>, PMID: <pub-id pub-id-type="pmid">33738265</pub-id></citation></ref>
<ref id="ref35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mah</surname> <given-names>C.</given-names></name> <name><surname>Jayawardana</surname> <given-names>T.</given-names></name> <name><surname>Leong</surname> <given-names>G.</given-names></name> <name><surname>Koentgen</surname> <given-names>S.</given-names></name> <name><surname>Lemberg</surname> <given-names>D.</given-names></name> <name><surname>Connor</surname> <given-names>S. J.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Assessing the relationship between the gut microbiota and inflammatory bowel disease therapeutics: a systematic review</article-title>. <source>Pathogens</source> <volume>12</volume>:<fpage>262</fpage>. doi: <pub-id pub-id-type="doi">10.3390/pathogens12020262</pub-id>, PMID: <pub-id pub-id-type="pmid">36839534</pub-id></citation></ref>
<ref id="ref36"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Marco-Ramell</surname> <given-names>A.</given-names></name> <name><surname>Palau-Rodriguez</surname> <given-names>M.</given-names></name> <name><surname>Alay</surname> <given-names>A.</given-names></name> <name><surname>Tulipani</surname> <given-names>S.</given-names></name> <name><surname>Urpi-Sarda</surname> <given-names>M.</given-names></name> <name><surname>Sanchez-Pla</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Evaluation and comparison of bioinformatic tools for the enrichment analysis of metabolomics data</article-title>. <source>BMC Bioinformatics</source> <volume>19</volume>:<fpage>1</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s12859-017-2006-0</pub-id>, PMID: <pub-id pub-id-type="pmid">29291722</pub-id></citation></ref>
<ref id="ref37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Marcos-Zambrano</surname> <given-names>L. J.</given-names></name> <name><surname>Karaduzovic-Hadziabdic</surname> <given-names>K.</given-names></name> <name><surname>Loncar Turukalo</surname> <given-names>T.</given-names></name> <name><surname>Przymus</surname> <given-names>P.</given-names></name> <name><surname>Trajkovik</surname> <given-names>V.</given-names></name> <name><surname>Aasmets</surname> <given-names>O.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Applications of machine learning in human microbiome studies: a review on feature selection, biomarker identification, disease prediction and treatment</article-title>. <source>Front. Microbiol.</source> <volume>12</volume>:<fpage>511</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fmicb.2021.634511</pub-id>, PMID: <pub-id pub-id-type="pmid">33737920</pub-id></citation></ref>
<ref id="ref38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Martin</surname> <given-names>A. M.</given-names></name> <name><surname>Yabut</surname> <given-names>J. M.</given-names></name> <name><surname>Choo</surname> <given-names>J. M.</given-names></name> <name><surname>Page</surname> <given-names>A. J.</given-names></name> <name><surname>Sun</surname> <given-names>E. W.</given-names></name> <name><surname>Jessup</surname> <given-names>C. F.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>The gut microbiome regulates host glucose homeostasis via peripheral serotonin</article-title>. <source>Proc. Natl. Acad. Sci.</source> <volume>116</volume>, <fpage>19802</fpage>&#x2013;<lpage>19804</lpage>. doi: <pub-id pub-id-type="doi">10.1073/pnas.1909311116</pub-id>, PMID: <pub-id pub-id-type="pmid">31527237</pub-id></citation></ref>
<ref id="ref39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>McDonald</surname> <given-names>D.</given-names></name> <name><surname>Hyde</surname> <given-names>E.</given-names></name> <name><surname>Debelius</surname> <given-names>J. W.</given-names></name> <name><surname>Morton</surname> <given-names>J. T.</given-names></name> <name><surname>Gonzalez</surname> <given-names>A.</given-names></name> <name><surname>Ackermann</surname> <given-names>G.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>American gut: an open platform for citizen science microbiome research</article-title>. <source>mSystems</source> <volume>3</volume>:<fpage>e00031</fpage>. doi: <pub-id pub-id-type="doi">10.1128/mSystems.00031-18</pub-id>, PMID: <pub-id pub-id-type="pmid">29795809</pub-id></citation></ref>
<ref id="ref40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mendes</surname> <given-names>V.</given-names></name> <name><surname>Galv&#x00E3;o</surname> <given-names>I.</given-names></name> <name><surname>Vieira</surname> <given-names>A. T.</given-names></name></person-group> (<year>2019</year>). <article-title>Mechanisms by which the gut microbiota influences cytokine production and modulates host inflammatory responses</article-title>. <source>J. Interf. Cytokine Res.</source> <volume>39</volume>, <fpage>393</fpage>&#x2013;<lpage>409</lpage>. doi: <pub-id pub-id-type="doi">10.1089/jir.2019.0011</pub-id></citation></ref>
<ref id="ref41"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Muller</surname> <given-names>E. E. L.</given-names></name></person-group> (<year>2019</year>). <article-title>Determining microbial niche breadth in the environment for better ecosystem fate predictions</article-title>. <source>mSystems</source> <volume>4</volume>:<fpage>19</fpage>. doi: <pub-id pub-id-type="doi">10.1128/msystems.00080-19</pub-id></citation></ref>
<ref id="ref42"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Negrut</surname> <given-names>R. L.</given-names></name> <name><surname>Cote</surname> <given-names>A.</given-names></name> <name><surname>Maghiar</surname> <given-names>A. M.</given-names></name></person-group> (<year>2023</year>). <article-title>Exploring the potential of Oral microbiome biomarkers for colorectal Cancer diagnosis and prognosis: a systematic review</article-title>. <source>Microorganisms</source> <volume>11</volume>:<fpage>1586</fpage>. doi: <pub-id pub-id-type="doi">10.3390/microorganisms11061586</pub-id>, PMID: <pub-id pub-id-type="pmid">37375087</pub-id></citation></ref>
<ref id="ref43"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ni</surname> <given-names>Y.</given-names></name> <name><surname>Mu</surname> <given-names>C.</given-names></name> <name><surname>He</surname> <given-names>X.</given-names></name> <name><surname>Zheng</surname> <given-names>K.</given-names></name> <name><surname>Guo</surname> <given-names>H.</given-names></name> <name><surname>Zhu</surname> <given-names>W.</given-names></name></person-group> (<year>2018</year>). <article-title>Characteristics of gut microbiota and its response to a Chinese herbal formula in elder patients with metabolic syndrome</article-title>. <source>Drug Discov. Ther.</source> <volume>12</volume>, <fpage>161</fpage>&#x2013;<lpage>169</lpage>. doi: <pub-id pub-id-type="doi">10.5582/ddt.2018.01036</pub-id></citation></ref>
<ref id="ref44"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ohland</surname> <given-names>C. L.</given-names></name> <name><surname>Jobin</surname> <given-names>C.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x2018;Microbial activities and intestinal homeostasis: a delicate balance between health and disease&#x2019;, cellular and molecular</article-title>. <source>Gastroenterol. Hepatol.</source> <volume>1</volume>, <fpage>28</fpage>&#x2013;<lpage>40</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jcmgh.2014.11.004</pub-id>, PMID: <pub-id pub-id-type="pmid">25729763</pub-id></citation></ref>
<ref id="ref45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Oudah</surname> <given-names>M.</given-names></name> <name><surname>Henschel</surname> <given-names>A.</given-names></name></person-group> (<year>2018</year>). <article-title>Taxonomy-aware feature engineering for microbiome classification</article-title>. <source>BMC Bioinformatics</source> <volume>19</volume>:<fpage>227</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s12859-018-2205-3</pub-id>, PMID: <pub-id pub-id-type="pmid">29907097</pub-id></citation></ref>
<ref id="ref46"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pasolli</surname> <given-names>E.</given-names></name> <name><surname>Truong</surname> <given-names>D. T.</given-names></name> <name><surname>Malik</surname> <given-names>F.</given-names></name> <name><surname>Waldron</surname> <given-names>L.</given-names></name> <name><surname>Segata</surname> <given-names>N.</given-names></name></person-group> (<year>2016</year>). <article-title>Machine learning Meta-analysis of large metagenomic datasets: tools and biological insights</article-title>. <source>PLoS Comput. Biol.</source> <volume>12</volume>:<fpage>e1004977</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pcbi.1004977</pub-id>, PMID: <pub-id pub-id-type="pmid">27400279</pub-id></citation></ref>
<ref id="ref47"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Pedregosa</surname> <given-names>F.</given-names></name> <name><surname>Varoquaux</surname> <given-names>G.</given-names></name> <name><surname>Gramfort</surname> <given-names>A.</given-names></name> <name><surname>Michel</surname> <given-names>V.</given-names></name> <name><surname>Thirion</surname> <given-names>B.</given-names></name> <name><surname>Grisel</surname> <given-names>O.</given-names></name> <etal/></person-group>. (<year>2011</year>). &#x2018;<italic>Scikit-learn: Machine learning in Python&#x2019;, Machine Learning in Python</italic>.</citation></ref>
<ref id="ref48"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Petersen</surname> <given-names>C.</given-names></name> <name><surname>Round</surname> <given-names>J. L.</given-names></name></person-group> (<year>2014</year>). <article-title>Defining dysbiosis and its influence on host immunity and disease</article-title>. <source>Cell. Microbiol.</source> <volume>16</volume>, <fpage>1024</fpage>&#x2013;<lpage>1033</lpage>. doi: <pub-id pub-id-type="doi">10.1111/cmi.12308</pub-id>, PMID: <pub-id pub-id-type="pmid">24798552</pub-id></citation></ref>
<ref id="ref49"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pickard</surname> <given-names>J. M.</given-names></name> <name><surname>Zeng</surname> <given-names>M. Y.</given-names></name> <name><surname>Caruso</surname> <given-names>R.</given-names></name> <name><surname>N&#x00FA;&#x00F1;ez</surname> <given-names>G.</given-names></name></person-group> (<year>2017</year>). <article-title>Gut microbiota: role in pathogen colonization, immune responses, and inflammatory disease</article-title>. <source>Immunol. Rev.</source> <volume>279</volume>, <fpage>70</fpage>&#x2013;<lpage>89</lpage>. doi: <pub-id pub-id-type="doi">10.1111/imr.12567</pub-id>, PMID: <pub-id pub-id-type="pmid">28856738</pub-id></citation></ref>
<ref id="ref50"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qin</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Cai</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>S.</given-names></name> <name><surname>Zhu</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>F.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>A metagenome-wide association study of gut microbiota in type 2 diabetes</article-title>. <source>Nature</source> <volume>490</volume>, <fpage>55</fpage>&#x2013;<lpage>60</lpage>. doi: <pub-id pub-id-type="doi">10.1038/nature11450</pub-id>, PMID: <pub-id pub-id-type="pmid">23023125</pub-id></citation></ref>
<ref id="ref51"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qin</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>R.</given-names></name> <name><surname>Raes</surname> <given-names>J.</given-names></name> <name><surname>Arumugam</surname> <given-names>M.</given-names></name> <name><surname>Burgdorf</surname> <given-names>K. S.</given-names></name> <name><surname>Manichanh</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2010</year>). <article-title>A human gut microbial gene catalogue established by metagenomic sequencing</article-title>. <source>Nature</source> <volume>464</volume>, <fpage>59</fpage>&#x2013;<lpage>65</lpage>. doi: <pub-id pub-id-type="doi">10.1038/nature08821</pub-id>, PMID: <pub-id pub-id-type="pmid">20203603</pub-id></citation></ref>
<ref id="ref52"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qin</surname> <given-names>N.</given-names></name> <name><surname>Yang</surname> <given-names>F.</given-names></name> <name><surname>Li</surname> <given-names>A.</given-names></name> <name><surname>Prifti</surname> <given-names>E.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Shao</surname> <given-names>L.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>Alterations of the human gut microbiome in liver cirrhosis</article-title>. <source>Nature</source> <volume>513</volume>, <fpage>59</fpage>&#x2013;<lpage>64</lpage>. doi: <pub-id pub-id-type="doi">10.1038/nature13568</pub-id></citation></ref>
<ref id="ref53"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qumsiyeh</surname> <given-names>E.</given-names></name> <name><surname>Showe</surname> <given-names>L.</given-names></name> <name><surname>Yousef</surname> <given-names>M.</given-names></name></person-group> (<year>2022</year>). <article-title>GediNET for discovering gene associations across diseases using knowledge based machine learning approach</article-title>. <source>Sci. Rep.</source> <volume>12</volume>:<fpage>19955</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41598-022-24421-0</pub-id>, PMID: <pub-id pub-id-type="pmid">36402891</pub-id></citation></ref>
<ref id="ref54"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Salamon</surname> <given-names>D.</given-names></name> <name><surname>Sroka-Oleksiak</surname> <given-names>A.</given-names></name> <name><surname>Kapusta</surname> <given-names>P.</given-names></name> <name><surname>Szopa</surname> <given-names>M.</given-names></name> <name><surname>Mrozi&#x0144;ska</surname> <given-names>S.</given-names></name> <name><surname>Ludwig-S&#x0142;omczy&#x0144;ska</surname> <given-names>A. H.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Characteristics of the gut microbiota in adult patients with type 1 and 2 diabetes based on the analysis of a fragment of 16S rRNA gene using next-generation sequencing</article-title>. <source>Pol. Arch. Intern. Med.</source> <volume>128</volume>, <fpage>336</fpage>&#x2013;<lpage>343</lpage>. doi: <pub-id pub-id-type="doi">10.20452/pamw.4246</pub-id>, PMID: <pub-id pub-id-type="pmid">29657308</pub-id></citation></ref>
<ref id="ref55"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sedighi</surname> <given-names>M.</given-names></name> <name><surname>Razavi</surname> <given-names>S.</given-names></name> <name><surname>Navab-Moghadam</surname> <given-names>F.</given-names></name> <name><surname>Khamseh</surname> <given-names>M. E.</given-names></name> <name><surname>Alaei-Shahmiri</surname> <given-names>F.</given-names></name> <name><surname>Mehrtash</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>Comparison of gut microbiota in adult patients with type 2 diabetes and healthy individuals</article-title>. <source>Microb. Pathog.</source> <volume>111</volume>, <fpage>362</fpage>&#x2013;<lpage>369</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.micpath.2017.08.038</pub-id></citation></ref>
<ref id="ref56"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Senliol</surname> <given-names>B.</given-names></name> <name><surname>Gulgezen</surname> <given-names>G.</given-names></name> <name><surname>Yu</surname> <given-names>L.</given-names></name> <name><surname>Cataltepe</surname> <given-names>Z.</given-names></name></person-group>. (<year>2008</year>) <italic>Fast correlation based filter (FCBF) with a different search strategy</italic>. In: 2008 23rd international symposium on computer and information sciences. 2008 23rd international symposium on computer and information sciences, pp. 1&#x2013;4.</citation></ref>
<ref id="ref57"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sharma</surname> <given-names>D.</given-names></name> <name><surname>Paterson</surname> <given-names>A. D.</given-names></name> <name><surname>Xu</surname> <given-names>W.</given-names></name></person-group> (<year>2020</year>). <article-title>TaxoNN: ensemble of neural networks on stratified microbiome data for disease prediction</article-title>. <source>Bioinformatics</source> <volume>36</volume>, <fpage>4544</fpage>&#x2013;<lpage>4550</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa542</pub-id>, PMID: <pub-id pub-id-type="pmid">32449747</pub-id></citation></ref>
<ref id="ref58"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Song</surname> <given-names>M.</given-names></name> <name><surname>Chan</surname> <given-names>A. T.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). <article-title>Influence of the gut microbiome, diet, and environment on risk of colorectal Cancer</article-title>. <source>Gastroenterology</source> <volume>158</volume>, <fpage>322</fpage>&#x2013;<lpage>340</lpage>. doi: <pub-id pub-id-type="doi">10.1053/j.gastro.2019.06.048</pub-id>, PMID: <pub-id pub-id-type="pmid">31586566</pub-id></citation></ref>
<ref id="ref59"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Soueidan</surname> <given-names>H.</given-names></name> <name><surname>Nikolski</surname> <given-names>M.</given-names></name></person-group> (<year>2016</year>). <article-title>Machine learning for metagenomics: methods and tools</article-title>. <source>arXiv</source> <volume>2016</volume>:<fpage>621</fpage>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.1510.06621</pub-id></citation></ref>
<ref id="ref60"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tabowei</surname> <given-names>G.</given-names></name> <name><surname>Gaddipati</surname> <given-names>G. N.</given-names></name> <name><surname>Mukhtar</surname> <given-names>M.</given-names></name> <name><surname>Alzubaidee</surname> <given-names>M. J.</given-names></name> <name><surname>Dwarampudi</surname> <given-names>R. S.</given-names></name> <name><surname>Mathew</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Microbiota Dysbiosis a cause of colorectal Cancer or not? A systematic review</article-title>. <source>Cureus</source> <volume>14</volume>, <volume>14</volume>:<fpage>e30893</fpage>. doi: <pub-id pub-id-type="doi">10.7759/cureus.30893</pub-id>, PMID: <pub-id pub-id-type="pmid">36465770</pub-id></citation></ref>
<ref id="ref61"><citation citation-type="journal"><person-group person-group-type="author"><collab id="coll1">The Human Microbiome Project Consortium</collab></person-group> (<year>2012</year>). <article-title>Structure, function and diversity of the healthy human microbiome</article-title>. <source>Nature</source> <volume>486</volume>, <fpage>207</fpage>&#x2013;<lpage>214</lpage>. doi: <pub-id pub-id-type="doi">10.1038/nature11234</pub-id>, PMID: <pub-id pub-id-type="pmid">22699609</pub-id></citation></ref>
<ref id="ref62"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Thomas</surname> <given-names>A. M.</given-names></name> <name><surname>Manghi</surname> <given-names>P.</given-names></name> <name><surname>Asnicar</surname> <given-names>F.</given-names></name> <name><surname>Pasolli</surname> <given-names>E.</given-names></name> <name><surname>Armanini</surname> <given-names>F.</given-names></name> <name><surname>Zolfo</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Metagenomic analysis of colorectal cancer datasets identifies cross-cohort microbial diagnostic signatures and a link with choline degradation</article-title>. <source>Nat. Med.</source> <volume>25</volume>, <fpage>667</fpage>&#x2013;<lpage>678</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41591-019-0405-7</pub-id>, PMID: <pub-id pub-id-type="pmid">30936548</pub-id></citation></ref>
<ref id="ref63"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tibshirani</surname> <given-names>R.</given-names></name></person-group> (<year>1996</year>). <article-title>Regression shrinkage and selection via the lasso</article-title>. <source>J. R. Stat. Soc. Series B (Methodological)</source> <volume>58</volume>, <fpage>267</fpage>&#x2013;<lpage>288</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.2517-6161.1996.tb02080.x</pub-id></citation></ref>
<ref id="ref64"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Top&#x00E7;uo&#x011F;lu</surname> <given-names>B. D.</given-names></name> <name><surname>Lesniak</surname> <given-names>N. A.</given-names></name> <name><surname>Ruffin</surname> <given-names>M. T.</given-names> <suffix>IV</suffix></name> <name><surname>Wiens</surname> <given-names>J.</given-names></name> <name><surname>Schloss</surname> <given-names>P. D.</given-names></name></person-group> (<year>2020</year>). <article-title>A framework for effective application of machine learning to microbiome-based classification problems</article-title>. <source>MBio</source> <volume>11</volume>, <fpage>e00434</fpage>&#x2013;<lpage>e00420</lpage>. doi: <pub-id pub-id-type="doi">10.1128/mBio.00434-20</pub-id></citation></ref>
<ref id="ref65"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Unlu Yazici</surname> <given-names>M.</given-names></name> <name><surname>Marron</surname> <given-names>J. S.</given-names></name> <name><surname>Bakir-Gungor</surname> <given-names>B.</given-names></name> <name><surname>Zou</surname> <given-names>F.</given-names></name> <name><surname>Yousef</surname> <given-names>M.</given-names></name></person-group> (<year>2023</year>). <article-title>Invention of 3Mint for feature grouping and scoring in multi-omics</article-title>. <source>Front. Genet.</source> <volume>14</volume>:<fpage>1093326</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fgene.2023.1093326</pub-id>, PMID: <pub-id pub-id-type="pmid">37007972</pub-id></citation></ref>
<ref id="ref66"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>X.-W.</given-names></name> <name><surname>Liu</surname> <given-names>Y.-Y.</given-names></name></person-group> (<year>2020</year>). <article-title>Comparative study of classifiers for human microbiome data</article-title>. <source>Med. Microecol.</source> <volume>4</volume>:<fpage>100013</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.medmic.2020.100013</pub-id>, PMID: <pub-id pub-id-type="pmid">34368751</pub-id></citation></ref>
<ref id="ref67"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yousef</surname> <given-names>M.</given-names></name> <name><surname>Abdallah</surname> <given-names>L.</given-names></name> <name><surname>Allmer</surname> <given-names>J.</given-names></name></person-group> (<year>2019</year>). <article-title>maTE: discovering expressed interactions between microRNAs and their targets</article-title>. <source>Bioinformatics</source> <volume>35</volume>, <fpage>4020</fpage>&#x2013;<lpage>4028</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btz204</pub-id>, PMID: <pub-id pub-id-type="pmid">30895309</pub-id></citation></ref>
<ref id="ref68"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yousef</surname> <given-names>M.</given-names></name> <name><surname>Goy</surname> <given-names>G.</given-names></name> <name><surname>Bakir-Gungor</surname> <given-names>B.</given-names></name></person-group> (<year>2022b</year>). <article-title>miRModuleNet: detecting miRNA-mRNA regulatory modules</article-title>. <source>Front. Genet.</source> <volume>13</volume>:<fpage>455</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fgene.2022.767455</pub-id>, PMID: <pub-id pub-id-type="pmid">35495139</pub-id></citation></ref>
<ref id="ref69"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yousef</surname> <given-names>M.</given-names></name> <name><surname>Goy</surname> <given-names>G.</given-names></name> <name><surname>Mitra</surname> <given-names>R.</given-names></name> <name><surname>Eischen</surname> <given-names>C. M.</given-names></name> <name><surname>Jabeer</surname> <given-names>A.</given-names></name> <name><surname>Bakir-Gungor</surname> <given-names>B.</given-names></name></person-group> (<year>2021a</year>). <article-title>miRcorrNet: machine learning-based integration of miRNA and mRNA expression profiles, combined with feature grouping and ranking</article-title>. <source>PeerJ</source> <volume>9</volume>:<fpage>e11458</fpage>. doi: <pub-id pub-id-type="doi">10.7717/peerj.11458</pub-id>, PMID: <pub-id pub-id-type="pmid">34055490</pub-id></citation></ref>
<ref id="ref70"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yousef</surname> <given-names>M.</given-names></name> <name><surname>Kumar</surname> <given-names>A.</given-names></name> <name><surname>Bakir-Gungor</surname> <given-names>B.</given-names></name></person-group> (<year>2021b</year>). <article-title>Application of biological domain knowledge based feature selection on gene expression data</article-title>. <source>Entropy</source> <volume>23</volume>:<fpage>2</fpage>. doi: <pub-id pub-id-type="doi">10.3390/e23010002</pub-id>, PMID: <pub-id pub-id-type="pmid">33374969</pub-id></citation></ref>
<ref id="ref71"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yousef</surname> <given-names>M.</given-names></name> <name><surname>Ozdemir</surname> <given-names>F.</given-names></name> <name><surname>Jaber</surname> <given-names>A.</given-names></name> <name><surname>Allmer</surname> <given-names>J.</given-names></name> <name><surname>Bakir-Gungor</surname> <given-names>B.</given-names></name></person-group> (<year>2022a</year>). <article-title>PriPath: Identifying dysregulated pathways from differential gene expression via grouping, scoring and modeling with an embedded machine learning approach</article-title>. <source>BMC Bioinformatics</source> <volume>24</volume>:<fpage>60</fpage>. doi: <pub-id pub-id-type="doi">10.21203/rs.3.rs-1449467/v1</pub-id></citation></ref>
<ref id="ref72"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yousef</surname> <given-names>M.</given-names></name> <name><surname>&#x00DC;lgen</surname> <given-names>E.</given-names></name> <name><surname>U&#x011F;ur Sezerman</surname> <given-names>O.</given-names></name></person-group> (<year>2021c</year>). <article-title>CogNet: classification of gene expression data based on ranked active-subnetwork-oriented KEGG pathway enrichment analysis</article-title>. <source>PeerJ Comput. Sci.</source> <volume>7</volume>:<fpage>e336</fpage>. doi: <pub-id pub-id-type="doi">10.7717/peerj-cs.336</pub-id>, PMID: <pub-id pub-id-type="pmid">33816987</pub-id></citation></ref>
<ref id="ref73"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yousef</surname> <given-names>M.</given-names></name> <name><surname>Voskergian</surname> <given-names>D.</given-names></name></person-group> (<year>2022</year>). <article-title>TextNetTopics: text classification based word grouping as topics and topics&#x2019; scoring</article-title>. <source>Front. Genet.</source> <volume>13</volume>:<fpage>893378</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fgene.2022.893378</pub-id>, PMID: <pub-id pub-id-type="pmid">35795215</pub-id></citation></ref>
<ref id="ref74"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Bhosle</surname> <given-names>A.</given-names></name> <name><surname>Bae</surname> <given-names>S.</given-names></name> <name><surname>McIver</surname> <given-names>L. J.</given-names></name> <name><surname>Pishchany</surname> <given-names>G.</given-names></name> <name><surname>Accorsi</surname> <given-names>E. K.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Discovery of bioactive microbial gene products in inflammatory bowel disease</article-title>. <source>Nature</source> <volume>606</volume>, <fpage>754</fpage>&#x2013;<lpage>760</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41586-022-04648-7</pub-id>, PMID: <pub-id pub-id-type="pmid">35614211</pub-id></citation></ref>
<ref id="ref75"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>W.</given-names></name> <name><surname>Liu</surname> <given-names>A.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Chen</surname> <given-names>G.</given-names></name> <name><surname>Li</surname> <given-names>Q.</given-names></name></person-group> (<year>2022</year>). <article-title>An adaptive direction-assisted test for microbiome compositional data</article-title>. <source>Bioinformatics</source> <volume>38</volume>, <fpage>3493</fpage>&#x2013;<lpage>3500</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btac361</pub-id>, PMID: <pub-id pub-id-type="pmid">35640978</pub-id></citation></ref>
<ref id="ref76"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zou</surname> <given-names>H.</given-names></name> <name><surname>Hastie</surname> <given-names>T.</given-names></name></person-group> (<year>2005</year>). <article-title>Regularization and variable selection via the elastic net</article-title>. <source>J. R. Stat. Soc. Series B (Statistical Methodology)</source> <volume>67</volume>, <fpage>301</fpage>&#x2013;<lpage>320</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.1467-9868.2005.00503.x</pub-id></citation></ref>
<ref id="ref77"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zwezerijnen-Jiwa</surname> <given-names>F. H.</given-names></name> <name><surname>Sivov</surname> <given-names>H.</given-names></name> <name><surname>Paizs</surname> <given-names>P.</given-names></name> <name><surname>Zafeiropoulou</surname> <given-names>K.</given-names></name> <name><surname>Kinross</surname> <given-names>J.</given-names></name></person-group> (<year>2023</year>). <article-title>A systematic review of microbiome-derived biomarkers for early colorectal cancer detection</article-title>. <source>Neoplasia</source> <volume>36</volume>:<fpage>100868</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neo.2022.100868</pub-id>, PMID: <pub-id pub-id-type="pmid">36566591</pub-id></citation></ref>
</ref-list>
</back>
</article>
