<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article article-type="methods-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1618472</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2025.1618472</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>BANSMDA: a computational model for predicting potential microbe-disease associations based on bilinear attention networks and sparse autoencoders</article-title>
<alt-title alt-title-type="left-running-head">Liu et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2025.1618472">10.3389/fgene.2025.1618472</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Liu</surname>
<given-names>Xianzhi</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2945242/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Liang</surname>
<given-names>Mingmin</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2842979/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yu</surname>
<given-names>Ge</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3048736/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Tang</surname>
<given-names>Shichang</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3072418/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Wu</surname>
<given-names>Ouxiang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zeng</surname>
<given-names>Bin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1999059/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Lei</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/664933/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Information Engineering</institution>, Hunan Vocational College of Electronic and Technology, <addr-line>Changsha</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>School of Intelligent Equipment</institution>, Hunan Vocational College of Electronic and Technology, <addr-line>Changsha</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>School of Continuing Education</institution>, <institution>Central South University of Forestry and Technology</institution>, <addr-line>Changsha</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Big Data Innovation and Entrepreneurship Education Center of Hunan Province</institution>, Changsha University, <addr-line>Changsha</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/827295/overview">Lei Chen</ext-link>, Shanghai Maritime University, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/901589/overview">Xin Jin</ext-link>, Biotechnology HPC Software Applications Institute (BHSAI), United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1932673/overview">Shengbo Wu</ext-link>, Tianjin University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Xianzhi Liu, <email>lxz19920125@163.com</email>; Mingmin Liang, <email>1175858629@qq.com</email>; Ouxiang Wu, <email>347517450@qq.com</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>08</day>
<month>08</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1618472</elocation-id>
<history>
<date date-type="received">
<day>26</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>07</day>
<month>07</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Liu, Liang, Yu, Tang, Wu, Zeng and Wang.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Liu, Liang, Yu, Tang, Wu, Zeng and Wang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Predicting the relationship between diseases and microbes can significantly enhance disease diagnosis and treatment, while providing crucial scientific support for public health, ecological health, and drug development.</p>
</sec>
<sec>
<title>Methods</title>
<p>In this manuscript, we introduce an innovative computational model named BANSMDA, which integrates Bilinear Attention Networks with sparse autoencoder to uncover hidden connections between microbes and diseases. In BANSMDA, we first constructed a heterogeneous microbe-disease network by integrating multiple Gaussian similarity measures for diseases and microbes, along with known microbe-disease associations. And then, we employed a BAN-based autoencoder and a sparse autoencoder module to learn node representations within this newly constructed heterogeneous network. Finally, we evaluated the prediction performance of BANSMDA using a 5-fold cross-validation framework.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>Experiments results showed that BANSMDA achieved superior performance compared to other cutting-edge methods. To further assess its effectiveness, we carried out case studies on two common diseases (including Asthma and Colorectal carcinoma) and two important microbial genera (including <italic>Escherichia</italic> and <italic>Bacteroides</italic>), and in the top 20 predicted microbes, there were 19 and 20 having been confirmed by published literature respectively. Besides, in the top 20 predicted diseases, there were 19 and 19 having been confirmed by published literature separately. Therefore, it is easy to conclude that BANSMDA can achieve satisfactory prediction ability.</p>
</sec>
</abstract>
<kwd-group>
<kwd>computational model</kwd>
<kwd>microbe-disease associations</kwd>
<kwd>bilinear attention networks</kwd>
<kwd>sparse autoencoder</kwd>
<kwd>prediction</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Genomics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>A multitude of studies has underscored the significant influence that parasitic microbial communities within the human body exert on our metabolic processes (<xref ref-type="bibr" rid="B18">Kau et al., 2011</xref>). These microbes offer a range of benefits to humans, including the collection and storage of energy, the facilitation of organic compound absorption, and the defense against external microbes and diseases (<xref ref-type="bibr" rid="B21">Kim et al., 2018</xref>). Moreover, shifts within these microbial populations can potentially influence our health (<xref ref-type="bibr" rid="B30">Racanelli et al., 2018</xref>). Research also indicates that the onset of chronic diseases is intricately linked to the symbiotic microbiota that reside within us, particularly anomalies in the gut microbiota&#x2019;s genome, which may lead to alterations in the human genome (<xref ref-type="bibr" rid="B32">Sampson et al., 2016</xref>). Furthermore, the diversity of microbial communities is closely associated with the incidence and progression of cardiovascular and neurodegenerative diseases, exerting a substantial impact on human health (<xref ref-type="bibr" rid="B36">Toya et al., 2020</xref>; <xref ref-type="bibr" rid="B5">Cryan and Dinan, 2012</xref>). Consequently, the deliberate modulation of the human microbiota&#x2019;s abundance presents a promising avenue for bolstering our disease resistance and enhancing global health (<xref ref-type="bibr" rid="B6">Desbonnet et al., 2010</xref>). Specifically, fine-tuning the equilibrium of the gut microbiota can aid in combating viral infections. Additionally, the supplementation of lactobacilli and bifidobacteria not only assists in pain relief but also plays a role in regulating emotions and reducing anxiety, highlighting the multifaceted benefits of these microbial allies (<xref ref-type="bibr" rid="B37">Turnbaugh et al., 2007</xref>).</p>
<p>Given the inextricable links between microbes and human health, scientists have embarked on numerous microbiome-based disease research projects since the 21st century (<xref ref-type="bibr" rid="B8">Gilbert et al., 2010</xref>; <xref ref-type="bibr" rid="B34">Sun et al., 2018</xref>). However, traditional wet-lab methods for detecting microbial-disease associations, such as culture-dependent and quantitative methods, are time-consuming, requiring extensive periods for cultivation, observation, and detection of a wide variety of microbes. These methods also suffer from a degree of arbitrariness and inherent risks. To surmount the limitations of biological research, the application of computational methods has been on the rise in recent years, spurred by rapid advancements in biotechnology. Additionally, experimentally validated databases linking microbes to diseases, such as HMDAD (<xref ref-type="bibr" rid="B26">Ma et al., 2017</xref>) and Disbiome (<xref ref-type="bibr" rid="B16">Janssens et al., 2018</xref>), have been established, providing invaluable data resources for scientific inquiry. These databases serve as a treasure trove of information, facilitating a deeper understanding of the complex interplay between microorganisms and human health. For instance, reference (<xref ref-type="bibr" rid="B28">Park et al., 2021</xref>) employs sophisticated computational approaches, including hierarchical long short-term memory (LSTM) networks and ensemble parsing models, to unravel the complex associations between microbes and diseases. Reference (<xref ref-type="bibr" rid="B24">Lu et al., 2023</xref>) employs a cutting-edge combination of autoencoders and graph convolutional networks to predict potential associations between microbes and diseases. Reference (<xref ref-type="bibr" rid="B3">Chen et al., 2024a</xref>) introduces a pioneering human microbiota disease association prediction model that is grounded in multi-view latent feature learning, and reference (<xref ref-type="bibr" rid="B10">Hu et al., 2023</xref>) introduces a microbe-disease association prediction model based on generative adversarial networks.</p>
<p>In this manuscript, we proposed an innovative forecasting framework named BANSMDA to infer possible microbe-disease associations by combining Bilinear Attention Networks (BAN) with sparse autoencoder (SAE). By fusing the nuanced feature interactions discerned by BAN (<xref ref-type="bibr" rid="B22">Liang et al., 2025</xref>) with the proficiency of SAE in feature dimensionality reduction and representation learning, BANSMDA is expected to deliver more precise and dependable predictions within the realm of microbe-disease associations. As depicted in <xref ref-type="fig" rid="F1">Figure 1</xref>, the key contributions of the BANSMDA encompass the following innovative aspects.<list list-type="simple">
<list-item>
<p>(1) A novel heterogeneous network <italic>B</italic> composed of microbes and diseases has been created by integrating the functional similarity network of microbes, the functional similarity network of diseases, and the existing microbe-disease associations.</p>
</list-item>
<list-item>
<p>(2) Utilize the BAN framework and the SEA framework respectively to derive node attribute representations within the heterogeneous network <italic>B.</italic>
</p>
</list-item>
<list-item>
<p>(3) Integrate the attribute representations of the two types of nodes, leveraging their multiple original features, to construct comprehensive node features within network <italic>B.</italic>
</p>
</list-item>
<list-item>
<p>(4) Calculate potential association scores for microbe-disease pairs using their feature matrices.</p>
</list-item>
</list>
</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The overall structure diagram of BANSMDA. <bold>(A)</bold> The heterogeneous microbe-disease network was established by amalgamating the microbe Gaussian similarity network, the disease Gaussian similarity network, and known associations between microbes and diseases. <bold>(B)</bold> Learning node representation based on BAN. <bold>(C)</bold> Learning node representation based on SAE. <bold>(D)</bold> Predicting the final scores of potential microbe-disease associations.</p>
</caption>
<graphic xlink:href="fgene-16-1618472-g001.tif">
<alt-text content-type="machine-generated">Diagram illustrating a computational model for analyzing disease-microbe interactions. Panel A depicts the creation of a heterogeneous network using disease and microbe Gaussian kernel similarities, shown with triangles and circles, and adjacency matrices. Panel B involves bilinear transformation and bilinear attention map. Panel C shows disease and microbe functional similarities, processed via stacked autoencoders. Panel D illustrates the concatenation of matrices leading to a prediction matrix. The flow includes various matrix operations and transformations highlighted by arrows.</alt-text>
</graphic>
</fig>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>Materials and methods</title>
<sec id="s2-1">
<title>Data sources</title>
<p>In this section, we would download known microbe-disease associations from public databases including HMDAD and Disbiome separately, among them, HMDAD (<xref ref-type="bibr" rid="B26">Ma et al., 2017</xref>) was compiled by Ma et al., in 2017, and after eliminating duplicate entries, we downloaded 450 distinct association pairs involving 39 diseases and 292 microbes. Besides, Disbiome (<xref ref-type="bibr" rid="B16">Janssens et al., 2018</xref>) was compiled by Janssens Y et al., in 2018, and after eliminating duplicate entries, we extracted 5,573 established associations between 240 diseases and 1,098 microbes. Detailed statistical information was shown in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>The specific statistical data.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Dataset</th>
<th align="left">Diseases</th>
<th align="left">Microbes</th>
<th align="left">Associations</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">HMDAD</td>
<td align="left">39</td>
<td align="left">292</td>
<td align="left">450</td>
</tr>
<tr>
<td align="left">Disbiome</td>
<td align="left">240</td>
<td align="left">1,098</td>
<td align="left">5,573</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec sec-type="methods" id="s3">
<title>Methods</title>
<sec id="s3-1">
<title>Microbe-disease incidence matrix</title>
<p>The incidence matrix <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is a square matrix used to represent a bipartite graph where one set of vertices represents diseases (<inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) and the other set represents microbes (<inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>). The matrix is structured such that the rows correspond to diseases and the columns correspond to microbes. Each entry <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> in the matrix <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> indicates the presence or absence of an interaction between disease <inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and microbe <inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Specifically, if <inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, it means there&#x2019;s a relationship between disease <inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and microbe <inline-formula id="inf10">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. If <inline-formula id="inf11">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, there&#x2019;s no known relationship <xref ref-type="disp-formula" rid="e1">Equation 1</xref> shows how it works:<disp-formula id="e1">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="" separators="|">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
</sec>
<sec id="s3-2">
<title>Microbe/Disease Gaussian kernel similarity</title>
<p>The similarity <inline-formula id="inf12">
<mml:math id="m13">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> between diseases <inline-formula id="inf13">
<mml:math id="m14">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf14">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, as measured by the Gaussian kernel, can be determined using <xref ref-type="disp-formula" rid="e1">Equation 2</xref>:<disp-formula id="e2">
<mml:math id="m16">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="italic">exp</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="|">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
</p>
<p>In the Gaussian kernel similarity, <inline-formula id="inf15">
<mml:math id="m17">
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="|">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> refers to the Euclidean distance between two diseases. The parameter <inline-formula id="inf16">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> as shown in <xref ref-type="disp-formula" rid="e3">Equation 3</xref>, is key in controlling how the distance affects the similarity measure:<disp-formula id="e3">
<mml:math id="m19">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
</mml:munderover>
</mml:mstyle>
<mml:msup>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="|">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
<p>Absolutely, the Gaussian kernel similarity <inline-formula id="inf17">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> can be applied to measure the similarity between microbes as well as shown in <xref ref-type="disp-formula" rid="e4">Equations 4</xref>, <xref ref-type="disp-formula" rid="e5">5</xref>:<disp-formula id="e4">
<mml:math id="m21">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="italic">exp</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="|">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
<disp-formula id="e5">
<mml:math id="m22">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:munderover>
</mml:mstyle>
<mml:msup>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="|">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
</sec>
<sec id="s3-3">
<title>Microbe/disease functional similarity</title>
<p>Under the premise that diseases with similar characteristics are likely to interact with analogous genes (<xref ref-type="bibr" rid="B39">Wei and Liu, 2020</xref>; <xref ref-type="bibr" rid="B40">Xu and Li, 2006</xref>), we proceeded to calculate the functional similarity of diseases based on the functional associations among genes implicated in these diseases. The recently unveiled HumanNet v2.0 database serves as a potent tool for efficiently accessing gene interactions (<xref ref-type="bibr" rid="B12">Hwang et al., 2019</xref>; <xref ref-type="bibr" rid="B23">Long et al., 2021</xref>), with each interaction being accompanied by a log-likelihood score (LLS). This LLS quantifies the likelihood of a functional connection existing between genes. For a pair of diseases, denoted as <inline-formula id="inf18">
<mml:math id="m23">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf19">
<mml:math id="m24">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, we initially extracted their respective associated gene sets, denoted as <inline-formula id="inf20">
<mml:math id="m25">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf21">
<mml:math id="m26">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Here, <inline-formula id="inf22">
<mml:math id="m27">
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the count of genes within set <inline-formula id="inf23">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf24">
<mml:math id="m29">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the count of genes within set <inline-formula id="inf25">
<mml:math id="m30">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. We then defined the functional association between a single gene g and a gene set <inline-formula id="inf26">
<mml:math id="m31">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> using <xref ref-type="disp-formula" rid="e6">Equation 6</xref>:<disp-formula id="e6">
<mml:math id="m32">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="italic">max</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>where <inline-formula id="inf27">
<mml:math id="m33">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and the functional similarity score between genes, represented by <inline-formula id="inf28">
<mml:math id="m34">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, is defined as shown in <xref ref-type="disp-formula" rid="e7">Equation 7</xref>:<disp-formula id="e7">
<mml:math id="m35">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="" separators="|">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mo>&#x2260;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>where <inline-formula id="inf29">
<mml:math id="m36">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> represents the normalized <inline-formula id="inf30">
<mml:math id="m37">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of genes, which is defined as shown in <xref ref-type="disp-formula" rid="e8">Equation 8</xref>:<disp-formula id="e8">
<mml:math id="m38">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mi mathvariant="italic">min</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mi mathvariant="italic">min</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>where <inline-formula id="inf31">
<mml:math id="m39">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf32">
<mml:math id="m40">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mi mathvariant="italic">min</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denote the maximum and minimum <inline-formula id="inf33">
<mml:math id="m41">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> in the HumanNet database, respectively. Ultimately, we articulated the disease functional similarity using <xref ref-type="disp-formula" rid="e10">Equation 10</xref>:<disp-formula id="e9">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>U</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>
</p>
<p>In terms of microbe functional similarity, we employed the methodology advanced by <xref ref-type="bibr" rid="B17">Kamneva. (2017)</xref>. To ascertain the functional similarity among microbes. To meticulously determine the functional similarity for any given pair of microbes, we initially sourced the protein-protein functional association network from the STRING v11 database (<xref ref-type="bibr" rid="B35">Szklarczyk et al., 2019</xref>). Utilizing the similarity scores derived therefrom, we constructed a microbe functional similarity matrix, <inline-formula id="inf34">
<mml:math id="m43">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>U</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, wherein each entry <inline-formula id="inf35">
<mml:math id="m44">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>U</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the degree of similarity between microbe <inline-formula id="inf36">
<mml:math id="m45">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and microbe <inline-formula id="inf37">
<mml:math id="m46">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</sec>
<sec id="s3-4">
<title>Constructing the heterogeneous network <inline-formula id="inf38">
<mml:math id="m47">
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</title>
<p>By fusing the microbe-disease adjacency matrix with the disease Gaussian kernel similarity matrix and the microbe Gaussian kernel similarity matrix, as shown in <xref ref-type="disp-formula" rid="e10">Equation 10</xref>, we have crafted a heterogeneous network:<disp-formula id="e10">
<mml:math id="m48">
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd>
<mml:mi>E</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msup>
<mml:mi>E</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
</mml:mtd>
<mml:mtd>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>where <inline-formula id="inf39">
<mml:math id="m49">
<mml:mrow>
<mml:msup>
<mml:mi>E</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> represents s transposition.</p>
</sec>
<sec id="s3-5">
<title>BAN model</title>
<p>Bilinear Attention Networks (BAN), introduced by Kim in 2018, are composed of a central component known as the bilinear attention mechanism, which is designed to learn the distribution of attention by considering the bilinear interactions between input channels. This network employs two critical techniques to enhance feature interaction and manage complex data relationships: bilinear transformation and attention mechanisms. Bilinear transformation, which uses a weight matrix and an additive bias to process input features, is adept at revealing nuanced relationships within complex datasets, providing a robust framework for analyzing interactions. Its formula can be expressed as:<disp-formula id="e11">
<mml:math id="m50">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mi>a</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mi>H</mml:mi>
<mml:mi>a</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>
</p>
<p>In <xref ref-type="disp-formula" rid="e11">Equation 11</xref>, <inline-formula id="inf40">
<mml:math id="m51">
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the input vector to the BAN, <inline-formula id="inf41">
<mml:math id="m52">
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is a trainable weight matrix, <inline-formula id="inf42">
<mml:math id="m53">
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the bias term, and <inline-formula id="inf43">
<mml:math id="m54">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the resulting output vector from the BAN. The forward propagation process of the BAN can be described as follows:<disp-formula id="e12">
<mml:math id="m55">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mi>x</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(12)</label>
</disp-formula>
<disp-formula id="e13">
<mml:math id="m56">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(13)</label>
</disp-formula>
<disp-formula id="e14">
<mml:math id="m57">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mi>x</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(14)</label>
</disp-formula>
</p>
<p>In <xref ref-type="disp-formula" rid="e12">Equation 12</xref>, H<sub>1</sub> denotes the weight matrix from the input layer to the hidden layer, b<sub>1</sub> represents the bias vector of the hidden layer, and x, defined in <xref ref-type="disp-formula" rid="e11">Equation 11</xref>, corresponds to the input vector. In <xref ref-type="disp-formula" rid="e13">Equation 13</xref>, H<sub>2</sub> and b<sub>2</sub> are the weight matrix from the hidden layer to the output layer and the output layer&#x2019;s bias vector, respectively. By substituting <xref ref-type="disp-formula" rid="e12">Equation 12</xref> into <xref ref-type="disp-formula" rid="e13">Equation 13</xref>, we derive the final output y and a streamlined forward propagation formula, <xref ref-type="disp-formula" rid="e14">Equation 14</xref>, which explicitly formalizes the computation process. The activation function used within the network is ReLU, as defined in <xref ref-type="disp-formula" rid="e15">formula 15</xref>, which introduces non-linearity to the model and helps in learning complex patterns. The feature vector that undergoes processing by this ReLU activation function is referred to as <inline-formula id="inf44">
<mml:math id="m58">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.<disp-formula id="e15">
<mml:math id="m59">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="" separators="|">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>z</mml:mi>
<mml:mo>&#x3e;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(15)</label>
</disp-formula>
</p>
<p>By feeding the heterogeneous network <inline-formula id="inf45">
<mml:math id="m60">
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> into the BAN, a low-dimensional matrix <inline-formula id="inf46">
<mml:math id="m61">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is produced, where indices <inline-formula id="inf47">
<mml:math id="m62">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf48">
<mml:math id="m63">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denote the disease nodes and microbial nodes, respectively. During the computation, the cross-entropy function is utilized for optimization purposes.</p>
</sec>
<sec id="s3-6">
<title>SAE model</title>
<p>To effectively capture both the local and global topological intrinsic features of nodes, we have further implemented an enhanced version of Random Walk with Restart (RWR) on the <inline-formula id="inf49">
<mml:math id="m64">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>U</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The RWR is defined as shown in <xref ref-type="disp-formula" rid="e16">Equation 16</xref>:<disp-formula id="e16">
<mml:math id="m65">
<mml:mrow>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c6;</mml:mi>
<mml:mi>X</mml:mi>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>l</mml:mi>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(16)</label>
</disp-formula>
</p>
<p>In <xref ref-type="disp-formula" rid="e16">Equation 16</xref>, <inline-formula id="inf50">
<mml:math id="m66">
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the restart probability. <inline-formula id="inf51">
<mml:math id="m67">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> signifies the transition probability matrix, and <inline-formula id="inf52">
<mml:math id="m68">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the initial probability vector for node <inline-formula id="inf53">
<mml:math id="m69">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. The definition of the initial probability vector is as shown in <xref ref-type="disp-formula" rid="e17">Equation 17</xref>:<disp-formula id="e17">
<mml:math id="m70">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="" separators="|">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(17)</label>
</disp-formula>
</p>
<p>Following the aforementioned RWR process, it becomes evident that we can derive a new matrix <inline-formula id="inf54">
<mml:math id="m71">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Similarly, after applying the improved RWR on <inline-formula id="inf55">
<mml:math id="m72">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>U</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, we can obtain a new matrix <inline-formula id="inf56">
<mml:math id="m73">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>Consequently, by amalgamating all the matrices <italic>E</italic>, <inline-formula id="inf57">
<mml:math id="m74">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>U</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf58">
<mml:math id="m75">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, as shown in <xref ref-type="disp-formula" rid="e18">Equation 18</xref>: we can construct a new disease matrix <inline-formula id="inf59">
<mml:math id="m76">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>:<disp-formula id="e18">
<mml:math id="m77">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mo>;</mml:mo>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>U</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(18)</label>
</disp-formula>
</p>
<p>Similarly, by integrating <italic>E</italic>, <inline-formula id="inf60">
<mml:math id="m78">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>U</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf61">
<mml:math id="m79">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> a new microbial matrix <inline-formula id="inf62">
<mml:math id="m80">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> can be obtained as shown in <xref ref-type="disp-formula" rid="e19">Equation 19</xref>:<disp-formula id="e19">
<mml:math id="m81">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>E</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mo>;</mml:mo>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>U</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(19)</label>
</disp-formula>
</p>
<p>Then, use the above two matrices as inputs to the sparse autoencoder (SAE). SAE excel in feature extraction and dimensionality reduction, enabling them to distill crucial features from intricate microbial data and reduce its complexity, which is particularly valuable for managing high-dimensional biomedical data. Additionally, the incorporation of sparsity penalties within SAE helps to constrain the activation of neurons in the hidden layer, thereby enhancing the model&#x2019;s feature extraction capabilities. This sparse representation not only boosts predictive accuracy but also contributes to the interpretability of the model. SAE consists of the following steps:</p>
<p>Encoding process: Input data <italic>x</italic> is converted into a hidden layer representation <italic>h</italic> through an encoder, and <inline-formula id="inf63">
<mml:math id="m82">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is used as a non-linear activation function after linear transformation. The specific formula is as shown in <xref ref-type="disp-formula" rid="e20">Equation 20</xref>:<disp-formula id="e20">
<mml:math id="m83">
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>x</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(20)</label>
</disp-formula>
</p>
<p>Among them, <inline-formula id="inf64">
<mml:math id="m84">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the weight matrix of the encoder, and <inline-formula id="inf65">
<mml:math id="m85">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the bias term.</p>
<p>Decoding process: The hidden layer representation <italic>h</italic> is reconstructed back to the original data <inline-formula id="inf66">
<mml:math id="m86">
<mml:mrow>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> through the decoder. The definition of x&#x2032; is shown in <xref ref-type="disp-formula" rid="e21">Equation 21</xref>. This process is also a linear transformation followed by a nonlinear activation function <inline-formula id="inf67">
<mml:math id="m87">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>:<disp-formula id="e21">
<mml:math id="m88">
<mml:mrow>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>x</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(21)</label>
</disp-formula>
</p>
<p>Among them, <inline-formula id="inf68">
<mml:math id="m89">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the weight matrix of the decoder, and <inline-formula id="inf69">
<mml:math id="m90">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the bias term.</p>
<p>Refactoring loss: Refactoring error is an indicator that measures the difference between the reconstructed data <inline-formula id="inf70">
<mml:math id="m91">
<mml:mrow>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and the original data <inline-formula id="inf71">
<mml:math id="m92">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, represented by the binary cross entropy (BCE) loss function. The specific form is shown in <xref ref-type="disp-formula" rid="e22">Equation 22</xref>:<disp-formula id="e22">
<mml:math id="m93">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>B</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>i</mml:mi>
</mml:munder>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi mathvariant="italic">log</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi mathvariant="italic">log</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(22)</label>
</disp-formula>
</p>
<p>Sparsity loss: To introduce sparsity, SAE adds a sparsity penalty term to the loss function, which is typically based on L1 regularization. The specific form is shown in <xref ref-type="disp-formula" rid="e23">Equation 23</xref>:<disp-formula id="e23">
<mml:math id="m94">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>j</mml:mi>
</mml:munder>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(23)</label>
</disp-formula>where <inline-formula id="inf72">
<mml:math id="m95">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the regularization coefficient and <inline-formula id="inf73">
<mml:math id="m96">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the activation value of the hidden layer.</p>
<p>Total loss function: The total loss function, which is the sum of the reconstruction loss and the sparsity loss, serves as the objective function for optimization during the training process. It can be articulated as shown in <xref ref-type="disp-formula" rid="e24">Equation 24</xref>:<disp-formula id="e24">
<mml:math id="m97">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(24)</label>
</disp-formula>
</p>
<p>Consequently, by feeding the disease matrix <inline-formula id="inf74">
<mml:math id="m98">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the microbe matrix <inline-formula id="inf75">
<mml:math id="m99">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> into the SAE individually, we can derive matrices <inline-formula id="inf76">
<mml:math id="m100">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf77">
<mml:math id="m101">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, respectively.</p>
</sec>
<sec id="s3-7">
<title>Microbe/disease feature matrix</title>
<p>Based on the processing results of BAN and SAE models, by integrating the disease matrix <inline-formula id="inf78">
<mml:math id="m102">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf79">
<mml:math id="m103">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf80">
<mml:math id="m104">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf81">
<mml:math id="m105">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>U</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the adjacency matrix <italic>E</italic>, inspired by <xref ref-type="bibr" rid="B41">Xuan et al. (2020)</xref> (<xref ref-type="bibr" rid="B23">Long et al., 2021</xref>), we can construct a new disease feature matrix <inline-formula id="inf82">
<mml:math id="m106">
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> as shown in <xref ref-type="disp-formula" rid="e25">Equation 25</xref>:<disp-formula id="e25">
<mml:math id="m107">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>E</mml:mi>
<mml:mo>;</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>U</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>E</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(25)</label>
</disp-formula>
</p>
<p>Similarly, integrating the microbial matrix <inline-formula id="inf83">
<mml:math id="m108">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf84">
<mml:math id="m109">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf85">
<mml:math id="m110">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf86">
<mml:math id="m111">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>U</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the adjacency matrix <italic>E</italic>, we can construct a new microbe feature matrix <inline-formula id="inf87">
<mml:math id="m112">
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> as shown in <xref ref-type="disp-formula" rid="e26">Equation 26</xref>:<disp-formula id="e26">
<mml:math id="m113">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mrow>
<mml:msup>
<mml:mi>E</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mo>;</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msup>
<mml:mi>E</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mo>;</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>U</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(26)</label>
</disp-formula>
</p>
</sec>
<sec id="s3-8">
<title>Calculating the final predicted scores of potential microbe-disease associations</title>
<p>The dot product of two vectors serves as an effective mechanism for modeling interactions, highlighting the shared aspects of these interactions while diminishing the distinct information they might carry. Consequently, for any given disease <inline-formula id="inf88">
<mml:math id="m114">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and microbe <inline-formula id="inf89">
<mml:math id="m115">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, we can determine their predicted association scores by computing the inner product of their feature representations, as shown in <xref ref-type="disp-formula" rid="e27">Equation 27</xref>:<disp-formula id="e27">
<mml:math id="m116">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#xb7;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(27)</label>
</disp-formula>
</p>
</sec>
<sec id="s3-9">
<title>Experiments and results</title>
<p>In this part, we began by conducting a sensitivity analysis of crucial parameters to improve the model&#x2019;s effectiveness. Next, we chose six state-of-the-art techniques to benchmark against BANSMDA. Additionally, to confirm the model&#x2019;s reliability, we selected two exemplary microbes and diseases for evaluation.</p>
</sec>
<sec id="s3-10">
<title>Parameter sensitivity analysis</title>
<p>Considering the actual situation of the model, we identified and analyzed four parameters that have a significant impact on the final prediction results. These include the <italic>L</italic>2 regularization parameter <inline-formula id="inf90">
<mml:math id="m117">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3bb;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>,which is named <inline-formula id="inf91">
<mml:math id="m118">
<mml:mrow>
<mml:msub>
<mml:mi>l</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>,in the BAN model, <inline-formula id="inf92">
<mml:math id="m119">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3c6;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> in the RWR of <xref ref-type="disp-formula" rid="e14">formula (14)</xref>, the learning rate <inline-formula id="inf93">
<mml:math id="m120">
<mml:mrow>
<mml:msub>
<mml:mi>l</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the average activation &#x3c1; in the SEA model.</p>
<p>In this section, our objective is to determine the optimal settings while maintaining the separation of the training and testing datasets. Specifically, The range of values for <inline-formula id="inf94">
<mml:math id="m121">
<mml:mrow>
<mml:msub>
<mml:mi>l</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is set to <inline-formula id="inf95">
<mml:math id="m122">
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mn>0.0001</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.0005</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.001</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.005</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.01</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.05</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>. The range of values for <inline-formula id="inf96">
<mml:math id="m123">
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is <inline-formula id="inf97">
<mml:math id="m124">
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mn>0.1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.3</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.4</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.5</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.6</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.7</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.8</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.9</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>. The range of values for <inline-formula id="inf98">
<mml:math id="m125">
<mml:mrow>
<mml:msub>
<mml:mi>l</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is <inline-formula id="inf99">
<mml:math id="m126">
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mn>0.0001</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.0005</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.001</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.005</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.01</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.05</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>. The range of values for &#x3c1; is {0.001,0.005,0.01,0.05,0.1}.Subsequently, we employed a 5 - fold cross - validation (CV) method to evaluate the area under the receiver operating characteristic curve (AUC) and the area under the precision - recall curve for the parameter configurations. In the parameter validation experiment, we first set one parameter, e.g., <italic>l</italic>
<sub>2</sub>_lamda, to a fixed value. Then, in each epoch, we changed the value of one of the other parameters. After that, we collected all AUC and AUPR values under such circumstances. Finally, we calculated the average of the obtained AUC and AUPR values respectively, and used them as the final results for that particular parameter setting. In the 5-fold CV experiment, we first randomly assigned 80% of the dataset, including both identified and unidentified associations, to the training set, with the remaining 20% reserved as the independent test set. We then divided the training set into five equal-sized subsets to perform 5-fold cross-validation. Using the HMDAD dataset, we independently conducted five cross-validation runs. After the cross-validation was completed, the model&#x2019;s performance was evaluated on different subsets of the training set. Finally, we used the pre-defined independent test set to assess the model&#x2019;s final performance. As shown in <xref ref-type="fig" rid="F2">Figure 2</xref>, the model achieves the best performance when the parameter value are configured as follows: <inline-formula id="inf100">
<mml:math id="m127">
<mml:mrow>
<mml:msub>
<mml:mi>l</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.01</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c6;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.4</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>l</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.005</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">&#x3c1;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.005</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>AUC and AUPR values on different parameter sensitivity analysis. <bold>(A)</bold> The AUC and AUPR values on different <inline-formula id="inf101">
<mml:math id="m128">
<mml:mrow>
<mml:msub>
<mml:mi>l</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <bold>(B)</bold> The AUC and AUPR values on different <inline-formula id="inf102">
<mml:math id="m129">
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <bold>(C)</bold> The AUC and AUPR values on different <inline-formula id="inf103">
<mml:math id="m130">
<mml:mrow>
<mml:msub>
<mml:mi>l</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>,<bold>(D)</bold> The AUC and AUPR values on different &#x3c1;.</p>
</caption>
<graphic xlink:href="fgene-16-1618472-g002.tif">
<alt-text content-type="machine-generated">Graphs display AUC and AUPR values across different parameters. (a) AUC remains stable, while AUPR varies between 0.61 and 0.65. (b) AUC is steady, AUPR peaks at 0.63, then declines. (c) AUC is consistent, AUPR ranges from 0.61 to 0.68. (d) AUC is constant around 9.8, AUPR fluctuates slightly around 6.3.</alt-text>
</graphic>
</fig>
</sec>
</sec>
<sec id="s4">
<title>Comparison with advanced methods</title>
<p>To further validate the predictive accuracy of BANSMDA, this section includes a comparative analysis with six prominent and competitive approaches. In the experiment, we employed the same 5-fold cross-validation technique on the HMDAD dataset for each method to ensure fair and consistent comparisons.</p>
<p>MOSFL-LNP (<xref ref-type="bibr" rid="B1">Chen et al., 2024b</xref>): This method involves preprocessing a similarity matrix, integrating low-order and high-order learning, optimizing and solving the associated equations, and finally normalizing the predicted association score matrix.</p>
<p>LRLSHMDA (<xref ref-type="bibr" rid="B38">Wang et al., 2017</xref>): This is a semi-supervised computational model utilizes the Gaussian interaction profile kernel similarity and the Laplacian regularized least squares classifier to predict the Potential Microbe-Disease associations.</p>
<p>BIRWMP (<xref ref-type="bibr" rid="B33">Shen et al., 2018</xref>): This method is a computational model based on bidirectional random walk, which predicts potential microbe-disease associations by conducting multipath analysis on microbe and disease similarity networks.</p>
<p>NTSHMDA (<xref ref-type="bibr" rid="B25">Luo and Long, 2020</xref>): A computational model based on neighborhood topology similarity, which is used for predicting the potential microbe-disease associations.</p>
<p>KATZHMDA (<xref ref-type="bibr" rid="B2">Chen et al., 2016</xref>): This is a computational method based on the KATZ algorithm, which calculates the association between microbes and diseases by considering the number and length of paths connecting two nodes in a microbe-disease heterogeneous network.</p>
<p>HMDA_Pred (<xref ref-type="bibr" rid="B7">Fan et al., 2020</xref>): This method is a novel computer model based on multi-data integration and network consistency projection, used for calculating the associations between microbes and diseases.</p>
<p>We assessed the performance of these models under their default parameter settings and five-fold cross-validation. By leveraging the HMDAD dataset, We comprehensively evaluated our model using four key metrics: AUC, AUPR, Accuracy, and F - score, all of which were obtained through averaging over five - fold cross - validation. The results are detailed in <xref ref-type="table" rid="T2">Table 2</xref> and visualized in <xref ref-type="fig" rid="F3">Figure 3</xref>, demonstrating the superior predictive performance and accuracy of the BANSMDA model compared to other methods.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Results of the compared methods.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Methods</th>
<th align="center">AUC</th>
<th align="center">AUPR</th>
<th align="center">Accuracy</th>
<th align="center">F1-score</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">BANSMDA</td>
<td align="center">
<bold>0.98011 &#xb1; 0.0023</bold>
</td>
<td align="center">
<bold>0.64431 &#xb1; 0.0032</bold>
</td>
<td align="center">0.91245</td>
<td align="center">
<bold>0.47276</bold>
</td>
</tr>
<tr>
<td align="left">MOSFL-LNP</td>
<td align="center">0.93062 &#xb1; 0.0012</td>
<td align="center">0.62327 &#xb1; 0.0013</td>
<td align="center">
<bold>0.91269</bold>
</td>
<td align="center">0.44111</td>
</tr>
<tr>
<td align="left">LRLSHMDA</td>
<td align="center">0.85827 &#xb1; 0.0035</td>
<td align="center">0.36511 &#xb1; 0.0312</td>
<td align="center">0.84887</td>
<td align="center">0.27597</td>
</tr>
<tr>
<td align="left">BIRWMP</td>
<td align="center">0.85669 &#xb1; 0.0015</td>
<td align="center">0.36395 &#xb1; 0.0156</td>
<td align="center">0.91008</td>
<td align="center">0.36711</td>
</tr>
<tr>
<td align="left">NTSHMDA</td>
<td align="center">0.77131 &#xb1; 0.0020</td>
<td align="center">0.0768 &#xb1; 0.0153</td>
<td align="center">0.73452</td>
<td align="center">0.08913</td>
</tr>
<tr>
<td align="left">KATZHMDA</td>
<td align="center">0.83502 &#xb1; 0.0034</td>
<td align="center">0.23771 &#xb1; 0.0048</td>
<td align="center">0.88393</td>
<td align="center">0.08701</td>
</tr>
<tr>
<td align="left">HMDA_PRED</td>
<td align="center">0.91875 &#xb1; 0.0026</td>
<td align="center">0.21276 &#xb1; 0.0074</td>
<td align="center">0.80877</td>
<td align="center">0.18082</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bold values are the maximum values of each column.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>AUC and AUPR curves of six competitive methods based on the HMDAD dataset. <bold>(A)</bold> AUC curves of six competitive methods based on the HMDAD dataset. <bold>(B)</bold> AUPR curves of six competitive methods based on the HMDAD dataset.</p>
</caption>
<graphic xlink:href="fgene-16-1618472-g003.tif">
<alt-text content-type="machine-generated">Multi-model ROC and Precision-Recall curve comparisons for six methods based on the HMDAD database. Top graph shows ROC curves with BANSFDA achieving the highest AUC of 0.98. Bottom graph shows Precision-Recall curves with BANSFDA having an AUPR of 0.64.</alt-text>
</graphic>
</fig>
<p>As detailed in <xref ref-type="table" rid="T2">Table 2</xref>, BANSMDA exhibits outstanding performance across three evaluation metrics: AUC, AUPR, and F1-Score. Specifically, compared to the MOSFL-LNP model, BANSMDA achieves a 5.31% improvement in AUC and a 3.37% improvement in AUPR. While slightly inferior to MOSFL-LNP in terms of Accuracy, the difference is negligible (only 0.00024). Collectively, these results confirm that BANSMDA is a highly efficient model for predicting microbe-disease associations.</p>
<p>As showen in <xref ref-type="fig" rid="F4">Figure 4</xref>, we visualize the learned encoder weights and hidden layer activations to demonstrate the interpretability of the sparse representations. The results show that each hidden unit specializes in specific input features and exhibits sparse, selective activation patterns across samples, supporting our claim that the SAE produces interpretable representations.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>The encoder weight heat map of SAE.</p>
</caption>
<graphic xlink:href="fgene-16-1618472-g004.tif">
<alt-text content-type="machine-generated">Heatmap showing encoder layer weights with input features on the x-axis and hidden units on the y-axis. The color gradient ranges from blue at -2.0 to red at 2.0, indicating weight magnitude.</alt-text>
</graphic>
</fig>
<p>
<xref ref-type="fig" rid="F4">Figure 4</xref> shows the heatmap of the encoder weights after training. We observe that each hidden unit (row) assigns substantial weights to only a small subset of input features (columns), with most weights close to zero. This suggests that each hidden neuron specializes in detecting specific feature patterns in the input rather than responding indiscriminately to all dimensions. Such specialization is desirable because it allows us to attribute meaningful feature combinations to individual hidden units, enhancing interpretability of the learned representation.</p>
<p>
<xref ref-type="fig" rid="F5">Figure 5</xref> depicts the hidden-layer activations at epoch 100. These heatmaps illustrate the activations of 32 hidden units (x-axis) across a batch of input samples (y-axis). We note two key observations: (1) the activations are sparse&#x2014;for each sample, only a few hidden units exhibit high activation values, while most remain near zero; (2) the activation patterns are distinct and consistent&#x2014;different hidden units activate for different samples, and the same unit responds consistently to similar samples across epochs. These findings indicate that the hidden layer learns a set of specialized, non-redundant detectors that respond selectively to specific input patterns.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>The encoder weight heat map of SAE.</p>
</caption>
<graphic xlink:href="fgene-16-1618472-g005.tif">
<alt-text content-type="machine-generated">Heatmap illustrating hidden layer activations at epoch one hundred. The horizontal axis represents hidden units, and the vertical axis represents samples. Colors range from yellow to dark blue, indicating activation levels from low to high.</alt-text>
</graphic>
</fig>
<sec id="s4-1">
<title>Case study</title>
<p>To rigorously assess the predictive performance of the BANSMDA model, we conducted case study validation using two prevalent diseases (Asthma and Colorectal carcinoma) and two clinically significant microbial genera (<italic>Escherichia</italic> and <italic>Bacteroides</italic>).</p>
<p>Asthma is a common chronic inflammatory disease originating in the lower airways, characterized by persistent airway inflammation (<xref ref-type="bibr" rid="B29">Polyxeni et al., 2021</xref>). Clinical manifestations include recurrent episodes of wheezing, coughing, chest tightness, and shortness of breath, typically exacerbated during nocturnal or early morning periods (<xref ref-type="bibr" rid="B15">James, 2015</xref>). A growing body of evidence from multiple references underscores a significant correlation between the pathogenesis of asthma and specific microbiota, such as <italic>Helicobacter pylori</italic> (<xref ref-type="bibr" rid="B43">Zhi et al., 2021</xref>), Proteobacteria (<xref ref-type="bibr" rid="B20">Kian, 2017</xref>), and Bacteroidetes (<xref ref-type="bibr" rid="B4">Chie, 2023</xref>). Based on the predictive scores, microbes associated with asthma were ranked in descending order according to their respective scores. As illustrated in <xref ref-type="table" rid="T3">Table 3</xref>, among the top 20 predicted microbes associated with Asthma, 19 have been confirmed by existing research indexed in PubMed.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Top 20 Asthma-associated candidate microbes on HMDAD.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Microbe</th>
<th align="left">Evidence</th>
<th align="left">Microbe</th>
<th align="left">Evidence</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<italic>Helicobacter pylori</italic>
</td>
<td align="left">PMID:33080611</td>
<td align="left">
<italic>Lactobacillus</italic>
</td>
<td align="left">PMID:33882482</td>
</tr>
<tr>
<td align="left">Proteobacteria</td>
<td align="left">PMID:29161086</td>
<td align="left">Burkholderia</td>
<td align="left">PMID:15297563</td>
</tr>
<tr>
<td align="left">Bacteroidetes</td>
<td align="left">PMID:38155860</td>
<td align="left">Faecalibacterium prausnitzii</td>
<td align="left">PMID:33709404</td>
</tr>
<tr>
<td align="left">Prevotella</td>
<td align="left">PMID:28542929</td>
<td align="left">Coxiellaceae</td>
<td align="left">NA</td>
</tr>
<tr>
<td align="left">
<italic>Staphylococcus</italic>
</td>
<td align="left">PMID:31980492</td>
<td align="left">
<italic>Clostridium</italic>
</td>
<td align="left">PMID:35349868</td>
</tr>
<tr>
<td align="left">
<italic>Haemophilus</italic>
</td>
<td align="left">PMID:37287344</td>
<td align="left">Clostridiales</td>
<td align="left">PMID:24798552</td>
</tr>
<tr>
<td align="left">Sphingomonadaceae</td>
<td align="left">PMID:21194740</td>
<td align="left">
<italic>Pseudomonas</italic>
</td>
<td align="left">PMID:36167555</td>
</tr>
<tr>
<td align="left">Comamonadaceae</td>
<td align="left">PMID:27433177</td>
<td align="left">Betaproteobacteria</td>
<td align="left">PMID:23053501</td>
</tr>
<tr>
<td align="left">
<italic>Clostridium difficile</italic>
</td>
<td align="left">PMID:32487252</td>
<td align="left">Propionibacterium</td>
<td align="left">PMID:29447223</td>
</tr>
<tr>
<td align="left">Oxalobacteraceae</td>
<td align="left">PMID:21194740</td>
<td align="left">Gammaproteobacteria</td>
<td align="left">PMID:27889361</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Notes: The top 10 microbes are listed in the first column, while the top 11&#x2013;20 microbes are listed in the third column.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Colorectal carcinoma ranks as the third most common cancer globally (<xref ref-type="bibr" rid="B4">Chie, 2023</xref>; <xref ref-type="bibr" rid="B13">In&#xe9;s et al., 2017</xref>). The gut microbiota is intricately involved in its development, with ecological imbalances capable of inducing colorectal carcinoma through chronic inflammatory pathways. Key bacterial taxa implicated in this multifaceted process include <italic>Clostridium</italic> (<xref ref-type="bibr" rid="B11">Hui et al., 2023</xref>), <italic>Bacteroides</italic> (<xref ref-type="bibr" rid="B42">Yasutoshi et al., 2024</xref>), and Enterobacteriaceae (<xref ref-type="bibr" rid="B31">Rashmi et al., 2016</xref>). As illustrated in <xref ref-type="table" rid="T4">Table 4</xref>, all of the top 20 predicted microbes associated with Colorectal carcinoma have been confirmed by existing studies in PubMed.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Top 20 Colorectal carcinoma-associated candidate microbes on HMDAD.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Microbe</th>
<th align="left">Evidence</th>
<th align="left">Microbe</th>
<th align="left">Evidence</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Bacteroidetes</td>
<td align="left">PMID:28643627</td>
<td align="left">Clostridia</td>
<td align="left">PMID:36941257</td>
</tr>
<tr>
<td align="left">Firmicutes</td>
<td align="left">PMID:37069401</td>
<td align="left">
<italic>Haemophilus</italic>
</td>
<td align="left">PMID:24725844</td>
</tr>
<tr>
<td align="left">Prevotella</td>
<td align="left">PMID:35935780</td>
<td align="left">
<italic>Clostridium</italic> coccoides</td>
<td align="left">PMID:28661219</td>
</tr>
<tr>
<td align="left">
<italic>Bacteroides</italic>
</td>
<td align="left">PMID:38266708</td>
<td align="left">
<italic>Fusobacterium</italic>
</td>
<td align="left">PMID:26311717</td>
</tr>
<tr>
<td align="left">Proteobacteria</td>
<td align="left">PMID:27721244</td>
<td align="left">
<italic>Staphylococcus</italic>
</td>
<td align="left">PMID:28506660</td>
</tr>
<tr>
<td align="left">
<italic>Helicobacter pylori</italic>
</td>
<td align="left">PMID:31368293</td>
<td align="left">Lachnospiraceae</td>
<td align="left">PMID:36893736</td>
</tr>
<tr>
<td align="left">
<italic>Clostridium difficile</italic>
</td>
<td align="left">PMID:26691472</td>
<td align="left">Enterobacteriaceae</td>
<td align="left">PMID:27015276</td>
</tr>
<tr>
<td align="left">
<italic>Staphylococcus aureus</italic>
</td>
<td align="left">PMID:25495422</td>
<td align="left">
<italic>Fusobacterium</italic> nucleatum</td>
<td align="left">PMID:37130518</td>
</tr>
<tr>
<td align="left">
<italic>Lactobacillus</italic>
</td>
<td align="left">PMID:35808840</td>
<td align="left">
<italic>Clostridium</italic>
</td>
<td align="left">PMID:36941257</td>
</tr>
<tr>
<td align="left">Actinobacteria</td>
<td align="left">PMID:27015276</td>
<td align="left">Veillonella</td>
<td align="left">PMID:37519587</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Notes: The top 10 microbes are listed in the first column, while the top 11&#x2013;20 microbes are listed in the third column.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>
<italic>Escherichia</italic> is a bacterium that embodies a dual identity, capable of functioning as both a symbiotic microbe and a pathogenic agent within the host&#x2019;s body (<xref ref-type="bibr" rid="B27">Olivier et al., 2010</xref>). Recent research has demonstrated that specific strains of <italic>Escherichia</italic> are capable of causing a range of intestinal infections, including diarrhea and enteritis (<xref ref-type="bibr" rid="B14">James, 2005</xref>). Moreover, <italic>Escherichia</italic> can extend its pathogenicity beyond the gut to cause extraintestinal infections through mechanisms like fecal contamination or hematogenous dissemination (<xref ref-type="bibr" rid="B19">Kevin et al., 2019</xref>). As illustrated in <xref ref-type="table" rid="T5">Table 5</xref>, among the top 20 predicted diseases associated with <italic>Escherichia</italic>, 19 have been confirmed by existing research indexed in PubMed.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Top 20 Escherichia-associated candidate diseases on HMDAD.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Disease</th>
<th align="left">Evidence</th>
<th align="left">Disease</th>
<th align="left">Evidence</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Type 1 diabetes</td>
<td align="left">PMID:36037202</td>
<td align="left">Psoriasis</td>
<td align="left">PMID:33924414</td>
</tr>
<tr>
<td align="left">Liver cirrhosis</td>
<td align="left">PMID:33466521</td>
<td align="left">Colorectal carcinoma</td>
<td align="left">PMID:28106826</td>
</tr>
<tr>
<td align="left">Irritable bowel syndrome (IBS)</td>
<td align="left">PMID:32966000</td>
<td align="left">Atopic dermatitis</td>
<td align="left">PMID:36335456</td>
</tr>
<tr>
<td align="left">Bacterial Vaginosis</td>
<td align="left">PMID:38751998</td>
<td align="left">Systemic inflammatory response syndrome</td>
<td align="left">PMID:34997430</td>
</tr>
<tr>
<td align="left">Periodontal</td>
<td align="left">PMID:33830141</td>
<td align="left">Obesity</td>
<td align="left">PMID:34385401</td>
</tr>
<tr>
<td align="left">Necrotizing Enterocolitis</td>
<td align="left">PMID:37894115</td>
<td align="left">Whipple&#x2019;s disease</td>
<td align="left">PMID:18500934</td>
</tr>
<tr>
<td align="left">Cystic fibrosis</td>
<td align="left">PMID:24178246</td>
<td align="left">Kidney stones</td>
<td align="left">PMID:36798915</td>
</tr>
<tr>
<td align="left">
<italic>Clostridium difficile</italic> infection (CDI)</td>
<td align="left">PMID:36267392</td>
<td align="left">Type 2 diabetes</td>
<td align="left">PMID:31399369</td>
</tr>
<tr>
<td align="left">Ileal Crohn&#x2019;s disease (CD)</td>
<td align="left">PMID:37800577</td>
<td align="left">Guttate psoriasis</td>
<td align="left">PMID:9,627,688</td>
</tr>
<tr>
<td align="left">Crohn&#x2019;s disease (CD)</td>
<td align="left">PMID:36182819</td>
<td align="left">Rheumatoid arthrits</td>
<td align="left">NA</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Notes: The top 10 diseases are listed in the first column, while the top 11&#x2013;20 diseases are listed in the third column.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Bacteroidetes are significant clinical pathogens that, when they breach the intestinal barrier, can induce severe pathology. This includes bacteremia and the formation of abscesses in various parts of the body (<xref ref-type="bibr" rid="B9">Hannah, 2007</xref>). As illustrated in <xref ref-type="table" rid="T6">Table 6</xref>, among the top 20 predicted diseases associated with Bacteroidetes, 19 have been confirmed by existing research indexed in PubMed.</p>
<table-wrap id="T6" position="float">
<label>TABLE 6</label>
<caption>
<p>Top 20 Bacteroidetes-associated candidate diseases on HMDAD.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Disease</th>
<th align="left">Evidence</th>
<th align="left">Disease</th>
<th align="left">Evidence</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Type 1 diabetes</td>
<td align="left">PMID:34361871</td>
<td align="left">Rheumatoid arthrits</td>
<td align="left">NA</td>
</tr>
<tr>
<td align="left">Liver cirrhosis</td>
<td align="left">PMID:37819146</td>
<td align="left">COPD</td>
<td align="left">PMID:37180432</td>
</tr>
<tr>
<td align="left">Irritable bowel syndrome (IBS)</td>
<td align="left">PMID:37616338</td>
<td align="left">Cystic fibrosis</td>
<td align="left">PMID:38179971</td>
</tr>
<tr>
<td align="left">Colorectal carcinoma</td>
<td align="left">PMID:38266708</td>
<td align="left">Crohn&#x2019;s disease (CD)</td>
<td align="left">PMID:35087228</td>
</tr>
<tr>
<td align="left">Infectious colitis</td>
<td align="left">PMID:36531989</td>
<td align="left">Constipation Irritable bowel syndrome (IBS)</td>
<td align="left">PMID:38073315</td>
</tr>
<tr>
<td align="left">Bacterial Vaginosis</td>
<td align="left">PMID:8357044</td>
<td align="left">Atopic sensitisation</td>
<td align="left">PMID:33741316</td>
</tr>
<tr>
<td align="left">Necrotizing Enterocolitis</td>
<td align="left">PMID:39013030</td>
<td align="left">Recurrent wheeze</td>
<td align="left">PMID:29600046</td>
</tr>
<tr>
<td align="left">Periodontal</td>
<td align="left">PMID:3279073</td>
<td align="left">Ulcerative colitis</td>
<td align="left">PMID:35087228</td>
</tr>
<tr>
<td align="left">Type 2 diabetes</td>
<td align="left">PMID:37349979</td>
<td align="left">Ileal Crohn&#x2019;s disease (CD)</td>
<td align="left">PMID:38282618</td>
</tr>
<tr>
<td align="left">Atopic dermatitis</td>
<td align="left">PMID:33551026</td>
<td align="left">
<italic>Clostridium difficile</italic> infection (CDI)</td>
<td align="left">PMID:30619112</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Notes:The top 10 diseases are listed in the first column, while the top 11&#x2013;20 diseases are listed in the third column.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>In summary, these case studies provide additional evidence for the ability of the BANSMDA model to predict potential associations between microbes and diseases.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s5">
<title>Discussion</title>
<p>In the present research, we developed the BANSMDA model, a predictive framework combining Bilinear Attention Networks (BAN) and Sparse Autoencoders (SAE) to identify microbe-disease associations. Our model demonstrates superior performance over existing methods in capturing intricate microbe-disease relationships. However, data scarcity and excessive parameters in the BAN component may induce overfitting, potentially compromising generalization capability in real-world scenarios. Future improvements should focus on integrating biological knowledge, refining model architecture to reduce parameter redundancy, and implementing data augmentation strategies to address data limitations. Balancing model complexity against sparse datasets remains a critical challenge for practical implementation.</p>
<p>The significant gap between AUC-ROC and AUPR values discrepancy reflects the extreme class imbalance in HMDAD (0.5%&#x2013;2% positive samples). AUPR specifically evaluates positive class identification, while AUC-ROC measures overall class discrimination. This imbalance fundamentally constrains AUPR performance, as demonstrated in prior literature. Potential solutions include rigorous negative sample validation and AUPR-optimized training objectives to enhance positive association detection.</p>
<p>In this study, hyperparameter validation was implemented through a combined approach of partial grid search and typical values, constrained by computational resources in this study. Specifically, given the limited sample size, the sparsity target and penalty coefficient were set to relatively low values to prevent over-regularization. Final results were obtained by averaging across multiple runs to minimize potential errors from computational limitations. Experimental results demonstrate that this approach yields hyperparameters enabling model performance approaching the theoretical optimum.</p>
</sec>
<sec sec-type="conclusion" id="s6">
<title>Conclusion</title>
<p>In this study, we introduce a novel model called BANSMDA to predict potential associations between microorganisms and diseases. And experimental results demonstrated the superior performance of BANSMDA. It is important to highlight that data related to microbes and diseases are often characterized by sparsity. While SAE can mitigate overfitting to some extent, the substantial number of parameters introduced by BAN models may still lead to overfitting, particularly when the volume of available data is limited. This, in turn, can compromise model performance. Future research could further enhance the model&#x2019;s performance by incorporating additional biological knowledge, refining the model architecture, or employing data augmentation techniques.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="http://www.cuilab.cn/hmdad">http://www.cuilab.cn/hmdad</ext-link>.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>XL: Writing &#x2013; review and editing, Formal Analysis, Software, Methodology, Investigation, Data curation, Writing &#x2013; original draft, Validation, Project administration, Conceptualization. ML: Writing &#x2013; original draft, Writing &#x2013; review and editing, Investigation, Formal Analysis, Software, Methodology, Data curation, Project administration, Validation, Conceptualization. GY: Writing &#x2013; review and editing, Investigation, Supervision, Validation, Funding acquisition. ST: Writing &#x2013; review and editing, Investigation, Software, Data curation, Project administration. OW: Writing &#x2013; review and editing, Investigation, Resources, Conceptualization, Visualization. BZ: Formal Analysis, Project administration, Funding acquisition, Writing &#x2013; review and editing. LW: Conceptualization, Resources, Funding acquisition, Writing &#x2013; review and editing, Validation, Formal Analysis, Supervision, Data curation, Project administration, Methodology.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work was partly sponsored by the National Natural Science Foundation of China (No. 62272064), the Natural Science Foundation of Hunan Province (No. 2023JJ60185 and No. 2025JJ90184), the Scientific Research Project of Hunan Provincial Department of Education (No. 23C0543 and No. 23C0544), and the Key project of Changsha Science and technology Plan (No. KQ2203001).</p>
</sec>
<ack>
<p>The authors thank the referees for suggestions that helped improve the paper substantially.</p>
</ack>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s11">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s12">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2024b</year>). <article-title>Predicting microbe-disease associations based on a linear neighborhood label propagation method with multi-order similarity fusion learning</article-title>. <source>Interdiscip. Sci. Comput. Life Sci.</source> <volume>16</volume> (<issue>2</issue>), <fpage>345</fpage>&#x2013;<lpage>360</lpage>. <pub-id pub-id-type="doi">10.1007/s12539-024-00607-0</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Y. A.</given-names>
</name>
<name>
<surname>You</surname>
<given-names>Z. H.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>G. Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X. S.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>A novel approach based on KATZ measure to predict associations of human microbiota with non-infectious diseases</article-title>. <source>Bioinformatics</source> <volume>33</volume> (<issue>5</issue>), <fpage>733</fpage>&#x2013;<lpage>739</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btw715</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2024a</year>). <article-title>MLFLHMDA: predicting human microbe-disease association based on multi-view latent feature learning</article-title>. <source>Front. Microbiol.</source> <volume>15</volume>, <fpage>1353278</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2024.1353278</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chie</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Sputum microbiota and inflammatory subtypes in asthma, COPD, and its overlap</article-title>. <source>J. Allergy Clin. Immunol. Glob.</source> <volume>3</volume> (<issue>1</issue>), <fpage>100194</fpage>. <pub-id pub-id-type="doi">10.1016/j.jacig.2023.100194</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cryan</surname>
<given-names>J. F.</given-names>
</name>
<name>
<surname>Dinan</surname>
<given-names>T. G.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Mind-altering microorganisms: the impact of the gut microbiota on brain and behaviour</article-title>. <source>Nat. Rev. Neurosci.</source> <volume>13</volume> (<issue>10</issue>), <fpage>701</fpage>&#x2013;<lpage>712</lpage>. <pub-id pub-id-type="doi">10.1038/nrn3346</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Desbonnet</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Garrett</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Clarke</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Kiely</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Cryan</surname>
<given-names>J. F.</given-names>
</name>
<name>
<surname>Dinan</surname>
<given-names>T. G.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Effects of the probiotic Bifidobacterium infantis in the maternal separation model of depression</article-title>. <source>Neuroscience</source> <volume>170</volume> (<issue>4</issue>), <fpage>1179</fpage>&#x2013;<lpage>1188</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroscience.2010.08.005</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Inferring disease-associated microbes based on multi-data integration and network consistency projection</article-title>. <source>Front. Bioeng. Biotechnol.</source> <volume>8</volume>, <fpage>831</fpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://10.3389/fbioe.2020.00831">https://10.3389/fbioe.2020.00831</ext-link>.</comment>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gilbert</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Meyer</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Antonopoulos</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Balaji</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Brown</surname>
<given-names>C. T.</given-names>
</name>
<name>
<surname>Brown</surname>
<given-names>C. T.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>Meeting report: the terabase metagenomics workshop and the vision of an Earth microbiome project</article-title>. <source>Stand. Genomic Sci.</source> <volume>3</volume> (<issue>3</issue>), <fpage>243</fpage>&#x2013;<lpage>248</lpage>. <pub-id pub-id-type="doi">10.4056/sigs.1433550</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hannah</surname>
<given-names>M. W.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Bacteroides: the good, the bad, and the nitty-gritty</article-title>. <source>Clin. Microbiol. Rev.</source> <volume>20</volume> (<issue>4</issue>), <fpage>593</fpage>&#x2013;<lpage>621</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://10.1128/CMR.00008-07">https://10.1128/CMR.00008-07</ext-link>.</comment>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>MADGAN:A microbe-disease association prediction model based on generative adversarial networks</article-title>. <source>Front. Microbiol.</source> <volume>14</volume>, <fpage>1159076</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2023.1159076</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hui</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>M. H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Therapeutic potential of Clostridium butyricum anticancer effects in colorectal cancer</article-title>. <source>Gut Microbes</source> <volume>15</volume> (<issue>1</issue>), <fpage>2186114</fpage>. <pub-id pub-id-type="doi">10.1080/19490976.2023.2186114</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hwang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>C. Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Hart</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Marcotte</surname>
<given-names>E. M.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>HumanNet v2: human gene networks for disease research</article-title>. <source>Nucleic Acids Res.</source> <volume>47</volume> (<issue>D1</issue>), <fpage>D573-D580</fpage>&#x2013;<lpage>80</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gky1126</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>In&#xe9;s</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>S&#xe1;nchez-de-Diego</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Pradilla Dieste</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cerrada</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Rodriguez Yoldi</surname>
<given-names>M. J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Colorectal carcinoma: a general overview and future perspectives in colorectal cancer</article-title>. <source>Int. J. Mol. Sci.</source> <volume>18</volume> (<issue>1</issue>), <fpage>197</fpage>. <pub-id pub-id-type="doi">10.3390/ijms18010197</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>James</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Pathogenic Escherichia coli</article-title>. <source>Int. J. Med. Microbiol.</source> <volume>295</volume> (<issue>6-7</issue>), <fpage>355</fpage>&#x2013;<lpage>356</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://10.1016/j.ijmm.2005.06.008">https://10.1016/j.ijmm.2005.06.008</ext-link>.</comment>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>James</surname>
<given-names>W. M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Asthma: definitions and pathophysiology</article-title>. <source>Int. Forum Allergy Rhinol.</source> <volume>5</volume> (<issue>1</issue>), <fpage>S2</fpage>&#x2013;<lpage>S6</lpage>. <pub-id pub-id-type="doi">10.1002/alr.21609</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Janssens</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Nielandt</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Bronselaer</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Debunne</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Verbeke</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Wynendaele</surname>
<given-names>E.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Disbiome database: linking the microbiome to disease</article-title>. <source>BMC Microbiol.</source> <volume>18</volume> (<issue>1</issue>), <fpage>50</fpage>. <pub-id pub-id-type="doi">10.1186/s12866-018-1197-5</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kamneva</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Genome composition and phylogeny of microbes predict their co-occurrence in the environment</article-title>. <source>PLoS Comput. Biol.</source> <volume>13</volume>, <fpage>e1005366</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1005366</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kau</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Ahern</surname>
<given-names>P. P.</given-names>
</name>
<name>
<surname>Griffin</surname>
<given-names>N. W.</given-names>
</name>
<name>
<surname>Goodman</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Gordon</surname>
<given-names>J. I.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Human nutrition, the gut microbiome and the immune system</article-title>. <source>Nature</source> <volume>474</volume> (<issue>7351</issue>), <fpage>327</fpage>&#x2013;<lpage>336</lpage>. <pub-id pub-id-type="doi">10.1038/nature10213</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kevin</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Omattage</surname>
<given-names>N. S.</given-names>
</name>
<name>
<surname>Spaulding</surname>
<given-names>C. N.</given-names>
</name>
<name>
<surname>Hultgren</surname>
<given-names>S. J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Reaching the end of the line: urinary tract infections</article-title>. <source>Microbiol. Spectr.</source> <volume>7</volume> (<issue>3</issue>). <pub-id pub-id-type="doi">10.1128/microbiolspec.bai-0014-2019</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kian</surname>
<given-names>F. C.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Potential role of the lung microbiome in shaping asthma phenotypes</article-title>. <source>Ann. Am. Thorac. Soc.</source> <volume>14</volume> (<issue>Suppl. ment_5</issue>), <fpage>S326</fpage>&#x2013;<lpage>S331</lpage>. <pub-id pub-id-type="doi">10.1513/AnnalsATS.201702-138AW</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Yun</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Oh</surname>
<given-names>Y. J.</given-names>
</name>
<name>
<surname>Choi</surname>
<given-names>H. J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Mind-altering with the gut: modulation of the gut-brain axis with probiotics</article-title>. <source>J. Microbiol.</source> <volume>56</volume> (<issue>3</issue>), <fpage>172</fpage>&#x2013;<lpage>182</lpage>. <pub-id pub-id-type="doi">10.1007/s12275-018-8032-4</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liang</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>BANNMDA: a computational model for predicting potential microbe-drug associations based on bilinear attention networks and nuclear norm minimization</article-title>. <source>Front. Microbiol.</source> <volume>15</volume>, <fpage>1497886</fpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://10.3389/fmicb.2024.1497886">https://10.3389/fmicb.2024.1497886</ext-link>.</comment>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Long</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Predicting human microbe-disease associations via graph attention networks with inductive matrix completion</article-title>. <source>Brief. Bioinform</source> <volume>22</volume> (<issue>3</issue>), <fpage>bbaa146</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbaa146</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Miao</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Liao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Predicting potential microbe-disease associations based on auto-encoder and graph convolution network</article-title>. <source>BMC Bioinforma.</source> <volume>24</volume> (<issue>1</issue>), <fpage>476</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-023-05611-7</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Luo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Long</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>NTSHMDA: prediction of human microbe-disease association based on random walk by integrating network topological similarity</article-title>. <source>IEEE/ACM Trans Comput. Biol Bioinf</source> <volume>17</volume> (<issue>4</issue>), <fpage>1341</fpage>&#x2013;<lpage>1351</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2018.2883041</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Geng</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>An analysis of human microbe-disease associations</article-title>. <source>Briefings Bioinforma.</source> <volume>18</volume> (<issue>1</issue>), <fpage>85</fpage>&#x2013;<lpage>97</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbw005</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Olivier</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Skurnik</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Picard</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Denamur</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>The population genetics of commensal Escherichia coli</article-title>. <source>Nat. Rev. Microbiol.</source> <volume>8</volume> (<issue>3</issue>), <fpage>207</fpage>&#x2013;<lpage>217</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://10.1038/nrmicro2298">https://10.1038/nrmicro2298</ext-link>.</comment>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Park</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Moon</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Choi</surname>
<given-names>Y. S.</given-names>
</name>
<name>
<surname>Rho</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Discovering microbe-disease associations from the literature using a hierarchical long short-term memory network and an ensemble parser model</article-title>. <source>Sci. Rep.</source> <volume>11</volume> (<issue>1</issue>), <fpage>4490</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-021-83966-8</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Polyxeni</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Photiades</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zervas</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Xanthou</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Samitas</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Genetics and epigenetics in asthma</article-title>. <source>Int. J. Mol. Sci.</source> <volume>22</volume> (<issue>5</issue>), <fpage>2412</fpage>. <pub-id pub-id-type="doi">10.3390/ijms22052412</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Racanelli</surname>
<given-names>A. C.</given-names>
</name>
<name>
<surname>Ann Kikkers</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Choi</surname>
<given-names>A. M. K.</given-names>
</name>
<name>
<surname>Cloonan</surname>
<given-names>S. M.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Autophagy and inflammation in chronic respiratory disease</article-title>. <source>Autophagy</source> <volume>14</volume> (<issue>2</issue>), <fpage>221</fpage>&#x2013;<lpage>232</lpage>. <pub-id pub-id-type="doi">10.1080/15548627.2017.1389823</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rashmi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ahn</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sampson</surname>
<given-names>J. N.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Xiong</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Fecal microbiota, fecal metabolome, and colorectal cancer interrelations</article-title>. <source>PLoS One</source> <volume>11</volume> (<issue>3</issue>), <fpage>e0152126</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0152126</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sampson</surname>
<given-names>T. R.</given-names>
</name>
<name>
<surname>Debelius</surname>
<given-names>J. W.</given-names>
</name>
<name>
<surname>Thron</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Janssen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shastri</surname>
<given-names>G. G.</given-names>
</name>
<name>
<surname>Ilhan</surname>
<given-names>Z. E.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Gut microbiota regulate motor deficits and neuroinflammation in a model of parkinson&#x2019;s disease</article-title>. <source>Cell</source> <volume>167</volume> (<issue>6</issue>), <fpage>1469</fpage>&#x2013;<lpage>1480</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2016.11.018</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Shen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>A novel approach based on Bi-Random walk to predict microbe-disease associations</article-title>,&#x201d; in <source>Intelligent computing methodologies</source>. <source>Lecture notes in computer science</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Huang</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Gromiha</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hussain</surname>
<given-names>A.</given-names>
</name>
</person-group> (<publisher-name>Springer International Publishing</publisher-name>), <volume>10956</volume>, <fpage>746</fpage>&#x2013;<lpage>752</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-319-95957-3_78</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>Y. Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>D. H.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>S. B.</given-names>
</name>
<name>
<surname>Ming</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J. Q.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>MDAD: a special resource for microbe-drug associations</article-title>. <source>Front. Cell. Infect. Microbiol.</source> <volume>8</volume>, <fpage>424</fpage>. <pub-id pub-id-type="doi">10.3389/fcimb.2018.00424</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Szklarczyk</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Gable</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Lyon</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Junge</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wyder</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Huerta-Cepas</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>STRING v11. Protein-protein association networks with increased coverage, supporting functional discovery in genome-wide experimental datasets</article-title>. <source>Nucleic Acids Res.</source> <volume>47</volume>, <fpage>D607-D613</fpage>&#x2013;<lpage>613</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gky1131</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Toya</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Corban</surname>
<given-names>M. T.</given-names>
</name>
<name>
<surname>Marrietta</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Horwath</surname>
<given-names>I. E.</given-names>
</name>
<name>
<surname>Lerman</surname>
<given-names>L. O.</given-names>
</name>
<name>
<surname>Murray</surname>
<given-names>J. A.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Coronary artery disease is associated with an altered gut microbiome composition</article-title>. <source>PLOS ONE</source> <volume>15</volume> (<issue>1</issue>), <fpage>e0227147</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0227147</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Turnbaugh</surname>
<given-names>P. J.</given-names>
</name>
<name>
<surname>Ley</surname>
<given-names>R. E.</given-names>
</name>
<name>
<surname>Hamady</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Fraser-Liggett</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Knight</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gordon</surname>
<given-names>J. I.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>The human microbiome project</article-title>. <source>Nature</source> <volume>449</volume> (<issue>7164</issue>), <fpage>804</fpage>&#x2013;<lpage>810</lpage>. <pub-id pub-id-type="doi">10.1038/nature06244</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Z. A.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>LRLSHMDA: laplacian regularized least squares for human microbe&#x2013;disease association prediction</article-title>. <source>Sci. Rep.</source> <volume>7</volume> (<issue>1</issue>), <fpage>7601</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-017-08127-2</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>iCircDA-MF: identification of circRNA-disease associations based on matrix factorization</article-title>. <source>Brief. Bioinform</source> <volume>21</volume> (<issue>4</issue>), <fpage>1356</fpage>&#x2013;<lpage>1367</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbz057</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Discovering disease-genes by topological features in human proteinprotein interaction network</article-title>. <source>Bioinformatics</source> <volume>22</volume> (<issue>22</issue>), <fpage>2800</fpage>&#x2013;<lpage>2805</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btl467</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xuan</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Sheng</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Nakaguchi</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Graph convolutional autoencoder and fully-connected autoencoder with attention mechanism based method for predicting drug&#x2013;disease associations</article-title>. <source>IEEE J. Biomed. Health Inf.</source> <volume>25</volume> (<issue>5</issue>), <fpage>1793</fpage>&#x2013;<lpage>1804</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2020.3039502</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yasutoshi</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Kawamura</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Okadome</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Ugai</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Haruki</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Arima</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Enrichment of Bacteroides fragilis and enterotoxigenic Bacteroides fragilis in CpG island methylator phenotype-high colorectal carcinoma</article-title>. <source>Clin. Microbiol. Infect.</source> <volume>30</volume> (<issue>5</issue>), <fpage>630</fpage>&#x2013;<lpage>636</lpage>. <pub-id pub-id-type="doi">10.1016/j.cmi.2024.01.013</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhi</surname>
<given-names>T. Z.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Bai</surname>
<given-names>C. Q.</given-names>
</name>
<name>
<surname>Ling</surname>
<given-names>C. H.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>F. L.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>The protective effects of Helicobacter pylori infection on allergic asthma</article-title>. <source>Int. Arch. Allergy Immunol.</source> <volume>182</volume> (<issue>1</issue>), <fpage>53</fpage>&#x2013;<lpage>64</lpage>. <pub-id pub-id-type="doi">10.1159/000508330</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>