<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Microbiol.</journal-id>
<journal-title>Frontiers in Microbiology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Microbiol.</abbrev-journal-title>
<issn pub-type="epub">1664-302X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmicb.2023.1207209</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Microbiology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>SAELGMDA: Identifying human microbe&#x02013;disease associations based on sparse autoencoder and LightGBM</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name><surname>Wang</surname> <given-names>Feixiang</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2301089/overview"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name><surname>Yang</surname> <given-names>Huandong</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x02020;</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Wu</surname> <given-names>Yan</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1541475/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Peng</surname> <given-names>Lihong</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/601035/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Li</surname> <given-names>Xiaoling</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x0002A;</sup></xref>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>School of Computer Science, Hunan University of Technology</institution>, <addr-line>Zhuzhou</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Gastrointestinal Surgery, Yidu Central Hospital of Weifang</institution>, <addr-line>Weifang</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>Geneis (Beijing) Co., Ltd.</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>The Second Department of Oncology, Beidahuang Industry Group General Hospital</institution>, <addr-line>Harbin</addr-line>, <country>China</country></aff>
<aff id="aff5"><sup>5</sup><institution>The Second Department of Oncology, Heilongjiang Second Cancer Hospital</institution>, <addr-line>Harbin</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Qi Zhao, University of Science and Technology Liaoning, China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Yulin Zhang, Shandong University of Science and Technology, China; Ju Xiang, Changsha Medical University, China</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Lihong Peng <email>plhhnu&#x00040;163.com</email></corresp>
<corresp id="c002">Xiaoling Li <email>athena2111&#x00040;163.com</email></corresp>
<fn fn-type="equal" id="fn001"><p>&#x02020;These authors have contributed equally to this work and share first authorship</p></fn></author-notes>
<pub-date pub-type="epub">
<day>21</day>
<month>06</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>14</volume>
<elocation-id>1207209</elocation-id>
<history>
<date date-type="received">
<day>17</day>
<month>04</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>18</day>
<month>05</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2023 Wang, Yang, Wu, Peng and Li.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Wang, Yang, Wu, Peng and Li</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license> </permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Identification of complex associations between diseases and microbes is important to understand the pathogenesis of diseases and design therapeutic strategies. Biomedical experiment-based Microbe-Disease Association (MDA) detection methods are expensive, time-consuming, and laborious.</p>
</sec>
<sec>
<title>Methods</title>
<p>Here, we developed a computational method called SAELGMDA for potential MDA prediction. First, microbe similarity and disease similarity are computed by integrating their functional similarity and Gaussian interaction profile kernel similarity. Second, one microbe-disease pair is presented as a feature vector by combining the microbe and disease similarity matrices. Next, the obtained feature vectors are mapped to a low-dimensional space based on a Sparse AutoEncoder. Finally, unknown microbe-disease pairs are classified based on Light Gradient boosting machine.</p>
</sec>
<sec>
<title>Results</title>
<p>The proposed SAELGMDA method was compared with four state-of-the-art MDA methods (MNNMDA, GATMDA, NTSHMDA, and LRLSHMDA) under five-fold cross validations on diseases, microbes, and microbe-disease pairs on the HMDAD and Disbiome databases. The results show that SAELGMDA computed the best accuracy, Matthews correlation coefficient, AUC, and AUPR under the majority of conditions, outperforming the other four MDA prediction models. In particular, SAELGMDA obtained the best AUCs of 0.8358 and 0.9301 under cross validation on diseases, 0.9838 and 0.9293 under cross validation on microbes, and 0.9857 and 0.9358 under cross validation on microbe-disease pairs on the HMDAD and Disbiome databases. Colorectal cancer, inflammatory bowel disease, and lung cancer are diseases that severely threat human health. We used the proposed SAELGMDA method to find possible microbes for the three diseases. The results demonstrate that there are potential associations between <italic>Clostridium coccoides</italic> and colorectal cancer and one between Sphingomonadaceae and inflammatory bowel disease. In addition, <italic>Veillonella</italic> may associate with autism. The inferred MDAs need further validation.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>We anticipate that the proposed SAELGMDA method contributes to the identification of new MDAs.</p>
</sec></abstract>
<kwd-group>
<kwd>microbe-disease association</kwd>
<kwd>feature representation</kwd>
<kwd>dimensional reduction</kwd>
<kwd>sparse autoencoder</kwd>
<kwd>LightGBM</kwd>
</kwd-group>
<contract-sponsor id="cn001">National Natural Science Foundation of China<named-content content-type="fundref-id">10.13039/501100001809</named-content></contract-sponsor>
<counts>
<fig-count count="6"/>
<table-count count="9"/>
<equation-count count="23"/>
<ref-count count="95"/>
<page-count count="16"/>
<word-count count="9796"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Systems Microbiology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1. Introduction</title>
<p>Human microbes are a class of organisms with simple structure and small size (Wu et al., <xref ref-type="bibr" rid="B78">2018</xref>; Cheng et al., <xref ref-type="bibr" rid="B11">2020</xref>). They widely distribute in various organs of the human body including the gut, gastrointestinal tract, lung, oral cavity, and skin (Lynch and Pedersen, <xref ref-type="bibr" rid="B49">2016</xref>). Its abnormality may cause diseases, such as cancers, inflammatory bowel disease (El Mouzan et al., <xref ref-type="bibr" rid="B18">2018</xref>), and asthma (Demirci et al., <xref ref-type="bibr" rid="B16">2019</xref>). Therefore, it is important to uncover potential associations between microbes and diseases. Identification of Microbe-Disease Associations (MDAs) helps capture the complex pathogenesis of various diseases and provides novel insights into its drug design. For example, a few methods have been developed to capture potential drugs against COVID-19 (Peng et al., <xref ref-type="bibr" rid="B57">2022a</xref>; Shen L. et al., <xref ref-type="bibr" rid="B61">2022</xref>; Tian et al., <xref ref-type="bibr" rid="B70">2022</xref>). Traditional experimental methods are expensive, time-consuming, and laborious (Chen et al., <xref ref-type="bibr" rid="B8">2019</xref>). Thus, much attention has been devoted to computational methods for new MDA prediction.</p>
<p>Many computational models have been designed to find potential MDAs based on known MDAs and biological features of diseases and microbes. These methods mainly contain network-based methods and machine learning-based methods. Network-based MDA prediction methods include the KATZ measurement (Zhang et al., <xref ref-type="bibr" rid="B88">2017</xref>; Li et al., <xref ref-type="bibr" rid="B37">2019</xref>), random walk with network topology structure (NTSHMDA) (Luo and Long, <xref ref-type="bibr" rid="B48">2018</xref>), and bi-random walk (Zou et al., <xref ref-type="bibr" rid="B95">2017</xref>; Luo and Long, <xref ref-type="bibr" rid="B48">2018</xref>; Yan et al., <xref ref-type="bibr" rid="B82">2019</xref>). Network-based methods effectively found a few new MDAs; however, they depend on known MDAs for similarity calculation and fail to screen possible microbes (or diseases) for a new disease (or microbes) that has no association prediction.</p>
<p>Machine learning-based MDA prediction methods contain Laplacian regularized least squares (LRLSHMDA) (Wang et al., <xref ref-type="bibr" rid="B72">2017</xref>), binary matrix completion (Shi et al., <xref ref-type="bibr" rid="B63">2018</xref>), graph regularized non-negative matrix factorization (He et al., <xref ref-type="bibr" rid="B24">2018</xref>), logistic matrix factorization with neighborhood regularization combining positive-unlabeled learning (Peng et al., <xref ref-type="bibr" rid="B56">2020</xref>), inductive matrix completion and graph attention networks (GATMDA) (Long et al., <xref ref-type="bibr" rid="B47">2021</xref>), and low-rank matrix completion combining the nuclear norm minimization (MNNMDA) (Liu H. et al., <xref ref-type="bibr" rid="B42">2023</xref>). Machine learning algorithms better improved MDA prediction.</p>
<p>In particular, deep learning has been increasingly applied to the area of bioinformatics, such as cardiotoxicity identification related to hERG channel blockers (Wang T. et al., <xref ref-type="bibr" rid="B73">2023</xref>), protein model quality assessment (Guo et al., <xref ref-type="bibr" rid="B23">2022</xref>; Liu J. et al., <xref ref-type="bibr" rid="B43">2023</xref>), metabolite-disease association discovery (Sun et al., <xref ref-type="bibr" rid="B67">2022</xref>), lncRNA-protein interaction prediction (Lihong et al., <xref ref-type="bibr" rid="B40">2021</xref>), lncRNA-miRNA association inference (Chen et al., <xref ref-type="bibr" rid="B7">2021</xref>; Wang et al., <xref ref-type="bibr" rid="B74">2022</xref>), lncRNA-disease association identification (Liang et al., <xref ref-type="bibr" rid="B39">2022</xref>; Zhang et al., <xref ref-type="bibr" rid="B91">2023</xref>), single-cell data analysis (Hu et al., <xref ref-type="bibr" rid="B26">2023</xref>; Xu et al., <xref ref-type="bibr" rid="B81">2023</xref>), drug-target interaction detection (Zhang et al., <xref ref-type="bibr" rid="B86">2022</xref>; Li et al., <xref ref-type="bibr" rid="B38">2023</xref>), and intercellular communication analyses (Peng et al., <xref ref-type="bibr" rid="B58">2022b</xref>). Similarly, deep learning has been widely applied to accurate MDA prediction. These methods include deep matrix factorization combining Bayesian personalized ranking (Liu et al., <xref ref-type="bibr" rid="B44">2020</xref>), multi-component graph attention network (Liu et al., <xref ref-type="bibr" rid="B41">2021</xref>), graph convolutional network (Hua et al., <xref ref-type="bibr" rid="B27">2022</xref>), metapath aggregated graph neural network (Chen and Lei, <xref ref-type="bibr" rid="B9">2022</xref>), dual network contrastive learning model (Cheng et al., <xref ref-type="bibr" rid="B10">2022</xref>), weighted meta-graph-based model (Long and Luo, <xref ref-type="bibr" rid="B46">2019</xref>), knowledge graph neural network (Jiang et al., <xref ref-type="bibr" rid="B30">2022</xref>), and relation graph convolutional network (Wang Y. et al., <xref ref-type="bibr" rid="B75">2023</xref>).</p>
<p>Deep learning efficiently implements accurate MDA identification. In this manuscript, we developed a computational MDA prediction method called SAELGMDA by combining a sparse autoencoder for feature extraction and Light Gradient Boosting Machine (LightGBM) for MDA classification.</p>
</sec>
<sec sec-type="materials and methods" id="s2">
<title>2. Materials and methods</title>
<sec>
<title>2.1. Data description</title>
<p>To construct a human MDA network, we investigated a human MDA database called HMDAD provided by Ma et al. (<xref ref-type="bibr" rid="B50">2017</xref>) (<ext-link ext-link-type="uri" xlink:href="http://www.cuilab.cn/hmdad">http://www.cuilab.cn/hmdad</ext-link>). The database contains 483 experimentally confirmed MDAs between 39 diseases and 292 microbes. We finally achieved 450 MDAs after filtering repetitive MDAs. In addition, Janssens et al. (<xref ref-type="bibr" rid="B29">2018</xref>) have collected a new MDA database named Disbiome. The database contains 5,573 experimentally confirmed human MDAs between 1,098 microbes and 240 diseases. Finally, we obtained 4,351 MDAs between 1,052 microbes and 218 diseases after filtering repetitive MDAs.</p>
<p>Consequently, an element <italic>X</italic><sub><italic>ij</italic></sub> in an MDA matrix <inline-formula><mml:math id="M1"><mml:mi>X</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> is represented as Eq. (1):</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M2"><mml:mrow><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mrow><mml:mo>{</mml:mo> <mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mn>1</mml:mn></mml:mtd><mml:mtd><mml:mrow><mml:mtext>&#x02003;&#x02003;if&#x000A0;disease&#x000A0;</mml:mtext><mml:msub><mml:mi>d</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mtext>&#x000A0;associates&#x000A0;with&#x000A0;microbe&#x000A0;</mml:mtext><mml:msub><mml:mi>m</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mrow><mml:mtext>&#x02003;&#x02003;otherwise</mml:mtext></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow> </mml:mrow></mml:mrow></mml:math></disp-formula>
<p>where <italic>n</italic><sub><italic>d</italic></sub> and <italic>n</italic><sub><italic>m</italic></sub> indicate the number of diseases and microbes, respectively. An MDA is taken as a positive sample if <italic>X</italic><sub><italic>ij</italic></sub> &#x0003D; 1, otherwise, it is taken as an unlabeled sample.</p>
</sec>
<sec>
<title>2.2. Methods</title>
<p>In this manuscript, we proposed an MDA prediction method called SAELGMDA by combining sparse autoencoder and LightGBM. First, disease similarity and microbe similarity are computed by integrating functional similarity and Gaussian Interaction Profile Kernel (GIPK) similarity. Second, one microbe&#x02013;disease pair is represented as one <italic>d</italic>-dimensional vector. Third, the obtained features for microbe&#x02013;disease pairs are mapped into a low-dimensional space via a sparse autoencoder. Finally, the low-dimensional features are fed to LightGBM for MDA classification. The pipeline of SAELGBM is illustrated in <xref ref-type="fig" rid="F1">Figure 1</xref>.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>The flowchart of SAELGMDA.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-14-1207209-g0001.tif"/>
</fig>
<sec>
<title>2.2.1. Functional similarity of diseases and microbes</title>
<p>We considered that similar diseases are more likely to associate with similar genes (Xu and Li, <xref ref-type="bibr" rid="B80">2006</xref>; Wei and Liu, <xref ref-type="bibr" rid="B76">2020</xref>) and computed disease functional similarity via disease-related genes. For two diseases <italic>d</italic><sub><italic>i</italic></sub> and <italic>d</italic><sub><italic>j</italic></sub> and corresponding associated gene sets <italic>G</italic><sub><italic>i</italic></sub> &#x0003D; {<italic>g</italic><sub><italic>i</italic><sub>1</sub></sub>, <italic>g</italic><sub><italic>i</italic><sub>2</sub></sub>, &#x02026;, <italic>g</italic><sub><italic>i</italic><sub><italic>a</italic></sub></sub>} and <italic>G</italic><sub><italic>j</italic></sub> &#x0003D; {<italic>g</italic><sub><italic>j</italic><sub>1</sub></sub>, <italic>g</italic><sub><italic>j</italic><sub>2</sub></sub>, &#x02026;, <italic>g</italic><sub><italic>j</italic><sub><italic>b</italic></sub></sub>}, the functional association between gene <italic>g</italic><sub><italic>k</italic></sub> and gene set <italic>G</italic> &#x0003D; {<italic>g</italic><sub>1</sub>, <italic>g</italic><sub>2</sub>, &#x02026;, <italic>g</italic><sub><italic>l</italic></sub>} is first defined by Eq. (2):</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M3"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo class="qopname">max</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:mi>G</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>F</mml:mi><mml:mi>S</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>FS</italic>(<italic>g</italic><sub><italic>k</italic></sub>, <italic>g</italic><sub><italic>t</italic></sub>) indicates the functional similarity between <italic>g</italic><sub><italic>k</italic></sub> and <italic>g</italic><sub><italic>t</italic></sub> by Eq. (3):</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M4"><mml:mrow><mml:mi>F</mml:mi><mml:mi>S</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>g</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo> <mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo></mml:mrow></mml:mtd><mml:mtd><mml:mrow><mml:mtext>&#x0000A0;if&#x000A0;</mml:mtext><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>L</mml:mi><mml:mi>L</mml:mi><mml:msup><mml:mi>S</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>g</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:mtd><mml:mtd><mml:mrow><mml:mtext>&#x000A0;if&#x000A0;</mml:mtext><mml:mi>k</mml:mi><mml:mo>&#x02260;</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow> </mml:mrow><mml:mtext>&#x000A0;</mml:mtext></mml:mrow></mml:math></disp-formula>
<p>where <italic>LLS</italic>&#x02032; denotes the normalized score of <italic>LLS</italic> by Eq. (4):</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M5"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>L</mml:mi><mml:mi>L</mml:mi><mml:msup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>L</mml:mi><mml:mi>L</mml:mi><mml:mi>S</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mi>L</mml:mi><mml:mi>L</mml:mi><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mo class="qopname">min</mml:mo></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>L</mml:mi><mml:mi>L</mml:mi><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mo class="qopname">max</mml:mo></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mi>L</mml:mi><mml:mi>L</mml:mi><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mo class="qopname">min</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>LLS</italic> represents association log-likelihood score used to evaluate the functional linkage probability between two genes provided by HumanNet (Hwang et al., <xref ref-type="bibr" rid="B28">2019</xref>; Long et al., <xref ref-type="bibr" rid="B47">2021</xref>), <italic>LLS</italic><sub>max</sub> and <italic>LLS</italic><sub>min</sub> denote its maximum and minimum values, respectively.</p>
<p>Finally, the functional similarity between <italic>d</italic><sub><italic>i</italic></sub> and <italic>d</italic><sub><italic>j</italic></sub> is computed by Eq. (5):</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M6"><mml:mrow><mml:mtable columnalign='left'><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mi>f</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>d</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>d</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle='true'><mml:msub><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:msub><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>&#x02208;</mml:mo><mml:mi>G</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>d</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mi>G</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>d</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mstyle><mml:mo>+</mml:mo><mml:mstyle displaystyle='true'><mml:msub><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:msub><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>&#x02208;</mml:mo><mml:mi>G</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>d</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mi>G</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>d</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>g</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mo>+</mml:mo><mml:mi>b</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:math></disp-formula>
<p>Microbe functional similarity matrix <italic>M</italic><sub><italic>f</italic></sub> is computed based on the method proposed by Kamneva (<xref ref-type="bibr" rid="B31">2017</xref>).</p>
</sec>
<sec>
<title>2.2.2. GIPK similarity of diseases and microbes</title>
<p>Based on the assumption that functionally similar diseases usually associate or disassociate with similar microbes, and disease Gaussian Interaction Profile Kernel (GIPK) similarity (Van Laarhoven et al., <xref ref-type="bibr" rid="B71">2011</xref>) is computed via experimentally validated MDA network. In particular, the GIPK similarity of two diseases <italic>d</italic><sub><italic>i</italic></sub> and <italic>d</italic><sub><italic>j</italic></sub> is computed by Eq. (6):</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:mi>I</mml:mi><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mi>I</mml:mi><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>|</mml:mo><mml:msup><mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M8"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup><mml:mo>/</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munderover></mml:mstyle><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:mi>I</mml:mi><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>|</mml:mo><mml:msup><mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>and <italic>IP</italic>(<italic>d</italic><sub><italic>i</italic></sub>) denotes associations between disease <italic>d</italic><sub><italic>i</italic></sub> and each microbe, that is, the <italic>i</italic>th row of <italic>X</italic>. &#x003B3;<sub><italic>d</italic></sub> denotes the normalized kernel bandwidth with original bandwidth <inline-formula><mml:math id="M9"><mml:msubsup><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> of 1, and <italic>n</italic><sub><italic>d</italic></sub> denotes the number of diseases.</p>
<p>Similarly, we computed the GIPK similarity matrix <italic>M</italic><sub><italic>G</italic></sub> of microbes.</p>
</sec>
<sec>
<title>2.2.3. Similarity integration for diseases and microbes</title>
<p>We may fail to compute the functional similarity for all diseases because not all diseases have related to genes. Thus, we combined disease GIPK similarity and functional similarity by Eq. (8):</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M10"><mml:mrow><mml:msub><mml:mi>S</mml:mi><mml:mi>D</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>d</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>d</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo> <mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mn>2</mml:mn></mml:mfrac><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mi>f</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>d</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>d</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>+</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mi>G</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>d</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>d</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mtd><mml:mtd><mml:mrow><mml:mtext>&#x000A0;if&#x000A0;</mml:mtext><mml:msub><mml:mi>D</mml:mi><mml:mi>f</mml:mi></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>d</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>d</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x02260;</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mi>f</mml:mi></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>d</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>d</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mtd><mml:mtd><mml:mrow><mml:mtext>&#x000A0;otherwise&#x000A0;</mml:mtext></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow> </mml:mrow></mml:mrow></mml:math></disp-formula>
<p>Similarly, the integrated microbe similarity <italic>S</italic><sub><italic>M</italic></sub> is computed.</p>
</sec>
<sec>
<title>2.2.4. Feature representation for microbe&#x02013;disease associations</title>
<p>For each microbe&#x02013;disease pair (<italic>m</italic><sub><italic>i</italic></sub>, <italic>d</italic><sub><italic>j</italic></sub>), feature vectors of <italic>m</italic><sub><italic>i</italic></sub> and <italic>d</italic><sub><italic>j</italic></sub> are obtained based on similarity matrices <italic>S</italic><sub><italic>D</italic></sub> and <italic>S</italic><sub><italic>M</italic></sub>, respectively. Particularly, the feature vector of <italic>d</italic><sub><italic>i</italic></sub> is denoted as the similarity between <italic>d</italic><sub><italic>i</italic></sub> and all diseases. The feature vector of <italic>m</italic><sub><italic>j</italic></sub> is denoted as the similarity between <italic>m</italic><sub><italic>j</italic></sub> and all microbes. Thus, one microbe&#x02013;disease pair is depicted as an (<italic>n</italic><sub><italic>d</italic></sub>&#x0002B;<italic>n</italic><sub><italic>m</italic></sub>)-dimensional feature vector after concatenation operation, where <italic>n</italic><sub><italic>d</italic></sub> and <italic>n</italic><sub><italic>m</italic></sub> indicate the number of diseases and microbes, respectively. In summary, there are <italic>n</italic>(<italic>n</italic> &#x0003D; <italic>n</italic><sub><italic>d</italic></sub>&#x000D7;<italic>n</italic><sub><italic>m</italic></sub>) samples (microbe&#x02013;disease pairs), and each sample <italic>x</italic><sub><italic>i</italic></sub> can be represented using a <italic>d</italic>(<italic>d</italic> &#x0003D; <italic>n</italic><sub><italic>d</italic></sub>&#x0002B;<italic>n</italic><sub><italic>m</italic></sub>)-dimensional vector. For <italic>x</italic><sub><italic>i</italic></sub>, its label <italic>y</italic><sub><italic>i</italic></sub> &#x0003D; 1. If its corresponding microbe&#x02013;disease pair is associated, otherwise <italic>y</italic><sub><italic>i</italic></sub> &#x0003D; 0. Consequently, an MDA matrix <italic>X</italic> with <italic>n</italic> samples is represented by Eq. (9):</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M11"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>X</mml:mi><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr></mml:mtr></mml:mtable></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mo>&#x022EF;</mml:mo><mml:mspace width="0.3em" class="thinspace"/><mml:mo>,</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr></mml:mtr></mml:mtable></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mo>&#x022EF;</mml:mo><mml:mspace width="0.3em" class="thinspace"/><mml:mo>,</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr></mml:mtr></mml:mtable></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mo>&#x022EF;</mml:mo><mml:mspace width="0.3em" class="thinspace"/><mml:mo>,</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr></mml:mtr></mml:mtable></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec>
<sec>
<title>2.2.5. Feature extraction based on sparse autoencoder</title>
<p>The obtained features for microbe&#x02013;disease pairs are highly dimensional and severely affect the classification accuracy of models. Deep learning demonstrates stronger feature learning ability than traditional dimensional reduction approaches. Thus, we designed a sparse autoencoder to reduce the feature dimensionality of each sample.</p>
<p>Sparse autoencoder (Andrew, <xref ref-type="bibr" rid="B1">2011</xref>) is an unsupervised neural network model. It minimizes the reconstruction error and enforces sparsity constraints on all hidden nodes to obtain a more robust and meaningful representation of features and further improves the prediction performance of classification models (Makhzani and Frey, <xref ref-type="bibr" rid="B52">2013</xref>). First, a high-dimensional feature vector for the microbe&#x02013;disease pair is fed to an encoder by Eq. (10):</p>
<disp-formula id="E10"><label>(10)</label><mml:math id="M12"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>H</mml:mi><mml:mo>=</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>W</mml:mi><mml:mi>X</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>b</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>X</italic> represents the input <italic>n</italic> samples with <italic>d</italic>-dimensional vector, <italic>H</italic> denotes the low-dimensional features after encoding, <italic>W</italic>, <italic>b</italic>, and <italic>f</italic>(&#x000B7;) represent the weight, bias, and encoding function of the encoder, respectively.</p>
<p>Next, a decoder restores the low-dimensional representation <italic>H</italic> to the same appearance as the input feature representation by Eq. (11):</p>
<disp-formula id="E11"><label>(11)</label><mml:math id="M13"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mover accent="true"><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:mi>g</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mi>H</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>W</italic>&#x02032;, <italic>b</italic>&#x02032;, and <italic>g</italic>(&#x000B7;) represent the weight, bias, and decoding function of the decoder, respectively, and <inline-formula><mml:math id="M14"><mml:mover accent="true"><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:math></inline-formula> denotes the learned feature representation.</p>
<p>To minimize the reconstruction error, we build a cost function by Eq. (12):</p>
<disp-formula id="E12"><label>(12)</label><mml:math id="M15"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:mi>E</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003BB;</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003A9;</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">sparsity</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B2;</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003A9;</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">weights</mml:mtext></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003BB; and &#x003B2; denote the sparsity regularization parameter and the coefficients for <italic>L</italic><sub>2</sub> regularization, respectively.</p>
<p>The first term <italic>MSE</italic> is mean square error. The term is used to measure the discrepancy between the input features <italic>X</italic> and the reconstructed features <inline-formula><mml:math id="M16"><mml:mover accent="true"><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:math></inline-formula> on training data by Eq. (13):</p>
<disp-formula id="E13"><label>(13)</label><mml:math id="M17"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The second term &#x003A9;<sub>sparsity</sub> is the Kullback&#x02013;Leibler divergence. The term is used to control sparsity based on the sparsity proportion &#x003C1; by Eq. (14):</p>
<disp-formula id="E14"><label>(14)</label><mml:math id="M18"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>&#x003A9;</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">sparsity</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munderover></mml:mstyle><mml:mi>K</mml:mi><mml:mi>L</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003C1;</mml:mi><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>&#x003C1;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>s</italic><sub><italic>l</italic></sub> and <inline-formula><mml:math id="M19"><mml:msub><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>&#x003C1;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> denote the number of neurons in the <italic>l</italic>th hidden layer and the average activity of the <italic>t</italic>th neuron, respectively, <inline-formula><mml:math id="M20"><mml:mi>K</mml:mi><mml:mi>L</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003C1;</mml:mi><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:mover accent="false"><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003C1;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> denotes the relative entropy between Bernoulli random variables with mean &#x003C1; and mean <inline-formula><mml:math id="M21"><mml:msub><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>&#x003C1;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. <inline-formula><mml:math id="M22"><mml:mi>K</mml:mi><mml:mi>L</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003C1;</mml:mi><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:mover accent="false"><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003C1;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> is computed by Eq. (15):</p>
<disp-formula id="E15"><label>(15)</label><mml:math id="M23"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>K</mml:mi><mml:mi>L</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003C1;</mml:mi><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:mover accent="false"><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003C1;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>&#x003C1;</mml:mi><mml:mo class="qopname">log</mml:mo><mml:mfrac><mml:mrow><mml:mi>&#x003C1;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>&#x003C1;</mml:mi></mml:mrow><mml:mo class="qopname">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>&#x003C1;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo class="qopname">log</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>&#x003C1;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>&#x003C1;</mml:mi></mml:mrow><mml:mo class="qopname">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The third term is <italic>L</italic><sub>2</sub> regularization term &#x003A9;<sub>weights</sub>. The term is used to control the weights and avoid overfitting by Eq. (16):</p>
<disp-formula id="E16"><label>(16)</label><mml:math id="M24"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>&#x003A9;</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">weights</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:munderover></mml:mstyle><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munderover></mml:mstyle><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:munderover></mml:mstyle><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>n</italic><sub><italic>l</italic></sub>, <italic>s</italic><sub><italic>l</italic></sub>, and <inline-formula><mml:math id="M25"><mml:msubsup><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> denote the number of layers, the number of units in the <italic>l</italic>th layer, and the weight, respectively.</p>
</sec>
<sec>
<title>2.2.6. MDA classification based on LightGBM</title>
<p>Each microbe&#x02013;disease pair is represented as a low-dimensional vector after dimensional reduction based on a sparse autoencoder. LightGBM (Ke et al., <xref ref-type="bibr" rid="B34">2017</xref>) is an optimized version of Gradient Boosting Decision Tree (GBDT) (Ye et al., <xref ref-type="bibr" rid="B84">2009</xref>). It obtains better performance in the area of bioinformatics. Next, the constructed low-dimensional vector is used as the input of LightGBM (Ke et al., <xref ref-type="bibr" rid="B34">2017</xref>), to classify each microbe&#x02013;disease pair. For an MDA dataset <inline-formula><mml:math id="M26"><mml:mi>D</mml:mi><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>, LightGBM intends to learn an approximation <inline-formula><mml:math id="M27"><mml:mover accent="true"><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:math></inline-formula> to a certain function <italic>f</italic>(<italic>x</italic>) by minimizing the expectation of the loss function <italic>L</italic>(<italic>y, f</italic>(<italic>x</italic>)) by Eq. (17):</p>
<disp-formula id="E17"><label>(17)</label><mml:math id="M28"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mover accent="true"><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:mo class="qopname">arg</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo class="qopname">min</mml:mo></mml:mrow><mml:mrow><mml:mi>f</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>L</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>y</mml:mi><mml:mo>,</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>LightGBM integrates <italic>T</italic> decision trees <inline-formula><mml:math id="M29"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:munderover><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> to approximate the final model <inline-formula><mml:math id="M30"><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:munderover><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>. The decision trees with <italic>J</italic> leaf nodes are expressed as <italic>w</italic><sub><italic>q</italic>(<italic>x</italic>)</sub>, where <italic>w</italic><sub><italic>q</italic>(<italic>x</italic>)</sub> denotes the weights of all samples on leaf nodes and <italic>q</italic>(<italic>x</italic>) denotes the decision rules. Hence, The loss function of LightGBM is defined by Eq. (18):</p>
<disp-formula id="E18"><label>(18)</label><mml:math id="M31"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>&#x00393;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mi>L</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The constant term in model (18) is removed for simplicity, and model (18) is transformed as Eq. (19):</p>
<disp-formula id="E19"><label>(19)</label><mml:math id="M32"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>&#x00393;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02245;</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>g</italic><sub><italic>i</italic></sub> and <italic>h</italic><sub><italic>i</italic></sub> denote the first-order and second-order derivatives of the loss function, respectively.</p>
<p>For a sample set, <italic>I</italic><sub><italic>j</italic></sub> related to leaf <italic>j</italic>, model (19) could be transformed as follows:</p>
<disp-formula id="E20"><label>(20)</label><mml:math id="M33"><mml:mrow><mml:msub><mml:mi>&#x00393;</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>J</mml:mi></mml:munderover><mml:mo stretchy='false'>(</mml:mo></mml:mstyle><mml:mo stretchy='false'>(</mml:mo><mml:mstyle displaystyle='true'><mml:munder><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mi>I</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow></mml:munder><mml:mrow><mml:msub><mml:mi>g</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mstyle><mml:mo stretchy='false'>)</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mn>2</mml:mn></mml:mfrac><mml:mo stretchy='false'>(</mml:mo><mml:mstyle displaystyle='true'><mml:munder><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mi>I</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow></mml:munder><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mstyle><mml:mo>+</mml:mo><mml:mi>&#x003BB;</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:msubsup><mml:mi>w</mml:mi><mml:mi>j</mml:mi><mml:mn>2</mml:mn></mml:msubsup><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:math></disp-formula>
<p>Given a tree structure <italic>q</italic>(<italic>x</italic>), the optimal leaf weight <inline-formula><mml:math id="M34"><mml:msubsup><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msubsup></mml:math></inline-formula> of each leaf node and the maximum value of a scoring function &#x00393;<sub><italic>k</italic></sub> that evaluate the quality of <italic>q</italic>(<italic>x</italic>) are defined by Eqs. (21) and (22):</p>
<disp-formula id="E21"><label>(21)</label><mml:math id="M35"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003BB;</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E22"><label>(22)</label><mml:math id="M36"><mml:mrow><mml:mtable columnalign='left'><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:msubsup><mml:mi>&#x00393;</mml:mi><mml:mi>T</mml:mi><mml:mo>*</mml:mo></mml:msubsup><mml:mo>=</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mn>2</mml:mn></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>J</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mfrac><mml:mrow><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mstyle displaystyle='true'><mml:msub><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msub><mml:mi>I</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mi>g</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow><mml:mrow><mml:mstyle displaystyle='true'><mml:msub><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mi>I</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mi>&#x003BB;</mml:mi></mml:mrow></mml:mstyle></mml:mrow></mml:mfrac></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:math></disp-formula>
<p>Consequently, the objective function is represented as Eq. (23):</p>
<disp-formula id="E23"><label>(23)</label><mml:math id="M37"><mml:mrow><mml:mtable columnalign='left'><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:mi>G</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mn>2</mml:mn></mml:mfrac><mml:mo stretchy='false'>(</mml:mo><mml:mfrac><mml:mrow><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mstyle displaystyle='true'><mml:msub><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mi>I</mml:mi><mml:mi>L</mml:mi></mml:msub></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mi>g</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow><mml:mrow><mml:mstyle displaystyle='true'><mml:msub><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mi>I</mml:mi><mml:mi>L</mml:mi></mml:msub></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mi>&#x003BB;</mml:mi></mml:mrow></mml:mstyle></mml:mrow></mml:mfrac><mml:mo>+</mml:mo><mml:mfrac><mml:mrow><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mstyle displaystyle='true'><mml:msub><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mi>I</mml:mi><mml:mi>R</mml:mi></mml:msub></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mi>g</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow><mml:mrow><mml:mstyle displaystyle='true'><mml:msub><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mi>I</mml:mi><mml:mi>R</mml:mi></mml:msub></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mi>&#x003BB;</mml:mi></mml:mrow></mml:mstyle></mml:mrow></mml:mfrac><mml:mo>&#x02212;</mml:mo><mml:mfrac><mml:mrow><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mstyle displaystyle='true'><mml:msub><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mi>I</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mi>g</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow><mml:mrow><mml:mstyle displaystyle='true'><mml:msub><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mi>I</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mi>&#x003BB;</mml:mi></mml:mrow></mml:mstyle></mml:mrow></mml:mfrac><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:math></disp-formula>
<p>where <italic>I</italic><sub><italic>L</italic></sub> and <italic>I</italic><sub><italic>R</italic></sub> denote the example sets on the left and right sides, respectively.</p>
</sec>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3. Results</title>
<sec>
<title>3.1. Experimental settings and evaluation metrics</title>
<p>Similar to RNMFMDA provided by Peng et al. (<xref ref-type="bibr" rid="B56">2020</xref>), the experiments were performed under three 5-fold cross validations (CVs) 20 times. For an MDA matrix <italic>X</italic><sub><italic>n</italic></sub>, the three CVs were as follows:</p>
<list list-type="bullet">
<list-item><p>five-fold CV 1 (<italic>CV</italic><sub>1</sub>): CV on diseases, i.e., in each round, 80% of <italic>n</italic><sub><italic>d</italic></sub> diseases in <italic>X</italic> was taken as training set and the remaining 20% was test set.</p></list-item>
<list-item><p>five-fold CV 2 (<italic>CV</italic><sub>2</sub>): CV on microbes, i.e., in each round, 80% of <italic>n</italic><sub><italic>m</italic></sub> microbes in <italic>X</italic> was taken as training set and the remaining 20% was test set.</p></list-item>
<list-item><p>five-fold Cv 3 (<italic>CV</italic><sub>3</sub>): CV on microbe&#x02013;disease pairs, i.e., in each round, 80% of entries (microbe&#x02013;disease pairs) in <italic>X</italic> were used as training set and the remaining 20% was test set.</p></list-item>
</list>
<p>In the sparse autoencoder, the neural network comprised an encoder and a decoder. The network structure was trained in Keras based on the TensorFlow backend. The structure comprised one input layer, three hidden layers, and an output layer. The number of each layer was 331, 256, 128, 96, and 64, respectively. The layers in the encoder and decoder were symmetric around the bottleneck. Tanh and ReLU were used as the activation functions in the output layer and the other layers, respectively. The optimization method used the Adam algorithm (Kingma and Ba, <xref ref-type="bibr" rid="B35">2014</xref>). The batch size was set to 32 because a smaller batch size can make the model converge faster. The parameters &#x003BB;, &#x003B2;, and &#x003C1; were set to 0.1, 0.0005, and 0.05, respectively. The final encoding size of the autoencoder is set to 64, that is, the features of MDAs were reduced to 64 dimensions.</p>
<p>For LightGBM, the parameters &#x0201C;num_leaves,&#x0201D; &#x0201C;learning_rate,&#x0201D; and &#x0201C;max_depth&#x0201D; denote the number of leaves in a tree, the speed of iteration, and the maximum depth of the tree, respectively. They were set to 31, 0.1, and &#x02013;1, respectively. &#x0201C;Feature_fraction&#x0201D; and &#x0201C;bagging_fraction&#x0201D; are two hyperparameters in the optimization process. The former denotes the fraction of features at each iteration and was set to 0.9. The latter denotes the fraction of data and applies to boost the training and reduce overfitting. It was set to 0.9. &#x0201C;min_data&#x0201D; denotes the minimum number of records in a leaf and is also used to reduce overfitting. The parameters in the other four comparison methods were set to the defaults in corresponding publications. One microbe&#x02013;disease pair is taken as a positive MDA when its association probability is greater than 50%, otherwise, it is taken as a negative MDA.</p>
<p>Four evaluation metrics were used to measure the performance of MDA prediction methods: accuracy, Matthews correlation coefficient (MCC) (Chicco and Jurman, <xref ref-type="bibr" rid="B12">2020</xref>), area under the ROC curve (AUC), and area under the Precision-Recall curve (AUPR). Higher values for the four evaluation metrics represent better performance.</p>
</sec>
<sec>
<title>3.2. Performance comparison of SAELGMDA with the other four methods</title>
<p>To evaluate the performance of SAELGMDA, we compared it with four state-of-the-art MDA identification algorithms (MNNMDA, GATMDA, NTSHMDA, and LRLSHMDA) under three CVs on the HMDAD and Disbiome datasets, that is, <italic>CV</italic><sub>1</sub>, <italic>CV</italic><sub>2</sub>, and <italic>CV</italic><sub>3</sub>.</p>
<sec>
<title>3.2.1. Performance comparison under <italic>CV</italic><sub>1</sub></title>
<p><xref ref-type="table" rid="T1">Table 1</xref> shows accuracies, MCCs, AUCs, and AUPRs of SAELGMDA and the other four methods under <italic>CV</italic><sub>1</sub>. The best performance in each column is described in <xref ref-type="table" rid="T1">Tables 1</xref>&#x02013;<bold>6</bold>. As shown in <xref ref-type="table" rid="T1">Table 1</xref>, SAELGMDA computed the best MCC, AUC, and AUPR on the HMDAD database and the best accuracy, MCC, AUC, and AUPR on the Disbiome database, significantly outperforming the other four MDA prediction methods under <italic>CV</italic><sub>1</sub>. Although accuracy was slightly less than MNNMDA and GATMDA on HMDAD, the difference was very tiny. Moreover, SAELGMDA outperformed the other methods, especially AUC and AUPR on the whole. In addition, although SAELGMDA outperformed the other four methods, all methods computed lower MCC and AUPR under <italic>CV</italic><sub>1</sub>, which may be caused by fewer diseases. <xref ref-type="fig" rid="F2">Figure 2</xref> shows the ROC and PR curves of the five methods on the two databases under <italic>CV</italic><sub>1</sub>.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>The performance of five MDA identification methods under <italic>CV</italic><sub>1</sub>.</p></caption> 
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Database</bold></th>
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="left"><bold>Accuracy</bold></th>
<th valign="top" align="left"><bold>MCC</bold></th>
<th valign="top" align="left"><bold>AUC</bold></th>
<th valign="top" align="left"><bold>AUPR</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" rowspan="5"> HMDAD</td>
<td valign="top" align="left">SAELGMDA</td>
<td valign="top" align="left">0.9497 &#x000B1; 0.0022</td>
<td valign="top" align="left"><bold>0.1855</bold> <bold>&#x000B1;0.0116</bold></td>
<td valign="top" align="left"><bold>0.8358</bold> <bold>&#x000B1;0.0109</bold></td>
<td valign="top" align="left"><bold>0.2155</bold> <bold>&#x000B1;0.0075</bold></td>
</tr>
<tr>
<td valign="top" align="left">MNNMDA</td>
<td valign="top" align="left"><bold>0.9588</bold> <bold>&#x000B1;0.0009</bold></td>
<td valign="top" align="left">0.1085 &#x000B1; 0.0109</td>
<td valign="top" align="left">0.6907 &#x000B1; 0.0040</td>
<td valign="top" align="left">0.1206 &#x000B1; 0.0021</td>
</tr>
<tr>
<td valign="top" align="left">GATMDA</td>
<td valign="top" align="left">0.9562 &#x000B1; 0.0009</td>
<td valign="top" align="left">0.0421 &#x000B1; 0.0018</td>
<td valign="top" align="left">0.5152 &#x000B1; 0.0003</td>
<td valign="top" align="left">0.0816 &#x000B1; 0.0014</td>
</tr>
<tr>
<td valign="top" align="left">NTSHMDA</td>
<td valign="top" align="left">0.9138 &#x000B1; 0.0006</td>
<td valign="top" align="left">0.0101 &#x000B1; 0.0008</td>
<td valign="top" align="left">0.6423 &#x000B1; 0.0085</td>
<td valign="top" align="left">0.0531 &#x000B1; 0.0007</td>
</tr>
<tr>
<td valign="top" align="left">LRLSHMDA</td>
<td valign="top" align="left">0.9421 &#x000B1; 0.0007</td>
<td valign="top" align="left">0.1182 &#x000B1; 0.0028</td>
<td valign="top" align="left">0.5343 &#x000B1; 0.0109</td>
<td valign="top" align="left">0.0769 &#x000B1; 0.0006</td>
</tr> <tr>
<td valign="top" align="left" rowspan="5"> Disbiome</td>
<td valign="top" align="left">SAELGMDA</td>
<td valign="top" align="left"><bold>0.9819</bold> <bold>&#x000B1;0.0000</bold></td>
<td valign="top" align="left"><bold>0.3431</bold> <bold>&#x000B1;0.0059</bold></td>
<td valign="top" align="left"><bold>0.9301</bold> <bold>&#x000B1;0.0002</bold></td>
<td valign="top" align="left"><bold>0.3469</bold> <bold>&#x000B1;0.0037</bold></td>
</tr>
<tr>
<td valign="top" align="left">MNNMDA</td>
<td valign="top" align="left">0.9814 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.1521 &#x000B1; 0.0008</td>
<td valign="top" align="left">0.6774 &#x000B1; 0.0010</td>
<td valign="top" align="left">0.1207 &#x000B1; 0.0004</td>
</tr>
<tr>
<td valign="top" align="left">GATMDA</td>
<td valign="top" align="left">0.9807 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.0542 &#x000B1; 0.0019</td>
<td valign="top" align="left">0.5214 &#x000B1; 0.0005</td>
<td valign="top" align="left">0.2166 &#x000B1; 0.0192</td>
</tr>
<tr>
<td valign="top" align="left">NTSHMDA</td>
<td valign="top" align="left">0.9416 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.0204 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.5898 &#x000B1; 0.0002</td>
<td valign="top" align="left">0.0235 &#x000B1; 0.0000</td>
</tr>
<tr>
<td valign="top" align="left">LRLSHMDA</td>
<td valign="top" align="left">0.9772 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.1469 &#x000B1; 0.0004</td>
<td valign="top" align="left">0.7200 &#x000B1; 0.0005</td>
<td valign="top" align="left">0.1109 &#x000B1; 0.0002</td>
</tr></tbody>
</table>
</table-wrap>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>The ROC and the PR curves of five different methods under <italic>CV</italic><sub>1</sub> on the two databases. <bold>(A, B)</bold> Denote the ROC curves on the HMDAD and Disbiome databases, respectively. <bold>(C, D)</bold> Denote the PR curves on the HMDAD and Disbiome databases, respectively.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-14-1207209-g0002.tif"/>
</fig>
</sec>
<sec>
<title>3.2.2. Performance comparison under <italic>CV</italic><sub>2</sub></title>
<p><xref ref-type="table" rid="T2">Table 2</xref> demonstrates the prediction performance of SAELGMDA and the other four methods under <italic>CV</italic><sub>2</sub>. The best performance in each column is described in boldface. As shown in <xref ref-type="table" rid="T2">Table 2</xref>, we observed that SAELGMDA computed the best accuracies, MCCs, and AUCs on the two databases under <italic>CV</italic><sub>2</sub>. In particular, SAELGMDA obtained better MCC and AUPR on the HMDAD database than ones on the Disbiome database, which may be caused by different data structures. In addition, all five MDA prediction methods computed lower MCC and AUPR on the Disbiome database. <xref ref-type="fig" rid="F3">Figure 3</xref> shows the ROC and PR curves of the five methods under <italic>CV</italic><sub>2</sub>.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>The performance of five MDA identification methods under <italic>CV</italic><sub>2</sub>.</p></caption> 
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Database</bold></th>
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="left"><bold>Accuracy</bold></th>
<th valign="top" align="left"><bold>MCC</bold></th>
<th valign="top" align="left"><bold>AUC</bold></th>
<th valign="top" align="left"><bold>AUPR</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" rowspan="5"> HMDAD</td>
<td valign="top" align="left">SAELGMDA</td>
<td valign="top" align="left"><bold>0.986</bold> <bold>&#x000B1;0.0000</bold></td>
<td valign="top" align="left"><bold>0.8017</bold> <bold>&#x000B1;0.0017</bold></td>
<td valign="top" align="left"><bold>0.9838</bold> <bold>&#x000B1;0.0001</bold></td>
<td valign="top" align="left"><bold>0.8706</bold> <bold>&#x000B1;0.0010</bold></td>
</tr>
<tr>
<td valign="top" align="left">MNNMDA</td>
<td valign="top" align="left">0.9654 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.344 &#x000B1; 0.0034</td>
<td valign="top" align="left">0.896 &#x000B1; 0.0016</td>
<td valign="top" align="left">0.7479 &#x000B1; 0.0052</td>
</tr>
<tr>
<td valign="top" align="left">GATMDA</td>
<td valign="top" align="left">0.9604 &#x000B1; 0.0001</td>
<td valign="top" align="left">0.4775 &#x000B1; 0.0065</td>
<td valign="top" align="left">0.7977 &#x000B1; 0.0020</td>
<td valign="top" align="left">0.4677 &#x000B1; 0.0096</td>
</tr>
<tr>
<td valign="top" align="left">NTSHMDA</td>
<td valign="top" align="left">0.9642 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.4449 &#x000B1; 0.0029</td>
<td valign="top" align="left">0.8614 &#x000B1; 0.0007</td>
<td valign="top" align="left">0.3718 &#x000B1; 0.0026</td>
</tr>
<tr>
<td valign="top" align="left">LRLSHMDA</td>
<td valign="top" align="left">0.9642 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.4451 &#x000B1; 0.0017</td>
<td valign="top" align="left">0.8596 &#x000B1; 0.0009</td>
<td valign="top" align="left">0.4068 &#x000B1; 0.0065</td>
</tr> <tr>
<td valign="top" align="left" rowspan="5"> Disbiome</td>
<td valign="top" align="left">SAELGMDA</td>
<td valign="top" align="left"><bold>0.9818</bold> <bold>&#x000B1;0.0000</bold></td>
<td valign="top" align="left"><bold>0.3437</bold> <bold>&#x000B1;0.0040</bold></td>
<td valign="top" align="left"><bold>0.9293</bold> <bold>&#x000B1;0.0003</bold></td>
<td valign="top" align="left">0.3378 &#x000B1; 0.0049</td>
</tr>
<tr>
<td valign="top" align="left">MNNMDA</td>
<td valign="top" align="left">0.9817 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.1907 &#x000B1; 0.0016</td>
<td valign="top" align="left">0.7744 &#x000B1; 0.0015</td>
<td valign="top" align="left"><bold>0.4117</bold> <bold>&#x000B1;0.0023</bold></td>
</tr>
<tr>
<td valign="top" align="left">GATMDA</td>
<td valign="top" align="left">0.9763 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.0915 &#x000B1; 0.0011</td>
<td valign="top" align="left">0.5761 &#x000B1; 0.0009</td>
<td valign="top" align="left">0.1069 &#x000B1; 0.0031</td>
</tr>
<tr>
<td valign="top" align="left">NTSHMDA</td>
<td valign="top" align="left">0.9723 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.0951 &#x000B1; 0.0002</td>
<td valign="top" align="left">0.7721 &#x000B1; 0.0002</td>
<td valign="top" align="left">0.0767 &#x000B1; 0.0000</td>
</tr>
<tr>
<td valign="top" align="left">LRLSHMDA</td>
<td valign="top" align="left">0.9657 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.1135 &#x000B1; 0.0002</td>
<td valign="top" align="left">0.7792 &#x000B1; 0.0002</td>
<td valign="top" align="left">0.0905 &#x000B1; 0.0001</td>
</tr></tbody>
</table>
</table-wrap>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>The ROC and the PR curves of five different methods under <italic>CV</italic><sub>2</sub> on the two databases. <bold>(A, B)</bold> Denote the ROC curves on the HMDAD and Disbiome databases, respectively. <bold>(C, D)</bold> Denote the PR curves on the HMDAD and Disbiome databases, respectively.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-14-1207209-g0003.tif"/>
</fig>
</sec>
<sec>
<title>3.2.3. Performance comparison under <italic>CV</italic><sub>3</sub></title>
<p><xref ref-type="table" rid="T3">Table 3</xref> shows the performance of SAELGMDA and the other four methods under <italic>CV</italic><sub>3</sub>. The best performance in each column is described in boldface under <italic>CV</italic><sub>3</sub>. The results from <xref ref-type="table" rid="T3">Table 3</xref> suggest that SAELGMDA achieved the best accuracies, MCCs, and AUCs, significantly outperforming the other four MDA prediction methods under <italic>CV</italic><sub>3</sub>. Moreover, the performance of all five methods under <italic>CV</italic><sub>3</sub> outperforms the ones under <italic>CV</italic><sub>1</sub> and <italic>CV</italic><sub>2</sub>, demonstrating that more samples help improve the classification performance. <xref ref-type="fig" rid="F4">Figure 4</xref> shows the ROC and PR curves of the five methods under <italic>CV</italic><sub>3</sub>.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>The performance of five MDA identification methods under <italic>CV</italic><sub>3</sub>.</p></caption> 
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Database</bold></th>
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="left"><bold>Accuracy</bold></th>
<th valign="top" align="left"><bold>MCC</bold></th>
<th valign="top" align="left"><bold>AUC</bold></th>
<th valign="top" align="left"><bold>AUPR</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" rowspan="5"> HMDAD</td>
<td valign="top" align="left">SAELGMDA</td>
<td valign="top" align="left"><bold>0.9859</bold> <bold>&#x000B1;0.0000</bold></td>
<td valign="top" align="left"><bold>0.7978</bold> <bold>&#x000B1;0.0010</bold></td>
<td valign="top" align="left"><bold>0.9857</bold> <bold>&#x000B1;0.0000</bold></td>
<td valign="top" align="left"><bold>0.8705</bold> <bold>&#x000B1;0.0008</bold></td>
</tr>
<tr>
<td valign="top" align="left">MNNMDA</td>
<td valign="top" align="left">0.9653 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.3401 &#x000B1; 0.0055</td>
<td valign="top" align="left">0.9511 &#x000B1; 0.0002</td>
<td valign="top" align="left">0.6465 &#x000B1; 0.0023</td>
</tr>
<tr>
<td valign="top" align="left">GATMDA</td>
<td valign="top" align="left">0.8935 &#x000B1; 0.0004</td>
<td valign="top" align="left">0.3427 &#x000B1; 0.0020</td>
<td valign="top" align="left">0.8638 &#x000B1; 0.0007</td>
<td valign="top" align="left">0.3230 &#x000B1; 0.0060</td>
</tr>
<tr>
<td valign="top" align="left">NTSHMDA</td>
<td valign="top" align="left">0.9613 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.1783 &#x000B1; 0.0338</td>
<td valign="top" align="left">0.8874 &#x000B1; 0.0003</td>
<td valign="top" align="left">0.3568 &#x000B1; 0.0026</td>
</tr>
<tr>
<td valign="top" align="left">LRLSHMDA</td>
<td valign="top" align="left">0.9453 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.0568 &#x000B1; 0.0011</td>
<td valign="top" align="left">0.7997 &#x000B1; 0.0002</td>
<td valign="top" align="left">0.1158 &#x000B1; 0.0002</td>
</tr> <tr>
<td valign="top" align="left" rowspan="5"> Disbiome</td>
<td valign="top" align="left">SAELGMDA</td>
<td valign="top" align="left"><bold>0.9826</bold> <bold>&#x000B1;0.0000</bold></td>
<td valign="top" align="left"><bold>0.3376</bold> <bold>&#x000B1;0.0004</bold></td>
<td valign="top" align="left"><bold>0.9358</bold> <bold>&#x000B1;0.0000</bold></td>
<td valign="top" align="left">0.3604 &#x000B1; 0.0004</td>
</tr>
<tr>
<td valign="top" align="left">MNNMDA</td>
<td valign="top" align="left">0.9815 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.1523 &#x000B1; 0.0012</td>
<td valign="top" align="left">0.9355 &#x000B1; 0.0000</td>
<td valign="top" align="left"><bold>0.4175</bold> <bold>&#x000B1;0.0002</bold></td>
</tr>
<tr>
<td valign="top" align="left">GATMDA</td>
<td valign="top" align="left">0.8461 &#x000B1; 0.0004</td>
<td valign="top" align="left">0.2032 &#x000B1; 0.0002</td>
<td valign="top" align="left">0.8332 &#x000B1; 0.0001</td>
<td valign="top" align="left">0.201 &#x000B1; 0.0004</td>
</tr>
<tr>
<td valign="top" align="left">NTSHMDA</td>
<td valign="top" align="left">0.9807 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.0207 &#x000B1; 0.0002</td>
<td valign="top" align="left">0.8146 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.0766 &#x000B1; 0.0000</td>
</tr>
<tr>
<td valign="top" align="left">LRLSHMDA</td>
<td valign="top" align="left">0.9781 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.0744 &#x000B1; 0.0002</td>
<td valign="top" align="left">0.7365 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.0625 &#x000B1; 0.0000</td>
</tr></tbody>
</table>
</table-wrap>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>The ROC and the PR curves of five different methods under <italic>CV</italic><sub>3</sub> on the two databases. <bold>(A, B)</bold> Denote the ROC curves on the HMDAD and Disbiome databases, respectively. <bold>(C, D)</bold> Denote the PR curves on the HMDAD and Disbiome databases, respectively.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-14-1207209-g0004.tif"/>
</fig>
</sec>
<sec>
<title>3.2.4. Performance comparison of LightGBM and two classification models</title>
<p>To measure the MDA classification performance of LightGBM, we compared it with two classical boosting algorithms, XGBoost and NGBoost. Extreme Gradient Boosting (XGBoost) is an ensemble learning method based on a gradient boost tree and can accurately cope with multicollinearity impact and complicated non-linearity interactions (Chen and Guestrin, <xref ref-type="bibr" rid="B6">2016</xref>; Zhu and Zhu, <xref ref-type="bibr" rid="B93">2019</xref>). Natural Gradient Boosting (NGBoost) uses natural gradients instead of regular gradients to implement flexible probabilistic forecast (Duan et al., <xref ref-type="bibr" rid="B17">2020</xref>). <xref ref-type="table" rid="T4">Tables 4</xref>&#x02013;<xref ref-type="table" rid="T6">6</xref> show the accuracy, MCC, AUC, and AUPR of LightGBM, NGBoost, and XGBoost on the Disbiome and HMDAD datasets under three cross validations. The results from <xref ref-type="table" rid="T4">Tables 4</xref>&#x02013;<xref ref-type="table" rid="T6">6</xref> indicate that LightGBM obtained better performance on the majority of conditions and can be used to improve MDA classification ability.</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>The performance of three classification models under <italic>CV</italic><sub>1</sub>.</p></caption> 
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Database</bold></th>
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="left"><bold>Accuracy</bold></th>
<th valign="top" align="left"><bold>MCC</bold></th>
<th valign="top" align="left"><bold>AUC</bold></th>
<th valign="top" align="left"><bold>AUPR</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" rowspan="3"> HMDAD</td>
<td valign="top" align="left">LightGBM</td>
<td valign="top" align="left">0.9497 &#x000B1; 0.0022</td>
<td valign="top" align="left"><bold>0.1855</bold> <bold>&#x000B1;0.0116</bold></td>
<td valign="top" align="left">0.8358 &#x000B1; 0.0109</td>
<td valign="top" align="left"><bold>0.2155</bold> <bold>&#x000B1;0.0075</bold></td>
</tr>
<tr>
<td valign="top" align="left">NGBoost</td>
<td valign="top" align="left"><bold>0.9526</bold> <bold>&#x000B1;0.0016</bold></td>
<td valign="top" align="left">0.1728 &#x000B1; 0.0107</td>
<td valign="top" align="left">0.8301 &#x000B1; 0.0097</td>
<td valign="top" align="left">0.1988 &#x000B1; 0.0056</td>
</tr>
<tr>
<td valign="top" align="left">XGBoost</td>
<td valign="top" align="left">0.946 &#x000B1; 0.0018</td>
<td valign="top" align="left">0.1832 &#x000B1; 0.0092</td>
<td valign="top" align="left"><bold>0.8385</bold> <bold>&#x000B1;0.0051</bold></td>
<td valign="top" align="left">0.1843 &#x000B1; 0.0050</td>
</tr> <tr>
<td valign="top" align="left" rowspan="3"> Disbiome</td>
<td valign="top" align="left">LightGBM</td>
<td valign="top" align="left">0.9819 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.3431 &#x000B1; 0.0059</td>
<td valign="top" align="left"><bold>0.9301</bold> <bold>&#x000B1;0.0002</bold></td>
<td valign="top" align="left">0.3469 &#x000B1; 0.0037</td>
</tr>
<tr>
<td valign="top" align="left">NGBoost</td>
<td valign="top" align="left"><bold>0.9826</bold> <bold>&#x000B1;0.0000</bold></td>
<td valign="top" align="left"><bold>0.3631</bold> <bold>&#x000B1;0.0032</bold></td>
<td valign="top" align="left">0.9284 &#x000B1; 0.0002</td>
<td valign="top" align="left"><bold>0.3598</bold> <bold>&#x000B1;0.0027</bold></td>
</tr>
<tr>
<td valign="top" align="left">XGBoost</td>
<td valign="top" align="left">0.9775 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.2706 &#x000B1; 0.0034</td>
<td valign="top" align="left">0.905 &#x000B1; 0.0003</td>
<td valign="top" align="left">0.2494 &#x000B1; 0.002</td>
</tr></tbody>
</table>
</table-wrap>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>The performance of three classification models under <italic>CV</italic><sub>2</sub>.</p></caption> 
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Database</bold></th>
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="left"><bold>Accuracy</bold></th>
<th valign="top" align="left"><bold>MCC</bold></th>
<th valign="top" align="left"><bold>AUC</bold></th>
<th valign="top" align="left"><bold>AUPR</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" rowspan="3"> HMDAD</td>
<td valign="top" align="left">LightGBM</td>
<td valign="top" align="left"><bold>0.986</bold> <bold>&#x000B1;0.0000</bold></td>
<td valign="top" align="left"><bold>0.8017</bold> <bold>&#x000B1;0.0017</bold></td>
<td valign="top" align="left"><bold>0.9838</bold> <bold>&#x000B1;0.0001</bold></td>
<td valign="top" align="left"><bold>0.8706</bold> <bold>&#x000B1;0.0010</bold></td>
</tr>
<tr>
<td valign="top" align="left">NGBoost</td>
<td valign="top" align="left">0.9854 &#x000B1; 0.0046</td>
<td valign="top" align="left">0.794 &#x000B1; 0.0511</td>
<td valign="top" align="left">0.9808 &#x000B1; 0.0102</td>
<td valign="top" align="left">0.8615 &#x000B1; 0.0447</td>
</tr>
<tr>
<td valign="top" align="left">XGBoost</td>
<td valign="top" align="left">0.9846 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.7814 &#x000B1; 0.0027</td>
<td valign="top" align="left">0.9803 &#x000B1; 0.0001</td>
<td valign="top" align="left">0.8434 &#x000B1; 0.0021</td>
</tr> <tr>
<td valign="top" align="left" rowspan="3"> Disbiome</td>
<td valign="top" align="left">LightGBM</td>
<td valign="top" align="left"><bold>0.9818</bold> <bold>&#x000B1;0.0000</bold></td>
<td valign="top" align="left"><bold>0.3437</bold> <bold>&#x000B1;0.0040</bold></td>
<td valign="top" align="left"><bold>0.9293</bold> <bold>&#x000B1;0.0003</bold></td>
<td valign="top" align="left">0.3378 &#x000B1; 0.0049</td>
</tr>
<tr>
<td valign="top" align="left">NGBoost</td>
<td valign="top" align="left">0.9817 &#x000B1; 0.0034</td>
<td valign="top" align="left">0.3382 &#x000B1; 0.0756</td>
<td valign="top" align="left">0.9284 &#x000B1; 0.0164</td>
<td valign="top" align="left"><bold>0.3597</bold> <bold>&#x000B1;0.0920</bold></td>
</tr>
<tr>
<td valign="top" align="left">XGBoost</td>
<td valign="top" align="left">0.9771 &#x000B1; 0.0054</td>
<td valign="top" align="left">0.2671 &#x000B1; 0.0619</td>
<td valign="top" align="left">0.904 &#x000B1; 0.0186</td>
<td valign="top" align="left">0.2502 &#x000B1; 0.0640</td>
</tr></tbody>
</table>
</table-wrap>
<table-wrap position="float" id="T6">
<label>Table 6</label>
<caption><p>The performance of three classification models under <italic>CV</italic><sub>3</sub>.</p></caption> 
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Database</bold></th>
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="left"><bold>Accuracy</bold></th>
<th valign="top" align="left"><bold>MCC</bold></th>
<th valign="top" align="left"><bold>AUC</bold></th>
<th valign="top" align="left"><bold>AUPR</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" rowspan="3"> HMDAD</td>
<td valign="top" align="left">LightGBM</td>
<td valign="top" align="left"><bold>0.9859</bold> <bold>&#x000B1;0.0000</bold></td>
<td valign="top" align="left"><bold>0.7978</bold> <bold>&#x000B1;0.0010</bold></td>
<td valign="top" align="left"><bold>0.9857</bold> <bold>&#x000B1;0.0000</bold></td>
<td valign="top" align="left"><bold>0.8705</bold> <bold>&#x000B1;0.0008</bold></td>
</tr>
<tr>
<td valign="top" align="left">NGBoost</td>
<td valign="top" align="left">0.9854 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.7905 &#x000B1; 0.0013</td>
<td valign="top" align="left">0.9821 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.8625 &#x000B1; 0.0013</td>
</tr>
<tr>
<td valign="top" align="left">XGBoost</td>
<td valign="top" align="left">0.9838 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.7679 &#x000B1; 0.0011</td>
<td valign="top" align="left">0.9804 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.835 &#x000B1; 0.0010</td>
</tr> <tr>
<td valign="top" align="left" rowspan="3"> Disbiome</td>
<td valign="top" align="left">SAELGMDA</td>
<td valign="top" align="left"><bold>0.9826</bold> <bold>&#x000B1;0.0000</bold></td>
<td valign="top" align="left">0.3376 &#x000B1; 0.0004</td>
<td valign="top" align="left"><bold>0.9358</bold> <bold>&#x000B1;0.0000</bold></td>
<td valign="top" align="left">0.3604 &#x000B1; 0.0004</td>
</tr>
<tr>
<td valign="top" align="left">LightGBM</td>
<td valign="top" align="left"><bold>0.9826</bold> <bold>&#x000B1;0.0000</bold></td>
<td valign="top" align="left"><bold>0.3396</bold> <bold>&#x000B1;0.0003</bold></td>
<td valign="top" align="left">0.9336 &#x000B1; 0.0000</td>
<td valign="top" align="left"><bold>0.3764</bold> <bold>&#x000B1;0.0002</bold></td>
</tr>
<tr>
<td valign="top" align="left">XGBoost</td>
<td valign="top" align="left">0.9805 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.2375 &#x000B1; 0.0039</td>
<td valign="top" align="left">0.9129 &#x000B1; 0.0000</td>
<td valign="top" align="left">0.2594 &#x000B1; 0.0002</td>
</tr></tbody>
</table>
</table-wrap>
</sec>
<sec>
<title>3.2.5. Computational time analysis</title>
<p>We compared the computational time of SAELGMDA with the other four MDA prediction models, MNNMDA, GATMDA, NTSHMDA, and LRLSHMDA. The experiments were run on a machine with an AMD EPYC 7302 CPU, a GeForce RTX 2080 Ti, and 256GB RAM on Ubuntu 20.04.4 LTS operating system. <xref ref-type="fig" rid="F5">Figure 5</xref> shows computational time (m) of the five MDA prediction models on five-fold cross validation for one time on two MDA datasets. As shown in <xref ref-type="fig" rid="F5">Figure 5</xref>, SAELGMDA is the most rapid method on the HMDAD dataset and the slowest one on the Disbiome dataset. SAELGMDA need only to spend 10.57 min, although it run slowly on the Disbiome database. In summary, SAELGMDA need not too much time on the two MDA datasets.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Computational time of five MDA methods.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-14-1207209-g0005.tif"/>
</fig>
</sec>
</sec>
<sec>
<title>3.3. Case study</title>
<p>In this section, we predicted potential MDAs on the two MDA databases. In addition, multiple evidence suggests that colorectal cancer, inflammatory bowel diseases, and lung cancer have dense linkages with microbes (Guarner and Malagelada, <xref ref-type="bibr" rid="B22">2003</xref>; M&#x000FC;ller and Macpherson, <xref ref-type="bibr" rid="B54">2006</xref>; Zhang et al., <xref ref-type="bibr" rid="B90">2015</xref>; M&#x000E1;rmol et al., <xref ref-type="bibr" rid="B53">2017</xref>; Chicco and Jurman, <xref ref-type="bibr" rid="B12">2020</xref>). In this section, we aim to find possible microbes for the three diseases using the proposed SAELGMDA method. For the three diseases, microbes that are known to associate with them were removed. Next, we computed the association scores between them and all microbes. Third, the computed scores were sorted in descending order. Finally, the top 20 microbes with the highest association scores with them were listed and confirmed by the existing publications.</p>
<sec>
<title>3.3.1. Finding new MDAs based on known MDAs</title>
<p>We further predicted new MDAs based on known MDAs using SAELGMDA. The predicted top 50 MDAs are shown in <xref ref-type="fig" rid="F6">Figure 6</xref>. In <xref ref-type="fig" rid="F6">Figure 6</xref>, sky blue solid lines and red dotted lines represent known and unknown MDAs obtained from SAELGMDA, respectively. Deep sky blue round rectangles represent microbes and green diamonds denote diseases.</p>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>The predicted top 50 MDAs on the HMDAD <bold>(A)</bold> and Disbiome <bold>(B)</bold> databases.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-14-1207209-g0006.tif"/>
</fig>
<p>On the HMDAD database, all predicted top 50 MDAs have been known to be associated with the database. SAELGMDA predicted that Actinobacteria and liver cirrhosis have the highest association probability with the ranking of 130 among all 11,388 microbe&#x02013;disease pairs. Actinobacteria have been reported to associate with liver disease (Bull-Otterson et al., <xref ref-type="bibr" rid="B5">2013</xref>). The expansion of Proteobacteria and Actinobacteria has a pathogenic effect on alcoholic liver disease (Bull-Otterson et al., <xref ref-type="bibr" rid="B5">2013</xref>).</p>
<p>In the Disbiome database, SAELGMDA predicted that <italic>Veillonella</italic> may associate with autism with a ranking of three among all 229,336 microbe&#x02013;disease pairs. Zhang et al. (<xref ref-type="bibr" rid="B87">2018</xref>) has reported that the abundance of <italic>Veillonella</italic> was severely decreased in stools of children suffering from autism spectrum disorder. The decreasing of its abundance has been also found in subjects involved in autism (Strati et al., <xref ref-type="bibr" rid="B66">2017</xref>). Furthermore, the decreased <italic>Veillonella</italic> may affect the fermentation of lactate in the autism children (Gronow et al., <xref ref-type="bibr" rid="B21">2010</xref>).</p>
</sec>
<sec>
<title>3.3.2. Colorectal cancer-related microbe identification</title>
<p>Colorectal cancer is the third most frequent cause of cancer mortality worldwide, severely threatening global life and health (Biller and Schrag, <xref ref-type="bibr" rid="B3">2021</xref>; Saeed et al., <xref ref-type="bibr" rid="B60">2021</xref>; Wong et al., <xref ref-type="bibr" rid="B77">2023</xref>). There are more than 1.85 million colorectal cancer cases and 850,000 colorectal cancer-related deaths each year. In total, 20% of patients with colorectal cancer have metastasis cancer among new colorectal cancer diagnoses. It has been reported that &#x0007E;70%&#x02013;75% of patients survive more than 1 year, 30%&#x02013;35% more than 3 years, and fewer than 20% more than 5 years among patients diagnosed with metastatic colorectal cancer. Although colonoscopy has been widely applied to the screen, its effect on colorectal cancer remains unclear (Bretthauer et al., <xref ref-type="bibr" rid="B4">2022</xref>). <xref ref-type="table" rid="T7">Table 7</xref> shows the top 20 microbes associated with colorectal cancer on the HMDAD database.</p>
<table-wrap position="float" id="T7">
<label>Table 7</label>
<caption><p>The top 20 microbes related to colorectal cancer inferred by SAELGMDA on the HMDAD database.</p></caption> 
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Rank</bold></th>
<th valign="top" align="left"><bold>Microbe</bold></th>
<th valign="top" align="left"><bold>Evidence</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1</td>
<td valign="top" align="left"><italic>Fusobacterium nucleatum</italic></td>
<td valign="top" align="left">Confirmed by HMDAD</td>
</tr> <tr>
<td valign="top" align="left">2</td>
<td valign="top" align="left">Firmicutes</td>
<td valign="top" align="left">Confirmed by HMDAD</td>
</tr> <tr>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Proteobacteria</td>
<td valign="top" align="left">PMID: 24 603 888, 27 194 068, 32 298 987</td>
</tr> <tr>
<td valign="top" align="left">4</td>
<td valign="top" align="left"><italic>Prevotella</italic></td>
<td valign="top" align="left">Confirmed by HMDAD</td>
</tr> <tr>
<td valign="top" align="left">5</td>
<td valign="top" align="left"><italic>Bacteroidetes</italic></td>
<td valign="top" align="left">Confirmed by HMDAD</td>
</tr> <tr>
<td valign="top" align="left">6</td>
<td valign="top" align="left">Clostridia</td>
<td valign="top" align="left">Confirmed by HMDAD</td>
</tr> <tr>
<td valign="top" align="left">7</td>
<td valign="top" align="left"><italic>Fusobacterium</italic></td>
<td valign="top" align="left">Confirmed by HMDAD</td>
</tr> <tr>
<td valign="top" align="left">8</td>
<td valign="top" align="left"><italic>Bacteroides</italic></td>
<td valign="top" align="left">Confirmed by HMDAD</td>
</tr> <tr>
<td valign="top" align="left">9</td>
<td valign="top" align="left"><italic>Pseudomonas</italic></td>
<td valign="top" align="left">PMID: 33 998 814, 25699023, 25 217 106</td>
</tr> <tr>
<td valign="top" align="left">10</td>
<td valign="top" align="left"><italic>Haemophilus</italic></td>
<td valign="top" align="left">PMID: 31 358 825, 26 549 775</td>
</tr> <tr>
<td valign="top" align="left">11</td>
<td valign="top" align="left">Actinobacteria</td>
<td valign="top" align="left">PMID: 35 899 111, 35 049 922</td>
</tr> <tr>
<td valign="top" align="left">12</td>
<td valign="top" align="left"><italic>Acinetobacter</italic></td>
<td valign="top" align="left">PMID: 32 738 757, 32 595 614</td>
</tr> <tr>
<td valign="top" align="left">13</td>
<td valign="top" align="left"><italic>Corynebacterium</italic></td>
<td valign="top" align="left">PMID: 313 873, 646 934</td>
</tr> <tr>
<td valign="top" align="left">14</td>
<td valign="top" align="left"><italic>Lactobacillus</italic></td>
<td valign="top" align="left">PMID: 36 162 222, 22 830 611, 35 808 840</td>
</tr> <tr>
<td valign="top" align="left">15</td>
<td valign="top" align="left"><italic>Streptococcus</italic></td>
<td valign="top" align="left">PMID: 9 771 449, 21 960 713, 21 247 505, 18 990 738, 16 845 563</td>
</tr> <tr>
<td valign="top" align="left">16</td>
<td valign="top" align="left"><italic>Clostridium difficile</italic></td>
<td valign="top" align="left">PMID: 26 691 472, 28 060 753, 21 152 135, 1 626 323</td>
</tr> <tr>
<td valign="top" align="left">17</td>
<td valign="top" align="left"><italic>Faecalibacterium prausnitzii</italic></td>
<td valign="top" align="left">PMID: 26 595 550, 35 625 865, 32 675 782</td>
</tr> <tr>
<td valign="top" align="left">18</td>
<td valign="top" align="left"><italic>Clostridium coccoides</italic></td>
<td valign="top" align="left">Unconfirmed</td>
</tr> <tr>
<td valign="top" align="left">19</td>
<td valign="top" align="left">Lachnospiraceae</td>
<td valign="top" align="left">PMID: 28 988 196, 36 893 736</td>
</tr> <tr>
<td valign="top" align="left">20</td>
<td valign="top" align="left"><italic>Helicobacter pylori</italic></td>
<td valign="top" align="left">PMID: 22 294 430, 16 579 836, 18 506 454, 31 393 968</td>
</tr></tbody>
</table>
</table-wrap>
<p>For colorectal cancer, as shown in <xref ref-type="table" rid="T7">Table 7</xref>, 19 microbes have been confirmed to have associations with colorectal cancer by the existing literature on the top 20 inferred microbes on the HMDAD database. For example, pseudomonas is distinctly less abundant in cancer tissues than normal tissues and has been increasingly taken as an emerging clinic-related opportunistic pathogen (Decker and Palmore, <xref ref-type="bibr" rid="B15">2014</xref>; Gao et al., <xref ref-type="bibr" rid="B19">2015</xref>). <italic>Haemophilus parainfluenzae</italic> demonstrates higher representation in colorectal cancer subjects but is scarcely investigated in control subjects (Kasai et al., <xref ref-type="bibr" rid="B33">2016</xref>). Research in 219 patients with colorectal cancer has suggested that clostridium difficile has a dense relationship with colorectal cancer (Yeom et al., <xref ref-type="bibr" rid="B85">2010</xref>). Helicobacter pylori infection has been reported to be a potential risk increase factor of left-sided colorectal cancer (Zhang et al., <xref ref-type="bibr" rid="B89">2012</xref>).</p>
<p>Moreover, we inferred that <italic>Clostridium coccoides</italic> has a possible association with colorectal cancer. <italic>Clostridium coccoides</italic> is taken as one of the most prevalent groups of bacteria in human intestines. They constitute &#x0007E;60% of mucin-adhered microbiota and comprise different species with high oxygen-sensitive anaerobes (such as <italic>Clostridium, Coprococcus, Eubacterium</italic>, and <italic>Ruminococcus</italic>). They contribute to the prevention of colonization of vancomycin-resistant <italic>Enterococcus</italic> in an antibiotic-treated mouse model (Grenda et al., <xref ref-type="bibr" rid="B20">2022</xref>). The association between <italic>Clostridium coccoides</italic> and colorectal cancer needs further validation.</p>
</sec>
<sec>
<title>3.3.3. Inflammatory bowel disease-related microbe identification</title>
<p>Inflammatory bowel disease is one of the idiopathic inflammatory bowel disorders that severely affect the gastrointestinal tract. It has become a global, chronic, and life-threatening disease over the last few decades. Mak et al. (<xref ref-type="bibr" rid="B51">2020</xref>) predicted that patients with inflammatory bowel disease may be an exponential increase worldwide. It typically includes Crohn&#x00027;s disease and ulcerative colitis. It manifests progressive and unpredictable features and is partially caused by bacteria that activate patient&#x00027;s immune system to protect against foreign substances (Lomax et al., <xref ref-type="bibr" rid="B45">2006</xref>; Kaplan and Windsor, <xref ref-type="bibr" rid="B32">2021</xref>). It has a close relationship with microbes. Identification of associated microbes for the disease helps us better equip to stem its global rise in future. <xref ref-type="table" rid="T8">Table 8</xref> lists the top 20 microbes associated with the disease on the HMDAD database.</p>
<table-wrap position="float" id="T8">
<label>Table 8</label>
<caption><p>The top 20 microbes related to inflammatory bowel disease inferred by SAELGMDA on the HMDAD database.</p></caption> 
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Rank</bold></th>
<th valign="top" align="left"><bold>Microbe</bold></th>
<th valign="top" align="left"><bold>Evidence</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1</td>
<td valign="top" align="left"><italic>Bacteroidetes</italic></td>
<td valign="top" align="left">PMID: 12 906 096, 27 999 802, 21 575 910</td>
</tr> <tr>
<td valign="top" align="left">2</td>
<td valign="top" align="left">Proteobacteria</td>
<td valign="top" align="left">Confirmed by HMDAD</td>
</tr> <tr>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Firmicutes</td>
<td valign="top" align="left">PMID: 19 235 886</td>
</tr> <tr>
<td valign="top" align="left">4</td>
<td valign="top" align="left">Lachnospiraceae</td>
<td valign="top" align="left">Confirmed by HMDAD</td>
</tr> <tr>
<td valign="top" align="left">5</td>
<td valign="top" align="left"><italic>Haemophilus</italic></td>
<td valign="top" align="left">PMID: 33 666 710, 30 685 379</td>
</tr> <tr>
<td valign="top" align="left">6</td>
<td valign="top" align="left">Actinobacteria</td>
<td valign="top" align="left">Confirmed by HMDAD</td>
</tr> <tr>
<td valign="top" align="left">7</td>
<td valign="top" align="left"><italic>Prevotella</italic></td>
<td valign="top" align="left">PMID: 28 542 929, 26 468 751</td>
</tr> <tr>
<td valign="top" align="left">8</td>
<td valign="top" align="left"><italic>Clostridium coccoides</italic></td>
<td valign="top" align="left">PMID: 27 687 331, 16 432 374</td>
</tr> <tr>
<td valign="top" align="left">9</td>
<td valign="top" align="left"><italic>Bifidobacterium</italic></td>
<td valign="top" align="left">PMID: 34 337 079, 25 793 197, 24 478 468, 25 391 346</td>
</tr> <tr>
<td valign="top" align="left">10</td>
<td valign="top" align="left"><italic>Lactobacillus</italic></td>
<td valign="top" align="left">PMID: 29 854 599, 32 509 162, 15 664 933</td>
</tr> <tr>
<td valign="top" align="left">11</td>
<td valign="top" align="left"><italic>Staphylococcus aureus</italic></td>
<td valign="top" align="left">PMID: 31 698 044</td>
</tr> <tr>
<td valign="top" align="left">12</td>
<td valign="top" align="left"><italic>Fusobacterium</italic></td>
<td valign="top" align="left">PMID: 27 139 617, 33 996 366, 25 576 662</td>
</tr> <tr>
<td valign="top" align="left">13</td>
<td valign="top" align="left">Clostridia</td>
<td valign="top" align="left">PMID: 22 508 484, 28 506 071</td>
</tr> <tr>
<td valign="top" align="left">14</td>
<td valign="top" align="left"><italic>Clostridium difficile</italic></td>
<td valign="top" align="left">PMID: 22 508 484, 28 506 071</td>
</tr> <tr>
<td valign="top" align="left">15</td>
<td valign="top" align="left"><italic>Helicobacter pylori</italic></td>
<td valign="top" align="left">PMID: 24 914 359, 19 760 778</td>
</tr> <tr>
<td valign="top" align="left">16</td>
<td valign="top" align="left"><italic>Streptococcus</italic></td>
<td valign="top" align="left">PMID: 30 392 911, 23 679 203, 28 618 865, 16 868 828</td>
</tr> <tr>
<td valign="top" align="left">17</td>
<td valign="top" align="left"><italic>Bacteroides vulgatus</italic></td>
<td valign="top" align="left">PMID: 12 906 096, 12 162 408</td>
</tr> <tr>
<td valign="top" align="left">18</td>
<td valign="top" align="left"><italic>Bacteroides</italic></td>
<td valign="top" align="left">PMID: 12 906 096, 12 162 408</td>
</tr> <tr>
<td valign="top" align="left">19</td>
<td valign="top" align="left">Oxalobacteraceae</td>
<td valign="top" align="left">PMID: 29228248</td>
</tr> <tr>
<td valign="top" align="left">20</td>
<td valign="top" align="left">Sphingomonadaceae</td>
<td valign="top" align="left">Unconfirmed</td>
</tr></tbody>
</table>
</table-wrap>
<p>As shown in <xref ref-type="table" rid="T8">Table 8</xref>, 19 microbes have been validated to link to inflammatory bowel disorders by existing literature on the predicted top 20 microbes associated with it on the HMDAD database. Researchers reported that Firmicutes were less represented in patients suffered from inflammatory bowel disease than healthy subjects (Sokol et al., <xref ref-type="bibr" rid="B65">2009</xref>). <italic>Streptococcus</italic> and <italic>Haemophilus</italic> were highly represented in patients with inflammatory bowel disease (Heidarian et al., <xref ref-type="bibr" rid="B25">2019</xref>). <italic>Prevotella</italic> was reduced in pediatric Crohn&#x00027;s disease (Lewis et al., <xref ref-type="bibr" rid="B36">2015</xref>). <italic>Clostridium coccoides</italic> was less abundant in patients with active inflammatory bowel disease than ones in remission (Prosberg et al., <xref ref-type="bibr" rid="B59">2016</xref>).</p>
<p>In addition, we predicted that Sphingomonadaceae dense links to inflammatory bowel disease. Sphingomonadaceae family has high abundance in marine waters, freshwater, and even drinking water. They can degrade lignin-derived compounds and refractory organic matter that comprise monocyclic and polycyclic aromatic hydrocarbons (Shen S. et al., <xref ref-type="bibr" rid="B62">2022</xref>). Sphingomonadaceae are significantly accommodated to bile salts through metabolic pathways (de Vries et al., <xref ref-type="bibr" rid="B14">2019</xref>). In addition, Sphingomonadaceae has a high linkage with triclosan degradation in nitrification and denitrification systems (Dai et al., <xref ref-type="bibr" rid="B13">2022</xref>). Microbial communities were adapted to Bisphenol A through the selection of Sphingomonadaceae populations including <italic>Sphingobium, Novosphingobium</italic>, and <italic>Sphingopyxis</italic>. The selected Sphingomonadaceae for Bisphenol A demonstrated higher Bisphenol A metabolic activity (Oh and Choi, <xref ref-type="bibr" rid="B55">2019</xref>). The association between Sphingomonadaceae and inflammatory bowel disease needs further validation.</p>
</sec>
<sec>
<title>3.3.4. Lung cancer-related microbe identification</title>
<p>Lung cancer is one of the leading causes of cancer-related deaths worldwide. It accounts for &#x0007E;18% of global cancer deaths (Sung et al., <xref ref-type="bibr" rid="B68">2021</xref>). More than 350 patients died from lung cancer each day in the United States (Siegel et al., <xref ref-type="bibr" rid="B64">2022</xref>). It has the highest incidence and mortality compared with other cancer types in China (Xia et al., <xref ref-type="bibr" rid="B79">2022</xref>). We used the proposed SAELGMDA model to identify potential microbes for lung cancer. <xref ref-type="table" rid="T9">Table 9</xref> lists the top 20 microbes associated with it on the Disbiome database. As shown in <xref ref-type="table" rid="T9">Table 9</xref>, all 20 top microbes have been confirmed to be associated with lung cancer by existing literatures or the Disbiome database. The results again validated the MDA prediction performance of SAELGMDA.</p>
<table-wrap position="float" id="T9">
<label>Table 9</label>
<caption><p>The top 20 microbes associated with lung cancer identified by SAELGMDA on the Disbiome database.</p></caption> 
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Rank</bold></th>
<th valign="top" align="left"><bold>Microbe</bold></th>
<th valign="top" align="left"><bold>Evidence</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1</td>
<td valign="top" align="left"><italic>Acidovorax</italic></td>
<td valign="top" align="left">Confirmed by Disbiome</td>
</tr> <tr>
<td valign="top" align="left">2</td>
<td valign="top" align="left"><italic>Parabacteroides</italic></td>
<td valign="top" align="left">PMID: 30 693 820, 32 010 563, 33 302 682, 33 302 682, 32 329 229, 30 693 820</td>
</tr> <tr>
<td valign="top" align="left">3</td>
<td valign="top" align="left"><italic>Diaphorobacter</italic></td>
<td valign="top" align="left">Confirmed by Disbiome</td>
</tr> <tr>
<td valign="top" align="left">4</td>
<td valign="top" align="left"><italic>Bifidobacterium</italic></td>
<td valign="top" align="left">Confirmed by Disbiome</td>
</tr> <tr>
<td valign="top" align="left">5</td>
<td valign="top" align="left"><italic>Roseburia</italic></td>
<td valign="top" align="left">PMID: 33 302 682, 32 227 387, 35 735 103</td>
</tr> <tr>
<td valign="top" align="left">6</td>
<td valign="top" align="left"><italic>Bacteroides</italic></td>
<td valign="top" align="left">PMID: 306 938 20, 36 498 063, 30 416 658,</td>
</tr> <tr>
<td valign="top" align="left">7</td>
<td valign="top" align="left"><italic>Lactobacillus</italic></td>
<td valign="top" align="left">PMID: 26 125 762, 36 361 537, 36 638 662</td>
</tr> <tr>
<td valign="top" align="left">8</td>
<td valign="top" align="left"><italic>Leptotrichia</italic></td>
<td valign="top" align="left">PMID: 34 432 217, 33 454 779</td>
</tr> <tr>
<td valign="top" align="left">9</td>
<td valign="top" align="left"><italic>Prevotella</italic></td>
<td valign="top" align="left">Confirmed by Disbiome</td>
</tr> <tr>
<td valign="top" align="left">10</td>
<td valign="top" align="left"><italic>Enterococcus</italic></td>
<td valign="top" align="left">PMID: 33 302 682, 27 717 798, 31 065 547, 33 111 503</td>
</tr> <tr>
<td valign="top" align="left">11</td>
<td valign="top" align="left"><italic>Streptococcus</italic></td>
<td valign="top" align="left">Confirmed by Disbiome</td>
</tr> <tr>
<td valign="top" align="left">12</td>
<td valign="top" align="left"><italic>Corynebacterium</italic></td>
<td valign="top" align="left">PMID: 350 388, 6 362 846, 6 998 933, 6 318 791</td>
</tr> <tr>
<td valign="top" align="left">13</td>
<td valign="top" align="left"><italic>Porphyromonas</italic></td>
<td valign="top" align="left">PMID: 33 279 803,32 615 270</td>
</tr> <tr>
<td valign="top" align="left">14</td>
<td valign="top" align="left"><italic>Alistipes</italic></td>
<td valign="top" align="left">PMID: 33 939 976, 34 793 492, 35 115 705</td>
</tr> <tr>
<td valign="top" align="left">15</td>
<td valign="top" align="left"><italic>Haemophilus</italic></td>
<td valign="top" align="left">PMID: 21 407 824, 21 407 824, 27 052 615, 21 098 042, 34 963 470</td>
</tr> <tr>
<td valign="top" align="left">16</td>
<td valign="top" align="left"><italic>Klebsiella</italic></td>
<td valign="top" align="left">PMID: 32 099 416, 24 706 703</td>
</tr> <tr>
<td valign="top" align="left">17</td>
<td valign="top" align="left"><italic>Dialister</italic></td>
<td valign="top" align="left">PMID: 30 416 658, 29 023 689, 34 063 829, 31 595 156</td>
</tr> <tr>
<td valign="top" align="left">18</td>
<td valign="top" align="left"><italic>Ruminococcus</italic></td>
<td valign="top" align="left">PMID: 32 227 387, 33 302 682, 36 737 654, 33 603 241, 32 240 032</td>
</tr> <tr>
<td valign="top" align="left">19</td>
<td valign="top" align="left"><italic>Pseudomonas</italic></td>
<td valign="top" align="left">PMID: 27 507 537, 25 801 231, 30 101 407</td>
</tr> <tr>
<td valign="top" align="left">20</td>
<td valign="top" align="left"><italic>Escherichia</italic></td>
<td valign="top" align="left">PMID: 18 496 688, 10.1158/1538-7445.AM2023-5185</td>
</tr></tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec>
<title>3.4. Discussion and conclusion</title>
<p>Systematic identification of associations between microbes and diseases significantly contributes to the understanding of the complex pathogenic mechanism of various diseases (Takahashi et al., <xref ref-type="bibr" rid="B69">2018</xref>; Zhou et al., <xref ref-type="bibr" rid="B92">2018</xref>; Yang et al., <xref ref-type="bibr" rid="B83">2022</xref>). In particular, computational pathogenic microorganism discovery helps to capture potential biomarkers from candidate compounds for human complex diseases (Barrows et al., <xref ref-type="bibr" rid="B2">2016</xref>; Zhu et al., <xref ref-type="bibr" rid="B94">2021</xref>).</p>
<p>Here, we developed a computational method called SAELGMDA to improve MDA prediction. First, microbe similarity and disease similarity were computed via their function similarity and GIPK similarity. Second, one microbe&#x02013;disease pair was represented as a feature vector based on microbe similarity matrix and disease similarity matrix. Third, the obtained high-dimensional features were mapped to a low-dimensional space based on a sparse autoencoder. Finally, unknown microbe&#x02013;disease pairs were classified using LightGBM.</p>
<p>Our proposed SAELGMDA method was compared with MNNMDA, GATMDA, LRLSHMDA, and NTSHMDA. Experimental results under <italic>CV</italic><sub>1</sub>, <italic>CV</italic><sub>2</sub>, and <italic>CV</italic><sub>3</sub> show that SAELGMDA outperforms the above four methods. SAELGMDA obtains the superior MDA identification ability. To investigate the MDA classification performance of LightGBM, we further compared it with XGBoost and NGBoost. The results demonstrate that LightGBM obtained better accuracy. Case studies demonstrate that there are possible associations between <italic>Clostridium coccoides</italic> and colorectal cancer, between Sphingomonadaceae and inflammatory bowel disease, and between <italic>Veillonella</italic> and autism and needs further validation.</p>
<p>We used two MDA databases (Disbiome and HMDAD) to investigate the performance of our proposed SAELGMDA method. The HMDAD dataset is a small dataset and Disbiome is a larger dataset. Under <italic>CV</italic><sub>1</sub>, the performance of SAELGMDA, GATMDA, and LRLSHMDA on the Disbiome dataset outperforms the ones on the HMDAD dataset, demonstrating more data contribute to the performance improvement for the three methods under <italic>CV</italic><sub>1</sub>. Under <italic>CV</italic><sub>2</sub> and <italic>CV</italic><sub>3</sub>, all five methods computed higher accuracy and AUC on the two datasets. However, MCC and AUPR computed by these five methods significantly decreased the Disbiome dataset compared with the HMDAD dataset. It may be caused by data imbalance; that is, the generalization ability of SAELGMDA is good when identifying potential associated microbes for a query disease. However, its generalization ability needs further improvement under <italic>CV</italic><sub>2</sub> and <italic>CV</italic><sub>3</sub>.</p>
<p>Although SAELGMDA outperformed the other four methods under the majority of condition on the HMDAD and Disbiome databases, the performance of all five MDA prediction methods, especially MCC and AUPR, remains an improvement. In future, we will integrate more biological data, such as microbe&#x02013;drug associations and disease&#x02013;gene associations, to extract effective features for microbe&#x02013;disease pairs. Furthermore, we will explore new dimensional reduction algorithms and classification models to improve MDA prediction by combining deep learning.</p>
</sec>
</sec>
<sec sec-type="data-availability" id="s4">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec sec-type="author-contributions" id="s5">
<title>Author contributions</title>
<p>FW and HY: conceptualization and validation. LP: funding acquisition. YW, LP, and XL: project administration. FW: writing&#x02014;original draft and software. HY, LP, and XL: writing&#x02014;reviewing and editing and investigation. FW and LP: methodology. All authors contributed to the article and approved the submitted version.</p>
</sec>
</body>
<back>
<sec sec-type="funding-information" id="s6">
<title>Funding</title>
<p>LP was supported by the National Natural Science Foundation of China under Grant No. 61803151.</p>
</sec>
<ack><p>We would like to thank all authors of the cited references.</p>
</ack>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>YW was employed by Geneis (Beijing) Co., Ltd., Beijing, China. The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s7">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Andrew</surname> <given-names>N.</given-names></name></person-group> (<year>2011</year>). <article-title>Sparse autoencoder</article-title>. <source>CS294A Lecture Notes</source> <volume>72</volume>, <fpage>1</fpage>&#x02013;<lpage>19</lpage>.</citation>
</ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Barrows</surname> <given-names>N. J.</given-names></name> <name><surname>Campos</surname> <given-names>R. K.</given-names></name> <name><surname>Powell</surname> <given-names>S. T.</given-names></name> <name><surname>Prasanth</surname> <given-names>K. R.</given-names></name> <name><surname>Schott-Lerner</surname> <given-names>G.</given-names></name> <name><surname>Soto-Acosta</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>A screen of FDA-approved drugs for inhibitors of Zika virus infection</article-title>. <source>Cell Host Microbe</source> <volume>20</volume>, <fpage>259</fpage>&#x02013;<lpage>270</lpage>. <pub-id pub-id-type="doi">10.1016/j.chom.2016.07.004</pub-id><pub-id pub-id-type="pmid">27476412</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Biller</surname> <given-names>L. H.</given-names></name> <name><surname>Schrag</surname> <given-names>D.</given-names></name></person-group> (<year>2021</year>). <article-title>Diagnosis and treatment of metastatic colorectal cancer: a review</article-title>. <source>JAMA</source> <volume>325</volume>, <fpage>669</fpage>&#x02013;<lpage>685</lpage>. <pub-id pub-id-type="doi">10.1001/jama.2021.0106</pub-id><pub-id pub-id-type="pmid">33591350</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bretthauer</surname> <given-names>M.</given-names></name> <name><surname>L&#x000F8;berg</surname> <given-names>M.</given-names></name> <name><surname>Wieszczy</surname> <given-names>P.</given-names></name> <name><surname>Kalager</surname> <given-names>M.</given-names></name> <name><surname>Emilsson</surname> <given-names>L.</given-names></name> <name><surname>Garborg</surname> <given-names>K.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Effect of colonoscopy screening on risks of colorectal cancer and related death</article-title>. <source>N. Engl. J. Med</source>. <volume>387</volume>, <fpage>1547</fpage>&#x02013;<lpage>1556</lpage>. <pub-id pub-id-type="doi">10.1056/NEJMoa2208375</pub-id><pub-id pub-id-type="pmid">36946845</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bull-Otterson</surname> <given-names>L.</given-names></name> <name><surname>Feng</surname> <given-names>W.</given-names></name> <name><surname>Kirpich</surname> <given-names>I.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Qin</surname> <given-names>X.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>Metagenomic analyses of alcohol induced pathogenic alterations in the intestinal microbiome and the effect of lactobacillus rhamnosus gg treatment</article-title>. <source>PLoS ONE</source> <volume>8</volume>, <fpage>e53028</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0053028</pub-id><pub-id pub-id-type="pmid">23326376</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>T.</given-names></name> <name><surname>Guestrin</surname> <given-names>C.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Xgboost: a scalable tree boosting system,&#x0201D;</article-title> in <source>Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>785</fpage>&#x02013;<lpage>794</lpage>. <pub-id pub-id-type="doi">10.1145/2939672.2939785</pub-id></citation>
</ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>T.-H.</given-names></name> <name><surname>Zhao</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>C.-C.</given-names></name> <name><surname>Zhu</surname> <given-names>C.-C.</given-names></name></person-group> (<year>2021</year>). <article-title>Deep-belief network for predicting potential mirna-disease associations</article-title>. <source>Brief. Bioinformatics</source> <volume>22</volume>, <fpage>bbaa186</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbaa186</pub-id><pub-id pub-id-type="pmid">34020550</pub-id></citation></ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>X.</given-names></name> <name><surname>Xie</surname> <given-names>D.</given-names></name> <name><surname>Zhao</surname> <given-names>Q.</given-names></name> <name><surname>You</surname> <given-names>Z.-H.</given-names></name></person-group> (<year>2019</year>). <article-title>Micrornas and complex diseases: from experimental results to computational models</article-title>. <source>Brief. Bioinformatics</source> <volume>20</volume>, <fpage>515</fpage>&#x02013;<lpage>539</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbx130</pub-id><pub-id pub-id-type="pmid">29045685</pub-id></citation></ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Lei</surname> <given-names>X.</given-names></name></person-group> (<year>2022</year>). <article-title>Metapath aggregated graph neural network and tripartite heterogeneous networks for microbe-disease prediction</article-title>. <source>Front. Microbiol</source>. <volume>13</volume>, <fpage>919380</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2022.919380</pub-id><pub-id pub-id-type="pmid">35711758</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Cheng</surname> <given-names>E.</given-names></name> <name><surname>Zhao</surname> <given-names>J.</given-names></name> <name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Song</surname> <given-names>S.</given-names></name> <name><surname>Xiong</surname> <given-names>S.</given-names></name> <name><surname>Sun</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>&#x0201C;Dual network contrastive learning for predicting microbe-disease associations,&#x0201D;</article-title> in <source>IEEE/ACM Transactions on Computational Biology and Bioinformatics</source> (<publisher-loc>New Jersey, NJ</publisher-loc>: <publisher-name>IEEE</publisher-name>). <pub-id pub-id-type="doi">10.1109/TCBB.2022.3228617</pub-id></citation>
</ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cheng</surname> <given-names>L.</given-names></name> <name><surname>Qi</surname> <given-names>C.</given-names></name> <name><surname>Zhuang</surname> <given-names>H.</given-names></name> <name><surname>Fu</surname> <given-names>T.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name></person-group> (<year>2020</year>). <article-title>gutmdisorder: a comprehensive database for dysbiosis of the gut microbiota in disorders and interventions</article-title>. <source>Nucleic Acids Res</source>. <volume>48</volume>, <fpage>D554</fpage>&#x02013;<lpage>D560</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkz843</pub-id><pub-id pub-id-type="pmid">32515792</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chicco</surname> <given-names>D.</given-names></name> <name><surname>Jurman</surname> <given-names>G.</given-names></name></person-group> (<year>2020</year>). <article-title>The advantages of the Matthews correlation coefficient (MCC) over f1 score and accuracy in binary classification evaluation</article-title>. <source>BMC Genom</source>. <volume>21</volume>, <fpage>1</fpage>&#x02013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1186/s12864-019-6413-7</pub-id><pub-id pub-id-type="pmid">31898477</pub-id></citation></ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dai</surname> <given-names>H.</given-names></name> <name><surname>Gao</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>D.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Cui</surname> <given-names>Y.</given-names></name> <name><surname>Zhao</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Family sphingomonadaceae as the key executor of triclosan degradation in both nitrification and denitrification systems</article-title>. <source>Chem. Eng. J</source>. <volume>442</volume>, <fpage>1362021</fpage>. <pub-id pub-id-type="doi">10.1016/j.cej.2022.136202</pub-id></citation>
</ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>de Vries</surname> <given-names>H. J.</given-names></name> <name><surname>Beyer</surname> <given-names>F.</given-names></name> <name><surname>Jarzembowska</surname> <given-names>M.</given-names></name> <name><surname>Lipi&#x00144;ska</surname> <given-names>J.</given-names></name> <name><surname>van den Brink</surname> <given-names>P.</given-names></name> <name><surname>Zwijnenburg</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Isolation and characterization of sphingomonadaceae from fouled membranes</article-title>. <source>NPJ Biofilms Microbiomes</source> <volume>5</volume>, <fpage>6</fpage>. <pub-id pub-id-type="doi">10.1038/s41522-018-0074-1</pub-id><pub-id pub-id-type="pmid">30701078</pub-id></citation></ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Decker</surname> <given-names>B. K.</given-names></name> <name><surname>Palmore</surname> <given-names>T. N.</given-names></name></person-group> (<year>2014</year>). <article-title>Hospital water and opportunities for infection prevention</article-title>. <source>Curr. Infect. Dis. Rep</source>. <volume>16</volume>, <fpage>1</fpage>&#x02013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1007/s11908-014-0432-y</pub-id><pub-id pub-id-type="pmid">25217106</pub-id></citation></ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Demirci</surname> <given-names>M.</given-names></name> <name><surname>Tokman</surname> <given-names>H.</given-names></name> <name><surname>Uysal</surname> <given-names>H.</given-names></name> <name><surname>Demiryas</surname> <given-names>S.</given-names></name> <name><surname>Karakullukcu</surname> <given-names>A.</given-names></name> <name><surname>Saribas</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Reduced <italic>Akkermansia muciniphila</italic> and <italic>Faecalibacterium prausnitzii</italic> levels in the gut microbiota of children with allergic asthma</article-title>. <source>Allergol. Immunopathol</source>. <volume>47</volume>, <fpage>365</fpage>&#x02013;<lpage>371</lpage>. <pub-id pub-id-type="doi">10.1016/j.aller.2018.12.009</pub-id><pub-id pub-id-type="pmid">30765132</pub-id></citation></ref>
<ref id="B17">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Duan</surname> <given-names>T.</given-names></name> <name><surname>Anand</surname> <given-names>A.</given-names></name> <name><surname>Ding</surname> <given-names>D. Y.</given-names></name> <name><surname>Thai</surname> <given-names>K. K.</given-names></name> <name><surname>Basu</surname> <given-names>S.</given-names></name> <name><surname>Ng</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>&#x0201C;Ngboost: natural gradient boosting for probabilistic prediction,&#x0201D;</article-title> in <source>International Conference on Machine Learning</source> (<publisher-loc>Vienna</publisher-loc>: <publisher-name>The International Machine Learning Society</publisher-name>), <fpage>2690</fpage>&#x02013;<lpage>2700</lpage>. PMLR.<pub-id pub-id-type="pmid">34185055</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>El Mouzan</surname> <given-names>M. I.</given-names></name> <name><surname>Winter</surname> <given-names>H. S.</given-names></name> <name><surname>Assiri</surname> <given-names>A. A.</given-names></name> <name><surname>Korolev</surname> <given-names>K. S.</given-names></name> <name><surname>Al Sarkhy</surname> <given-names>A. A.</given-names></name> <name><surname>Dowd</surname> <given-names>S. E.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Microbiota profile in new-onset pediatric crohn&#x00027;s disease: data from a non-western population</article-title>. <source>Gut Pathog</source>. <volume>10</volume>, <fpage>1</fpage>&#x02013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1186/s13099-018-0276-3</pub-id><pub-id pub-id-type="pmid">30519287</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>Z.</given-names></name> <name><surname>Guo</surname> <given-names>B.</given-names></name> <name><surname>Gao</surname> <given-names>R.</given-names></name> <name><surname>Zhu</surname> <given-names>Q.</given-names></name> <name><surname>Qin</surname> <given-names>H.</given-names></name></person-group> (<year>2015</year>). <article-title>Microbiota disbiosis is associated with colorectal cancer</article-title>. <source>Front. Microbiol</source>. <volume>6</volume>, <fpage>20</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2015.00020</pub-id><pub-id pub-id-type="pmid">25699023</pub-id></citation></ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Grenda</surname> <given-names>T.</given-names></name> <name><surname>Grenda</surname> <given-names>A.</given-names></name> <name><surname>Domaradzki</surname> <given-names>P.</given-names></name> <name><surname>Krawczyk</surname> <given-names>P.</given-names></name> <name><surname>Kwiatek</surname> <given-names>K.</given-names></name></person-group> (<year>2022</year>). <article-title>Probiotic potential of <italic>Clostridium</italic> spp.&#x02014;advantages and doubts</article-title>. <source>Curr. Issues Mol. Biol</source>. <volume>44</volume>, <fpage>3118</fpage>&#x02013;<lpage>3130</lpage>. <pub-id pub-id-type="doi">10.3390/cimb44070215</pub-id><pub-id pub-id-type="pmid">35877439</pub-id></citation></ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gronow</surname> <given-names>S.</given-names></name> <name><surname>Welnitz</surname> <given-names>S.</given-names></name> <name><surname>Lapidus</surname> <given-names>A.</given-names></name> <name><surname>Nolan</surname> <given-names>M.</given-names></name> <name><surname>Ivanova</surname> <given-names>N.</given-names></name> <name><surname>Glavina Del Rio</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2010</year>). <article-title>Complete genome sequence of <italic>Veillonella parvula</italic> type strain (te3t). <italic>Stand. Genomic Sci</italic></article-title>., <volume>2</volume>, <fpage>57</fpage>&#x02013;<lpage>65</lpage>. <pub-id pub-id-type="doi">10.4056/sigs.521107</pub-id><pub-id pub-id-type="pmid">21304678</pub-id></citation></ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Guarner</surname> <given-names>F.</given-names></name> <name><surname>Malagelada</surname> <given-names>J.-R.</given-names></name></person-group> (<year>2003</year>). <article-title>Gut flora in health and disease</article-title>. <source>Lancet</source> <volume>361</volume>, <fpage>512</fpage>&#x02013;<lpage>519</lpage>. <pub-id pub-id-type="doi">10.1016/S0140-6736(03)12489-0</pub-id><pub-id pub-id-type="pmid">12583961</pub-id></citation></ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Guo</surname> <given-names>S.-S.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name> <name><surname>Zhou</surname> <given-names>X.-G.</given-names></name> <name><surname>Zhang</surname> <given-names>G.-J.</given-names></name></person-group> (<year>2022</year>). <article-title>Deepumqa: ultrafast shape recognition-based protein model quality assessment using deep learning</article-title>. <source>Bioinformatics</source> <volume>38</volume>, <fpage>1895</fpage>&#x02013;<lpage>1903</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btac056</pub-id><pub-id pub-id-type="pmid">35134108</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>B.-S.</given-names></name> <name><surname>Peng</surname> <given-names>L.-H.</given-names></name> <name><surname>Li</surname> <given-names>Z.</given-names></name></person-group> (<year>2018</year>). <article-title>Human microbe-disease association prediction with graph regularized non-negative matrix factorization</article-title>. <source>Front. Microbiol</source>. <volume>9</volume>, <fpage>2560</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2018.02560</pub-id><pub-id pub-id-type="pmid">30443240</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Heidarian</surname> <given-names>F.</given-names></name> <name><surname>Alebouyeh</surname> <given-names>M.</given-names></name> <name><surname>Shahrokh</surname> <given-names>S.</given-names></name> <name><surname>Balaii</surname> <given-names>H.</given-names></name> <name><surname>Zali</surname> <given-names>M. R.</given-names></name></person-group> (<year>2019</year>). <article-title>Altered fecal bacterial composition correlates with disease activity in inflammatory bowel disease and the extent of il8 induction</article-title>. <source>Curr. Res. Transl. Med</source>. <volume>67</volume>, <fpage>41</fpage>&#x02013;<lpage>50</lpage>. <pub-id pub-id-type="doi">10.1016/j.retram.2019.01.002</pub-id><pub-id pub-id-type="pmid">30685379</pub-id></citation></ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>H.</given-names></name> <name><surname>Feng</surname> <given-names>Z.</given-names></name> <name><surname>Lin</surname> <given-names>H.</given-names></name> <name><surname>Cheng</surname> <given-names>J.</given-names></name> <name><surname>Lyu</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Gene function and cell surface protein association analysis based on single-cell multiomics data</article-title>. <source>Comput. Biol. Med</source>., <volume>157</volume>, <fpage>106733</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2023.106733</pub-id><pub-id pub-id-type="pmid">36924730</pub-id></citation></ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hua</surname> <given-names>M.</given-names></name> <name><surname>Yu</surname> <given-names>S.</given-names></name> <name><surname>Liu</surname> <given-names>T.</given-names></name> <name><surname>Yang</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>H.</given-names></name></person-group> (<year>2022</year>). <article-title>MVGCNMDA: multi-view graph augmentation convolutional network for uncovering disease-related microbes</article-title>. <source>Interdiscip. Sci. Comput. Life Sci</source>. <volume>14</volume>, <fpage>669</fpage>&#x02013;<lpage>682</lpage>. <pub-id pub-id-type="doi">10.1007/s12539-022-00514-2</pub-id><pub-id pub-id-type="pmid">35428964</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hwang</surname> <given-names>S.</given-names></name> <name><surname>Kim</surname> <given-names>C. Y.</given-names></name> <name><surname>Yang</surname> <given-names>S.</given-names></name> <name><surname>Kim</surname> <given-names>E.</given-names></name> <name><surname>Hart</surname> <given-names>T.</given-names></name> <name><surname>Marcotte</surname> <given-names>E. M.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Humannet v2: human gene networks for disease research</article-title>. <source>Nucleic Acids Res</source>. <volume>47</volume>, <fpage>D573</fpage>&#x02013;<lpage>D580</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gky1126</pub-id><pub-id pub-id-type="pmid">30418591</pub-id></citation></ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Janssens</surname> <given-names>Y.</given-names></name> <name><surname>Nielandt</surname> <given-names>J.</given-names></name> <name><surname>Bronselaer</surname> <given-names>A.</given-names></name> <name><surname>Debunne</surname> <given-names>N.</given-names></name> <name><surname>Verbeke</surname> <given-names>F.</given-names></name> <name><surname>Wynendaele</surname> <given-names>E.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Disbiome database: linking the microbiome to disease</article-title>. <source>BMC Microbiol</source>. <volume>18</volume>, <fpage>1</fpage>&#x02013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1186/s12866-018-1197-5</pub-id><pub-id pub-id-type="pmid">29866037</pub-id></citation></ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jiang</surname> <given-names>C.</given-names></name> <name><surname>Tang</surname> <given-names>M.</given-names></name> <name><surname>Jin</surname> <given-names>S.</given-names></name> <name><surname>Huang</surname> <given-names>W.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name></person-group> (<year>2022</year>). <article-title>Kgnmda: a knowledge graph neural network method for predicting microbe-disease associations</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinform</source>. <volume>20</volume>, <fpage>1147</fpage>&#x02013;<lpage>1155</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2022.3184362</pub-id><pub-id pub-id-type="pmid">35724280</pub-id></citation></ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kamneva</surname> <given-names>O. K.</given-names></name></person-group> (<year>2017</year>). <article-title>Genome composition and phylogeny of microbes predict their co-occurrence in the environment</article-title>. <source>PLoS Comput. Biol</source>. <volume>13</volume>, <fpage>e1005366</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1005366</pub-id><pub-id pub-id-type="pmid">28152007</pub-id></citation></ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kaplan</surname> <given-names>G. G.</given-names></name> <name><surname>Windsor</surname> <given-names>J. W.</given-names></name></person-group> (<year>2021</year>). <article-title>The four epidemiological stages in the global evolution of inflammatory bowel disease</article-title>. <source>Nat. Rev. Gastroenterol. Hepatol</source>. <volume>18</volume>, <fpage>56</fpage>&#x02013;<lpage>66</lpage>. <pub-id pub-id-type="doi">10.1038/s41575-020-00360-x</pub-id><pub-id pub-id-type="pmid">33033392</pub-id></citation></ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kasai</surname> <given-names>C.</given-names></name> <name><surname>Sugimoto</surname> <given-names>K.</given-names></name> <name><surname>Moritani</surname> <given-names>I.</given-names></name> <name><surname>Tanaka</surname> <given-names>J.</given-names></name> <name><surname>Oya</surname> <given-names>Y.</given-names></name> <name><surname>Inoue</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>Comparison of human gut microbiota in control subjects and patients with colorectal carcinoma in adenoma: terminal restriction fragment length polymorphism and next-generation sequencing analyses</article-title>. <source>Oncol. Rep</source>. <volume>35</volume>, <fpage>325</fpage>&#x02013;<lpage>333</lpage>. <pub-id pub-id-type="doi">10.3892/or.2015.4398</pub-id><pub-id pub-id-type="pmid">26549775</pub-id></citation></ref>
<ref id="B34">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ke</surname> <given-names>G.</given-names></name> <name><surname>Meng</surname> <given-names>Q.</given-names></name> <name><surname>Finley</surname> <given-names>T.</given-names></name> <name><surname>Wang</surname> <given-names>T.</given-names></name> <name><surname>Chen</surname> <given-names>W.</given-names></name> <name><surname>Ma</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>&#x0201C;LightGBM: a highly efficient gradient boosting decision tree,&#x0201D;</article-title> in Advances in <source>Neural Information Processing Systems, Vol. 30</source> (<publisher-loc>Long Beach, CA</publisher-loc>: <publisher-name>MIT Press</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>9</lpage>.</citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kingma</surname> <given-names>D. P.</given-names></name> <name><surname>Ba</surname> <given-names>J.</given-names></name></person-group> (<year>2014</year>). <article-title>ADAM: a method for stochastic optimization</article-title>. <source>arXiv</source> [preprint]. <pub-id pub-id-type="doi">10.48550/arXiv.1412.6980</pub-id></citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lewis</surname> <given-names>J. D.</given-names></name> <name><surname>Chen</surname> <given-names>E. Z.</given-names></name> <name><surname>Baldassano</surname> <given-names>R. N.</given-names></name> <name><surname>Otley</surname> <given-names>A. R.</given-names></name> <name><surname>Griffiths</surname> <given-names>A. M.</given-names></name> <name><surname>Lee</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Inflammation, antibiotics, and diet as environmental stressors of the gut microbiome in pediatric crohn&#x00027;s disease</article-title>. <source>Cell Host Microbe</source> <volume>18</volume>, <fpage>489</fpage>&#x02013;<lpage>500</lpage>. <pub-id pub-id-type="doi">10.1016/j.chom.2015.09.008</pub-id><pub-id pub-id-type="pmid">28799909</pub-id></citation></ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>S.</given-names></name> <name><surname>Xie</surname> <given-names>M.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name></person-group> (<year>2019</year>). <article-title>A novel approach based on bipartite network recommendation and katz model to predict potential micro-disease associations</article-title>. <source>Front. Genet</source>. <volume>10</volume>, <fpage>1147</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2019.01147</pub-id><pub-id pub-id-type="pmid">31803235</pub-id></citation></ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>T.-H.</given-names></name> <name><surname>Wang</surname> <given-names>C.-C.</given-names></name> <name><surname>Zhang</surname> <given-names>L.</given-names></name> <name><surname>Chen</surname> <given-names>X.</given-names></name></person-group> (<year>2023</year>). <article-title>Snrmpacdc: computational model focused on siamese network and random matrix projection for anticancer synergistic drug combination prediction</article-title>. <source>Brief. Bioinformatics</source> <volume>24</volume>, <fpage>bbac503</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbac503</pub-id><pub-id pub-id-type="pmid">36418927</pub-id></citation></ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liang</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.-Q.</given-names></name> <name><surname>Liu</surname> <given-names>N.-N.</given-names></name> <name><surname>Wu</surname> <given-names>Y.-N.</given-names></name> <name><surname>Gu</surname> <given-names>C.-L.</given-names></name> <name><surname>Wang</surname> <given-names>Y.-L.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Magcnse: predicting lncrna-disease associations using multi-view attention graph convolutional network and stacking ensemble model</article-title>. <source>BMC Bioinformatics</source> <volume>23</volume>, <fpage>1</fpage>&#x02013;<lpage>22</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-022-04715-w</pub-id><pub-id pub-id-type="pmid">35590258</pub-id></citation></ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lihong</surname> <given-names>P.</given-names></name> <name><surname>Wang</surname> <given-names>C.</given-names></name> <name><surname>Tian</surname> <given-names>X.</given-names></name> <name><surname>Zhou</surname> <given-names>L.</given-names></name> <name><surname>Li</surname> <given-names>K.</given-names></name></person-group> (<year>2021</year>). <article-title>Finding lncRNA-protein interactions based on deep learning with dual-net neural architecture</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinform</source>. <volume>19</volume>, <fpage>3456</fpage>&#x02013;<lpage>3468</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2021.3116232</pub-id><pub-id pub-id-type="pmid">34587091</pub-id></citation></ref>
<ref id="B41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>D.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name> <name><surname>Luo</surname> <given-names>Y.</given-names></name> <name><surname>He</surname> <given-names>Q.</given-names></name> <name><surname>Deng</surname> <given-names>L.</given-names></name></person-group> (<year>2021</year>). <article-title>MGATMDA: predicting microbe-disease associations via multi-component graph attention network</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinformatics</source> <volume>19</volume>, <fpage>3578</fpage>&#x02013;<lpage>3585</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2021.3116318</pub-id><pub-id pub-id-type="pmid">34587092</pub-id></citation></ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>H.</given-names></name> <name><surname>Bing</surname> <given-names>P.</given-names></name> <name><surname>Zhang</surname> <given-names>M.</given-names></name> <name><surname>Tian</surname> <given-names>G.</given-names></name> <name><surname>Ma</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>MNNMDA: predicting human microbe-disease association via a method to minimize matrix nuclear norm</article-title>. <source>Comput Struct. Biotechnol. J</source>. <volume>21</volume>, <fpage>1414</fpage>&#x02013;<lpage>1423</lpage>. <pub-id pub-id-type="doi">10.1016/j.csbj.2022.12.053</pub-id><pub-id pub-id-type="pmid">36824227</pub-id></citation></ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>J.</given-names></name> <name><surname>Zhao</surname> <given-names>K.</given-names></name> <name><surname>Zhang</surname> <given-names>G.</given-names></name></person-group> (<year>2023</year>). <article-title>Improved model quality assessment using sequence and structural information by enhanced deep neural networks</article-title>. <source>Brief. Bioinformatics</source> <volume>24</volume>, <fpage>bbac507</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbac507</pub-id><pub-id pub-id-type="pmid">36460624</pub-id></citation></ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>S.-L.</given-names></name> <name><surname>Zhang</surname> <given-names>J.-F.</given-names></name> <name><surname>Zhang</surname> <given-names>W.</given-names></name> <name><surname>Zhou</surname> <given-names>S.</given-names></name> <name><surname>Li</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>DMFMDA: prediction of microbe-disease associations based on deep matrix factorization using bayesian personalized ranking</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinformatics</source> <volume>18</volume>, <fpage>1763</fpage>&#x02013;<lpage>1772</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2020.3018138</pub-id><pub-id pub-id-type="pmid">32816678</pub-id></citation></ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lomax</surname> <given-names>A. E.</given-names></name> <name><surname>Linden</surname> <given-names>D. R.</given-names></name> <name><surname>Mawe</surname> <given-names>G. M.</given-names></name> <name><surname>Sharkey</surname> <given-names>K. A.</given-names></name></person-group> (<year>2006</year>). <article-title>Effects of gastrointestinal inflammation on enteroendocrine cells and enteric neural reflex circuits</article-title>. <source>Auton. Neurosci</source> <volume>126</volume>, <fpage>250</fpage>&#x02013;<lpage>257</lpage>. <pub-id pub-id-type="doi">10.1016/j.autneu.2006.02.015</pub-id><pub-id pub-id-type="pmid">16616704</pub-id></citation></ref>
<ref id="B46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Long</surname> <given-names>Y.</given-names></name> <name><surname>Luo</surname> <given-names>J.</given-names></name></person-group> (<year>2019</year>). <article-title>Wmghmda: a novel weighted meta-graph-based model for predicting human microbe-disease association on heterogeneous information network</article-title>. <source>BMC Bioinformatics</source> <volume>20</volume>, <fpage>1</fpage>&#x02013;<lpage>18</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-019-3066-0</pub-id><pub-id pub-id-type="pmid">31675979</pub-id></citation></ref>
<ref id="B47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Long</surname> <given-names>Y.</given-names></name> <name><surname>Luo</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Xia</surname> <given-names>Y.</given-names></name></person-group> (<year>2021</year>). <article-title>Predicting human microbe-disease associations via graph attention networks with inductive matrix completion</article-title>. <source>Brief. Bioinformatics</source> <volume>22</volume>, <fpage>bbaa146</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbaa146</pub-id><pub-id pub-id-type="pmid">32725163</pub-id></citation></ref>
<ref id="B48">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Luo</surname> <given-names>J.</given-names></name> <name><surname>Long</surname> <given-names>Y.</given-names></name></person-group> (<year>2018</year>). <article-title>NTSHMDA: prediction of human microbe-disease association based on random walk by integrating network topological similarity</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinformatics</source> <volume>17</volume>, <fpage>1341</fpage>&#x02013;<lpage>1351</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2018.2883041</pub-id><pub-id pub-id-type="pmid">30489271</pub-id></citation></ref>
<ref id="B49">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lynch</surname> <given-names>S. V.</given-names></name> <name><surname>Pedersen</surname> <given-names>O.</given-names></name></person-group> (<year>2016</year>). <article-title>The human intestinal microbiome in health and disease</article-title>. <source>N. Engl. J. Med</source>. <volume>375</volume>, <fpage>2369</fpage>&#x02013;<lpage>2379</lpage>. <pub-id pub-id-type="doi">10.1056/NEJMra1600266</pub-id><pub-id pub-id-type="pmid">32675857</pub-id></citation></ref>
<ref id="B50">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ma</surname> <given-names>W.</given-names></name> <name><surname>Zhang</surname> <given-names>L.</given-names></name> <name><surname>Zeng</surname> <given-names>P.</given-names></name> <name><surname>Huang</surname> <given-names>C.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Geng</surname> <given-names>B.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>An analysis of human microbe-disease associations</article-title>. <source>Brief. Bioinformatics</source> <volume>18</volume>, <fpage>85</fpage>&#x02013;<lpage>97</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbw005</pub-id><pub-id pub-id-type="pmid">26883326</pub-id></citation></ref>
<ref id="B51">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mak</surname> <given-names>W. Y.</given-names></name> <name><surname>Zhao</surname> <given-names>M.</given-names></name> <name><surname>Ng</surname> <given-names>S. C.</given-names></name> <name><surname>Burisch</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). <article-title>The epidemiology of inflammatory bowel disease: east meets west</article-title>. <source>J. Gastroenterol. Hepatol</source>. <volume>35</volume>, <fpage>380</fpage>&#x02013;<lpage>389</lpage>. <pub-id pub-id-type="doi">10.1111/jgh.14872</pub-id><pub-id pub-id-type="pmid">31596960</pub-id></citation></ref>
<ref id="B52">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Makhzani</surname> <given-names>A.</given-names></name> <name><surname>Frey</surname> <given-names>B.</given-names></name></person-group> (<year>2013</year>). <article-title>K-sparse autoencoders</article-title>. <source>arXiv</source>. [preprint]. <pub-id pub-id-type="doi">10.48550/arXiv.1312.5663</pub-id></citation>
</ref>
<ref id="B53">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>M&#x000E1;rmol</surname> <given-names>I.</given-names></name> <name><surname>S&#x000E1;nchez-de Diego</surname> <given-names>C.</given-names></name> <name><surname>Pradilla Dieste</surname> <given-names>A.</given-names></name> <name><surname>Cerrada</surname> <given-names>E.</given-names></name> <name><surname>Rodriguez Yoldi</surname> <given-names>M. J.</given-names></name></person-group> (<year>2017</year>). <article-title>Colorectal carcinoma: a general overview and future perspectives in colorectal cancer</article-title>. <source>Int. J. Mol. Sci</source>. <volume>18</volume>, <fpage>197</fpage>. <pub-id pub-id-type="doi">10.3390/ijms18010197</pub-id><pub-id pub-id-type="pmid">28106826</pub-id></citation></ref>
<ref id="B54">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>M&#x000FC;ller</surname> <given-names>C.</given-names></name> <name><surname>Macpherson</surname> <given-names>A.</given-names></name></person-group> (<year>2006</year>). <article-title>Layers of mutualism with commensal bacteria protect us from intestinal inflammation</article-title>. <source>Gut</source> <volume>55</volume>, <fpage>276</fpage>&#x02013;<lpage>284</lpage>. <pub-id pub-id-type="doi">10.1136/gut.2004.054098</pub-id><pub-id pub-id-type="pmid">16407387</pub-id></citation></ref>
<ref id="B55">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Oh</surname> <given-names>S.</given-names></name> <name><surname>Choi</surname> <given-names>D.</given-names></name></person-group> (<year>2019</year>). <article-title>Microbial community enhances biodegradation of bisphenol a through selection of sphingomonadaceae</article-title>. <source>Microb. Ecol</source>. <volume>77</volume>, <fpage>631</fpage>&#x02013;<lpage>639</lpage>. <pub-id pub-id-type="doi">10.1007/s00248-018-1263-4</pub-id><pub-id pub-id-type="pmid">30251120</pub-id></citation></ref>
<ref id="B56">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Peng</surname> <given-names>L.</given-names></name> <name><surname>Shen</surname> <given-names>L.</given-names></name> <name><surname>Liao</surname> <given-names>L.</given-names></name> <name><surname>Liu</surname> <given-names>G.</given-names></name> <name><surname>Zhou</surname> <given-names>L.</given-names></name></person-group> (<year>2020</year>). <article-title>RNMFMDA: a microbe-disease association identification method based on reliable negative sample selection and logistic matrix factorization with neighborhood regularization</article-title>. <source>Front. Microbiol</source>. <volume>11</volume>, <fpage>592430</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2020.592430</pub-id><pub-id pub-id-type="pmid">33193260</pub-id></citation></ref>
<ref id="B57">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Peng</surname> <given-names>L.</given-names></name> <name><surname>Wang</surname> <given-names>C.</given-names></name> <name><surname>Tian</surname> <given-names>G.</given-names></name> <name><surname>Liu</surname> <given-names>G.</given-names></name> <name><surname>Li</surname> <given-names>G.</given-names></name> <name><surname>Lu</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2022a</year>). <article-title>Analysis of CT scan images for covid-19 pneumonia based on a deep ensemble framework with densenet, swin transformer, and regnet</article-title>. <source>Front. Microbiol</source>. <volume>13</volume>, <fpage>993523</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2022.995323</pub-id><pub-id pub-id-type="pmid">36212877</pub-id></citation></ref>
<ref id="B58">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Peng</surname> <given-names>L.</given-names></name> <name><surname>Wang</surname> <given-names>F.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Tan</surname> <given-names>J.</given-names></name> <name><surname>Huang</surname> <given-names>L.</given-names></name> <name><surname>Tian</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2022b</year>). <article-title>Cell-cell communication inference and analysis in the tumour microenvironments from single-cell transcriptomics: data resources and computational strategies</article-title>. <source>Brief. Bioinformatics</source> <volume>23</volume>, <fpage>bbac234</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbac234</pub-id><pub-id pub-id-type="pmid">35753695</pub-id></citation></ref>
<ref id="B59">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Prosberg</surname> <given-names>M.</given-names></name> <name><surname>Bendtsen</surname> <given-names>F.</given-names></name> <name><surname>Vind</surname> <given-names>I.</given-names></name> <name><surname>Petersen</surname> <given-names>A. M.</given-names></name> <name><surname>Gluud</surname> <given-names>L. L.</given-names></name></person-group> (<year>2016</year>). <article-title>The association between the gut microbiota and the inflammatory bowel disease activity: a systematic review and meta-analysis</article-title>. <source>Scand. J. Gastroenterol</source>. <volume>51</volume>, <fpage>1407</fpage>&#x02013;<lpage>1415</lpage>. <pub-id pub-id-type="doi">10.1080/00365521.2016.1216587</pub-id><pub-id pub-id-type="pmid">27687331</pub-id></citation></ref>
<ref id="B60">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Saeed</surname> <given-names>M.</given-names></name> <name><surname>Shoaib</surname> <given-names>A.</given-names></name> <name><surname>Kandimalla</surname> <given-names>R.</given-names></name> <name><surname>Javed</surname> <given-names>S.</given-names></name> <name><surname>Almatroudi</surname> <given-names>A.</given-names></name> <name><surname>Gupta</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Microbe-based therapies for colorectal cancer: advantages and limitations</article-title>. <source>Semin. Cancer Biol</source>. <volume>86</volume>(<issue>Pt 3</issue>), <fpage>652</fpage>&#x02013;<lpage>665</lpage>. <pub-id pub-id-type="doi">10.1016/j.semcancer.2021.05.018</pub-id><pub-id pub-id-type="pmid">34020027</pub-id></citation></ref>
<ref id="B61">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shen</surname> <given-names>L.</given-names></name> <name><surname>Liu</surname> <given-names>F.</given-names></name> <name><surname>Huang</surname> <given-names>L.</given-names></name> <name><surname>Liu</surname> <given-names>G.</given-names></name> <name><surname>Zhou</surname> <given-names>L.</given-names></name> <name><surname>Peng</surname> <given-names>L.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>VDA-RWLRLS: an anti-sars-cov-2 drug prioritizing framework combining an unbalanced bi-random walk and laplacian regularized least squares</article-title>. <source>Comput. Biol. Med</source>. <volume>140</volume>, <fpage>105119</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2021.105119</pub-id><pub-id pub-id-type="pmid">34902608</pub-id></citation></ref>
<ref id="B62">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shen</surname> <given-names>S.</given-names></name> <name><surname>Anazawa</surname> <given-names>T.</given-names></name> <name><surname>Matsuda</surname> <given-names>T.</given-names></name> <name><surname>Shimizu</surname> <given-names>Y.</given-names></name></person-group> (<year>2022</year>). <article-title>Draft genome sequences of Sphingomonadaceae strains isolated from a freshwater lake</article-title>. <source>Microbiol. Resour. Announc</source>. <volume>11</volume>, e00070-22. <pub-id pub-id-type="doi">10.1128/mra.00070-22</pub-id><pub-id pub-id-type="pmid">35384702</pub-id></citation></ref>
<ref id="B63">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shi</surname> <given-names>J.-Y.</given-names></name> <name><surname>Huang</surname> <given-names>H.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.-N.</given-names></name> <name><surname>Cao</surname> <given-names>J.-B.</given-names></name> <name><surname>Yiu</surname> <given-names>S.-M.</given-names></name></person-group> (<year>2018</year>). <article-title>Bmcmda: a novel model for predicting human microbe-disease associations via binary matrix completion</article-title>. <source>BMC Bioinformatics</source> <volume>19</volume>, <fpage>85</fpage>&#x02013;<lpage>92</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-018-2274-3</pub-id><pub-id pub-id-type="pmid">30367598</pub-id></citation></ref>
<ref id="B64">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Siegel</surname> <given-names>R. L.</given-names></name> <name><surname>Miller</surname> <given-names>K. D.</given-names></name> <name><surname>Fuchs</surname> <given-names>H. E.</given-names></name> <name><surname>Jemal</surname> <given-names>A.</given-names></name></person-group> (<year>2022</year>). <article-title>Cancer statistics, 2022</article-title>. <source>CA Cancer J. Clin</source>. <volume>72</volume>, <fpage>7</fpage>&#x02013;<lpage>33</lpage>. <pub-id pub-id-type="doi">10.3322/caac.21708</pub-id><pub-id pub-id-type="pmid">35020204</pub-id></citation></ref>
<ref id="B65">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sokol</surname> <given-names>H.</given-names></name> <name><surname>Seksik</surname> <given-names>P.</given-names></name> <name><surname>Furet</surname> <given-names>J.</given-names></name> <name><surname>Firmesse</surname> <given-names>O.</given-names></name> <name><surname>Nion-Larmurier</surname> <given-names>I.</given-names></name> <name><surname>Beaugerie</surname> <given-names>L.</given-names></name> <etal/></person-group>. (<year>2009</year>). <article-title>Low counts of faecalibacterium prausnitzii in colitis microbiota</article-title>. <source>Inflamm. Bowel Dis</source>. <volume>15</volume>, <fpage>1183</fpage>&#x02013;<lpage>1189</lpage>. <pub-id pub-id-type="doi">10.1002/ibd.20903</pub-id><pub-id pub-id-type="pmid">19235886</pub-id></citation></ref>
<ref id="B66">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Strati</surname> <given-names>F.</given-names></name> <name><surname>Cavalieri</surname> <given-names>D.</given-names></name> <name><surname>Albanese</surname> <given-names>D.</given-names></name> <name><surname>De Felice</surname> <given-names>C.</given-names></name> <name><surname>Donati</surname> <given-names>C.</given-names></name> <name><surname>Hayek</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>New evidences on the altered gut microbiota in autism spectrum disorders</article-title>. <source>Microbiome</source> <volume>5</volume>, <fpage>1</fpage>&#x02013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1186/s40168-017-0242-1</pub-id><pub-id pub-id-type="pmid">28222761</pub-id></citation></ref>
<ref id="B67">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>F.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name> <name><surname>Zhao</surname> <given-names>Q.</given-names></name></person-group> (<year>2022</year>). <article-title>A deep learning method for predicting metabolite-disease associations via graph neural network</article-title>. <source>Brief. Bioinformatics</source> <volume>23</volume>, <fpage>bbac266</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbac266</pub-id><pub-id pub-id-type="pmid">35817399</pub-id></citation></ref>
<ref id="B68">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sung</surname> <given-names>H.</given-names></name> <name><surname>Ferlay</surname> <given-names>J.</given-names></name> <name><surname>Siegel</surname> <given-names>R. L.</given-names></name> <name><surname>Laversanne</surname> <given-names>M.</given-names></name> <name><surname>Soerjomataram</surname> <given-names>I.</given-names></name> <name><surname>Jemal</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Global cancer statistics 2020: globocan estimates of incidence and mortality worldwide for 36 cancers in 185 countries</article-title>. <source>CA Cancer J. Clin</source>. <volume>71</volume>, <fpage>209</fpage>&#x02013;<lpage>249</lpage>. <pub-id pub-id-type="doi">10.3322/caac.21660</pub-id><pub-id pub-id-type="pmid">33538338</pub-id></citation></ref>
<ref id="B69">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Takahashi</surname> <given-names>M. K.</given-names></name> <name><surname>Tan</surname> <given-names>X.</given-names></name> <name><surname>Dy</surname> <given-names>A. J.</given-names></name> <name><surname>Braff</surname> <given-names>D.</given-names></name> <name><surname>Akana</surname> <given-names>R. T.</given-names></name> <name><surname>Furuta</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>A low-cost paper-based synthetic biology platform for analyzing gut microbiota and host biomarkers</article-title>. <source>Nat. Commun</source>. <volume>9</volume>, <fpage>3347</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-018-05864-4</pub-id><pub-id pub-id-type="pmid">30131493</pub-id></citation></ref>
<ref id="B70">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tian</surname> <given-names>G.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Wang</surname> <given-names>C.</given-names></name> <name><surname>Chen</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>G.</given-names></name> <name><surname>Xu</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>A deep ensemble learning-based automated detection of covid-19 using lung CT images and vision transformer and convnext</article-title>. <source>Front. Microbiol</source>. <volume>13</volume>, <fpage>1024104</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2022.1024104</pub-id><pub-id pub-id-type="pmid">36406463</pub-id></citation></ref>
<ref id="B71">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Van Laarhoven</surname> <given-names>T.</given-names></name> <name><surname>Nabuurs</surname> <given-names>S. B.</given-names></name> <name><surname>Marchiori</surname> <given-names>E.</given-names></name></person-group> (<year>2011</year>). <article-title>Gaussian interaction profile kernels for predicting drug-target interaction</article-title>. <source>Bioinformatics</source> <volume>27</volume>, <fpage>3036</fpage>&#x02013;<lpage>3043</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btr500</pub-id><pub-id pub-id-type="pmid">21893517</pub-id></citation></ref>
<ref id="B72">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>F.</given-names></name> <name><surname>Huang</surname> <given-names>Z.-A.</given-names></name> <name><surname>Chen</surname> <given-names>X.</given-names></name> <name><surname>Zhu</surname> <given-names>Z.</given-names></name> <name><surname>Wen</surname> <given-names>Z.</given-names></name> <name><surname>Zhao</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>LRLSHMDA: Laplacian regularized least squares for human microbe-disease association prediction</article-title>. <source>Sci. Rep</source>. <volume>7</volume>, <fpage>7601</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-017-08127-2</pub-id><pub-id pub-id-type="pmid">28790448</pub-id></citation></ref>
<ref id="B73">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>T.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name> <name><surname>Zhao</surname> <given-names>Q.</given-names></name></person-group> (<year>2023</year>). <article-title>Investigating cardiotoxicity related with hERG channel blockers using molecular fingerprints and graph attention mechanism</article-title>. <source>Comput. Biol. Med</source>. <volume>153</volume>, <fpage>106464</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2022.106464</pub-id><pub-id pub-id-type="pmid">36584603</pub-id></citation></ref>
<ref id="B74">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>W.</given-names></name> <name><surname>Zhang</surname> <given-names>L.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name> <name><surname>Zhao</surname> <given-names>Q.</given-names></name> <name><surname>Shuai</surname> <given-names>J.</given-names></name></person-group> (<year>2022</year>). <article-title>Predicting the potential human lncrna-mirna interactions based on graph convolution network with conditional random field</article-title>. <source>Brief. Bioinformatics</source> <volume>23</volume>,<fpage> bbac463</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbac463</pub-id><pub-id pub-id-type="pmid">36305458</pub-id></citation></ref>
<ref id="B75">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Lei</surname> <given-names>X.</given-names></name> <name><surname>Pan</surname> <given-names>Y.</given-names></name></person-group> (<year>2023</year>). <article-title>Microbe-disease association prediction using rgcn through microbe-drug-disease network</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinformatics</source>. <volume>1</volume>, <fpage>1</fpage>&#x02013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2023.3247035</pub-id><pub-id pub-id-type="pmid">37027603</pub-id></citation></ref>
<ref id="B76">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wei</surname> <given-names>H.</given-names></name> <name><surname>Liu</surname> <given-names>B.</given-names></name></person-group> (<year>2020</year>). <article-title>ICIRCDA-MF: identification of circrna-disease associations based on matrix factorization</article-title>. <source>Brief. Bioinformatics</source> <volume>21</volume>, <fpage>1356</fpage>&#x02013;<lpage>1367</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbz057</pub-id><pub-id pub-id-type="pmid">31197324</pub-id></citation></ref>
<ref id="B77">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wong</surname> <given-names>A. H.</given-names></name> <name><surname>Ma</surname> <given-names>B.</given-names></name> <name><surname>Lui</surname> <given-names>R. N.</given-names></name></person-group> (<year>2023</year>). <article-title>New developments in targeted therapy for metastatic colorectal cancer</article-title>. <source>Ther. Adv. Med. Oncol</source>. <volume>15</volume>, <fpage>17588359221148540</fpage>. <pub-id pub-id-type="doi">10.1177/17588359221148540</pub-id><pub-id pub-id-type="pmid">36844139</pub-id></citation></ref>
<ref id="B78">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>C.</given-names></name> <name><surname>Gao</surname> <given-names>R.</given-names></name> <name><surname>Zhang</surname> <given-names>D.</given-names></name> <name><surname>Han</surname> <given-names>S.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name></person-group> (<year>2018</year>). <article-title>Prwhmda: human microbe-disease association prediction by random walk on the heterogeneous network with pso</article-title>. <source>Int. J. Biol. Sci</source>. <volume>14</volume>, <fpage>849</fpage>. <pub-id pub-id-type="doi">10.7150/ijbs.24539</pub-id><pub-id pub-id-type="pmid">29989079</pub-id></citation></ref>
<ref id="B79">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xia</surname> <given-names>C.</given-names></name> <name><surname>Dong</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Cao</surname> <given-names>M.</given-names></name> <name><surname>Sun</surname> <given-names>D.</given-names></name> <name><surname>He</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Cancer statistics in china and united states, 2022: profiles, trends, and determinants</article-title>. <source>Chin. Med. J</source>. <volume>135</volume>, <fpage>584</fpage>&#x02013;<lpage>590</lpage>. <pub-id pub-id-type="doi">10.1097/CM9.0000000000002108</pub-id><pub-id pub-id-type="pmid">35143424</pub-id></citation></ref>
<ref id="B80">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name></person-group> (<year>2006</year>). <article-title>Discovering disease-genes by topological features in human protein-protein interaction network</article-title>. <source>Bioinformatics</source> <volume>22</volume>, <fpage>2800</fpage>&#x02013;<lpage>2805</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btl467</pub-id><pub-id pub-id-type="pmid">16954137</pub-id></citation></ref>
<ref id="B81">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>J.</given-names></name> <name><surname>Xu</surname> <given-names>J.</given-names></name> <name><surname>Meng</surname> <given-names>Y.</given-names></name> <name><surname>Lu</surname> <given-names>C.</given-names></name> <name><surname>Cai</surname> <given-names>L.</given-names></name> <name><surname>Zeng</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Graph embedding and gaussian mixture variational autoencoder network for end-to-end analysis of single-cell RNA sequencing data</article-title>. <source>Cell Rep. Methods 3</source>, <fpage>100382</fpage>. <pub-id pub-id-type="doi">10.1016/j.crmeth.2022.100382</pub-id><pub-id pub-id-type="pmid">36814845</pub-id></citation></ref>
<ref id="B82">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yan</surname> <given-names>C.</given-names></name> <name><surname>Duan</surname> <given-names>G.</given-names></name> <name><surname>Wu</surname> <given-names>F.-X.</given-names></name> <name><surname>Pan</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name></person-group> (<year>2019</year>). <article-title>BRWMDA: predicting microbe-disease associations based on similarities and bi-random walk on disease and microbe networks</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinformatics</source> <volume>17</volume>, <fpage>1595</fpage>&#x02013;<lpage>1604</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2019.2907626</pub-id><pub-id pub-id-type="pmid">30932846</pub-id></citation></ref>
<ref id="B83">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>M.</given-names></name> <name><surname>Yang</surname> <given-names>H.</given-names></name> <name><surname>Ji</surname> <given-names>L.</given-names></name> <name><surname>Hu</surname> <given-names>X.</given-names></name> <name><surname>Tian</surname> <given-names>G.</given-names></name> <name><surname>Wang</surname> <given-names>B.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>A multi-omics machine learning framework in predicting the survival of colorectal cancer patients</article-title>. <source>Comput. Biol. Med</source>. <volume>146</volume>,<fpage> 105516</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2022.105516</pub-id><pub-id pub-id-type="pmid">35468406</pub-id></citation></ref>
<ref id="B84">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ye</surname> <given-names>J.</given-names></name> <name><surname>Chow</surname> <given-names>J.-H.</given-names></name> <name><surname>Chen</surname> <given-names>J.</given-names></name> <name><surname>Zheng</surname> <given-names>Z.</given-names></name></person-group> (<year>2009</year>). <article-title>&#x0201C;Stochastic gradient boosted distributed decision trees,&#x0201D; in <italic>Proceedings of the 18th ACM Conference on Information and Knowledge Management</italic></article-title> (Hong Kong), <fpage>2061</fpage>&#x02013;<lpage>2064</lpage>. <pub-id pub-id-type="doi">10.1145/1645953.1646301</pub-id></citation>
</ref>
<ref id="B85">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yeom</surname> <given-names>C. H.</given-names></name> <name><surname>Cho</surname> <given-names>M. M.</given-names></name> <name><surname>Baek</surname> <given-names>S. K.</given-names></name> <name><surname>Bae</surname> <given-names>O. S.</given-names></name> <etal/></person-group>. (<year>2010</year>). <article-title>Risk factors for the development of <italic>Clostridium difficile</italic> associated colitis after colorectal cancer surgery</article-title>. <source>J. Korean Soc. Coloproctol</source>. <volume>26</volume>, <fpage>329</fpage>&#x02013;<lpage>333</lpage>. <pub-id pub-id-type="doi">10.3393/jksc.2010.26.5.329</pub-id><pub-id pub-id-type="pmid">21152135</pub-id></citation></ref>
<ref id="B86">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>L.</given-names></name> <name><surname>Wang</surname> <given-names>C.-C.</given-names></name> <name><surname>Chen</surname> <given-names>X.</given-names></name></person-group> (<year>2022</year>). <article-title>Predicting drug-target binding affinity through molecule representation block based on multi-head attention and skip connection</article-title>. <source>Brief. Bioinformatics</source> <volume>23</volume>, <fpage>bbac468</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbac468</pub-id><pub-id pub-id-type="pmid">36411674</pub-id></citation></ref>
<ref id="B87">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>M.</given-names></name> <name><surname>Ma</surname> <given-names>W.</given-names></name> <name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>He</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name></person-group> (<year>2018</year>). <article-title>Analysis of gut microbiota profiles and microbe-disease associations in children with autism spectrum disorders in china</article-title>. <source>Sci. Rep</source>. <volume>8</volume>, <fpage>13981</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-018-32219-2</pub-id><pub-id pub-id-type="pmid">30228282</pub-id></citation></ref>
<ref id="B88">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>W.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>F.</given-names></name> <name><surname>Luo</surname> <given-names>F.</given-names></name> <name><surname>Tian</surname> <given-names>G.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>Predicting potential drug-drug interactions by integrating chemical, biological, phenotypic and network data</article-title>. <source>BMC Bioinformatics</source> <volume>18</volume>, <fpage>1</fpage>&#x02013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-016-1415-9</pub-id><pub-id pub-id-type="pmid">28056782</pub-id></citation></ref>
<ref id="B89">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Hoffmeister</surname> <given-names>M.</given-names></name> <name><surname>Weck</surname> <given-names>M. N.</given-names></name> <name><surname>Chang-Claude</surname> <given-names>J.</given-names></name> <name><surname>Brenner</surname> <given-names>H.</given-names></name></person-group> (<year>2012</year>). <article-title>Helicobacter pylori infection and colorectal cancer risk: evidence from a large population-based case-control study in germany</article-title>. <source>Am. J. Epidemiol</source>. <volume>175</volume>, <fpage>441</fpage>&#x02013;<lpage>450</lpage>. <pub-id pub-id-type="doi">10.1093/aje/kwr331</pub-id><pub-id pub-id-type="pmid">22908208</pub-id></citation></ref>
<ref id="B90">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Y.-J.</given-names></name> <name><surname>Li</surname> <given-names>S.</given-names></name> <name><surname>Gan</surname> <given-names>R.-Y.</given-names></name> <name><surname>Zhou</surname> <given-names>T.</given-names></name> <name><surname>Xu</surname> <given-names>D.-P.</given-names></name> <name><surname>Li</surname> <given-names>H.-B.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Impacts of gut bacteria on human health and diseases</article-title>. <source>Int. J. Mol. Sci</source>. <volume>16</volume>, <fpage>7493</fpage>&#x02013;<lpage>7519</lpage>. <pub-id pub-id-type="doi">10.3390/ijms16047493</pub-id><pub-id pub-id-type="pmid">25849657</pub-id></citation></ref>
<ref id="B91">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Xu</surname> <given-names>J.</given-names></name> <name><surname>Wu</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>N.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Liang</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>CAPSNET-LDA: predicting lncrna-disease associations using attention mechanism and capsule network based on multi-view data</article-title>. <source>Brief. Bioinformatics</source> <volume>24</volume>, <fpage>bbac531</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbac531</pub-id><pub-id pub-id-type="pmid">36511221</pub-id></citation></ref>
<ref id="B92">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>Y.</given-names></name> <name><surname>Xu</surname> <given-names>Z. Z.</given-names></name> <name><surname>He</surname> <given-names>Y.</given-names></name> <name><surname>Yang</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>L.</given-names></name> <name><surname>Lin</surname> <given-names>Q.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Gut microbiota offers universal biomarkers across ethnicity in inflammatory bowel disease diagnosis and infliximab response prediction</article-title>. <source>MSystems</source> <volume>3</volume>, <fpage>e00188</fpage>&#x02013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.1128/mSystems.00188-17</pub-id><pub-id pub-id-type="pmid">29404425</pub-id></citation></ref>
<ref id="B93">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>S.</given-names></name> <name><surname>Zhu</surname> <given-names>F.</given-names></name></person-group> (<year>2019</year>). <article-title>Cycling comfort evaluation with instrumented probe bicycle</article-title>. <source>Transp. Res. Part A. Policy Pract</source>. <volume>129</volume>, <fpage>217</fpage>&#x02013;<lpage>231</lpage>. <pub-id pub-id-type="doi">10.1016/j.tra.2019.08.009</pub-id></citation>
</ref>
<ref id="B94">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>T.</given-names></name> <name><surname>Dai</surname> <given-names>Q.</given-names></name> <name><surname>He</surname> <given-names>P.-A.</given-names></name></person-group> (<year>2021</year>). <article-title>Identification of potential immune-related biomarkers in gastrointestinal cancers</article-title>. <source>Curr. Bioinform</source>. <volume>16</volume>, <fpage>1203</fpage>&#x02013;<lpage>1213</lpage>. <pub-id pub-id-type="doi">10.2174/1574893615666210106121335</pub-id></citation>
</ref>
<ref id="B95">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zou</surname> <given-names>S.</given-names></name> <name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name></person-group> (<year>2017</year>). <article-title>A novel approach for predicting microbe-disease associations by bi-random walk on the heterogeneous network</article-title>. <source>PLoS ONE</source> <volume>12</volume>, <fpage>e0184394</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0184394</pub-id><pub-id pub-id-type="pmid">28880967</pub-id></citation></ref>
</ref-list> 
</back>
</article> 

