<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Microbiol.</journal-id>
<journal-title>Frontiers in Microbiology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Microbiol.</abbrev-journal-title>
<issn pub-type="epub">1664-302X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmicb.2024.1366272</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Microbiology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>LCASPMDA: a computational model for predicting potential microbe-drug associations based on learnable graph convolutional attention networks and self-paced iterative sampling ensemble</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Yang</surname> <given-names>Zinuo</given-names></name>
<uri xlink:href="https://loop.frontiersin.org/people/2656032/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Wang</surname> <given-names>Lei</given-names></name>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/664933/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhang</surname> <given-names>Xiangrui</given-names></name>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Zeng</surname> <given-names>Bin</given-names></name>
<uri xlink:href="https://loop.frontiersin.org/people/1999059/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Zhang</surname> <given-names>Zhen</given-names></name>
<xref ref-type="corresp" rid="c002"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2128527/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Liu</surname> <given-names>Xin</given-names></name>
<xref ref-type="corresp" rid="c003"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2128600/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff><institution>Big Data Innovation and Entrepreneurship Education Center of Hunan Province, Changsha University</institution>, <addr-line>Changsha</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0002">
<p>Edited by: George Tsiamis, University of Patras, Greece</p>
</fn>
<fn fn-type="edited-by" id="fn0003">
<p>Reviewed by: Jia Qu, Changzhou University, China</p>
<p>Fatemeh Zare-Mirakabad, Amirkabir University of Technology, Iran</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Lei Wang, <email>wanglei@xtu.edu.cn</email></corresp>
<corresp id="c002">Zhen Zhang, <email>155299243@qq.com</email></corresp>
<corresp id="c003">Xin Liu, <email>xin.liu@ccsu.edu.cn</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>23</day>
<month>05</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1366272</elocation-id>
<history>
<date date-type="received">
<day>06</day>
<month>01</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>06</day>
<month>05</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2024 Yang, Wang, Zhang, Zeng, Zhang and Liu.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Yang, Wang, Zhang, Zeng, Zhang and Liu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec id="sec300">
<title>Introduction</title>
<p>Numerous studies show that microbes in the human body are very closely linked to the human host and can affect the human host by modulating the efficacy and toxicity of drugs. However, discovering potential microbe-drug associations through traditional wet labs is expensive and time-consuming, hence, it is important and necessary to develop effective computational models to detect possible microbe-drug associations.</p>
</sec>
<sec id="sec301">
<title>Methods</title>
<p>In this manuscript, we proposed a new prediction model named LCASPMDA by combining the learnable graph convolutional attention network and the self-paced iterative sampling ensemble strategy to infer latent microbe-drug associations. In LCASPMDA, we first constructed a heterogeneous network based on newly downloaded known microbe-drug associations. Then, we adopted the learnable graph convolutional attention network to learn the hidden features of nodes in the heterogeneous network. After that, we utilized the self-paced iterative sampling ensemble strategy to select the most informative negative samples to train the Multi-Layer Perceptron classifier and put the newly-extracted hidden features into the trained MLP classifier to infer possible microbe-drug associations.</p>
</sec>
<sec id="sec302">
<title>Results and discussion</title>
<p>Intensive experimental results on two different public databases including the MDAD and the aBiofilm showed that LCASPMDA could achieve better performance than state-of-the-art baseline methods in microbe-drug association prediction.</p>
</sec>
</abstract>
<kwd-group>
<kwd>prediction model</kwd>
<kwd>drug-microbe association</kwd>
<kwd>learnable graph convolutional attention network</kwd>
<kwd>self-paced iterative sampling ensemble</kwd>
<kwd>multi-layer perceptron classifier</kwd>
</kwd-group>
<counts>
<fig-count count="9"/>
<table-count count="6"/>
<equation-count count="12"/>
<ref-count count="51"/>
<page-count count="15"/>
<word-count count="10280"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Systems Microbiology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<label>1</label>
<title>Introduction</title>
<p>The human body contains trillions of microbes, including bacteria, archaea, fungi, protozoa, and viruses, which constitute the human microbiota and interact closely with the human host (<xref ref-type="bibr" rid="ref39">The Human Microbiome Project Consortium, 2012</xref>; <xref ref-type="bibr" rid="ref35">Sommer and B&#x00E4;ckhed, 2013</xref>). These microbes can be found in the skin, oral cavity, nasal cavity, gastrointestinal tract, genitourinary tract and other parts of the human body, and play an important role in regulating human health. For example, they can regulate the pathology of the gastrointestinal tract and harmonize the homeostasis of the internal environment in order to promote the metabolic functions of the body (<xref ref-type="bibr" rid="ref12">Gill et al., 2006</xref>; <xref ref-type="bibr" rid="ref41">Ventura et al., 2009</xref>). The microbiome and host mucosal sites interact in a synergistic manner to protect against pathogens (<xref ref-type="bibr" rid="ref28">Macpherson and Harris, 2004</xref>). Microorganisms promote the synthesis of sugar metabolism and facilitate the synthesis of vitamins required for t-cell reactions (<xref ref-type="bibr" rid="ref17">Kau et al., 2011</xref>). But microorganisms also have adverse effects on the human body. For instance, studies have proved that dysbiosis of microbial communities can induce diabetes (<xref ref-type="bibr" rid="ref46">Wen et al., 2008</xref>), inflammatory bowel disease (<xref ref-type="bibr" rid="ref9">Durack and Lynch, 2019</xref>) and even cancer (<xref ref-type="bibr" rid="ref34">Schwabe and Jobin, 2013</xref>). And additionally, pathogens such as bacteria and viruses have been proven to be able to cause as many as 27 infectious diseases such as COVID-19 (<xref ref-type="bibr" rid="ref47">Xiang et al., 2020</xref>). Moreover, in recent years, due to the abuse and irrational use of drugs, microbes have developed resistance to some drugs, which has brought serious challenges to clinical medicine and drug development. In addition, recent studies have also shown that the efficacy of drugs is significantly influenced by the microbial metabolism (<xref ref-type="bibr" rid="ref29">McCoubrey et al., 2022</xref>). When drugs are functioning in the human body, microorganisms play an important role in drug absorption and metabolism, thereby modulating drug efficacy and toxicity (<xref ref-type="bibr" rid="ref51">Zimmermann et al., 2019</xref>). Concetta et al. reported that gut microbiota can interact with anticancer drugs, thus affecting the therapeutic efficiency and toxic side effects of drugs. They considered the probiotics, prebiotics, synbiotics, biologics and antibiotics as emerging strategies for microbiota control, which might improve treatment outcomes or ensure that patients have a better quality of life during anticancer treatment (<xref ref-type="bibr" rid="ref31">Panebianco et al., 2018</xref>). Therefore, the discovery of potential microbial-drug associations is one of the key problems to be solved in the field of precision medicine, and the need to develop an efficient computational model to discover potential microbial-drug associations is becoming more and more urgent.</p>
<p>Since traditional wet tests are very expensive, time-consuming and inefficient, moreover, in recent years, the advances in bioinformatics technology have given birth to lots of public microbial drug association databases, including MDAD (<xref ref-type="bibr" rid="ref36">Sun et al., 2018</xref>), aBiofilm (<xref ref-type="bibr" rid="ref33">Rajput et al., 2018</xref>), and DrugVirus (<xref ref-type="bibr" rid="ref2">Andersen et al., 2020</xref>), researchers have developed more and more feasible and efficient computational models based on these publicly available databases to infer potential microbe-drug associations (<xref ref-type="bibr" rid="ref21">Long et al., 2022</xref>), which can be roughly divided into five main categories: network-based, matrix decomposition, matrix complementation, regularization and neural networks. For example, <xref ref-type="bibr" rid="ref50">Zhu et al. (2019)</xref> designed a method called HMDAKATZ to detect latent associations between microbes and drugs by combining microbe-drug heterogeneous networks with the KATZ metrics. Long et al. proposed a prediction model named GCNMDA by adopting graph neural networks and conditional random fields with attentional mechanisms to learn deep representations of microbes and drugs (<xref ref-type="bibr" rid="ref22">Long et al. 2020a</xref>), and a calculation model called EGATMDA (<xref ref-type="bibr" rid="ref23">Long et al., 2020b</xref>) to predict potential associations between microorganisms and drugs by adopting a graph convolutional network with graph-level attention mechanism to learn the importance of different heterogeneous networks and a graph convolutional network with node-level attention to learn the embedding of nodes in the heterogeneous networks. <xref ref-type="bibr" rid="ref8">Deng et al. (2022)</xref> devised a method called Graph2MDA to detect possible associations between microbes and drugs, in which, multimodal attribute maps were constructed as inputs of the variogram self-encoder to obtain informative and interpretable latent features of microbes and drugs. <xref ref-type="bibr" rid="ref38">Tan et al. (2022)</xref> constructed a novel prediction model GSAMDA by integrating the graph attention network and the sparse self-encoder, in which, the graph attention network and the sparse self-encoder were adopted to extract topological features and node features of microbes and drugs in heterogeneous networks, respectively. <xref ref-type="bibr" rid="ref27">Ma et al. (2023)</xref> employed a two-layer graphical attention network to learn the features of microbes and drugs, and subsequently adopted a convolutional neural network classifier to detect potential microbe-drug associations. MHBVDA combined two new methods, such as the Matrix Decomposition for Heterogeneous Graph Inference (MDHGI) and the Bounded Nucleus Paradigm Regularization (BNNR), to construct virus-drug heterogeneous networks by using multi-source heterogeneous data of viruses and drugs, and then reconstructed the adjacency matrix of the network to predict the missing virus-drug associations (<xref ref-type="bibr" rid="ref32">Qu et al., 2023</xref>). NIRBMMDA first obtained two potential microbe-drug association matrices to calculate drug-microbe associations for similar drugs and microbe-drug associations for similar microbes by using different thresholds to find similar neighbors of drugs or microbes, respectively, and then obtained another potential microbe-drug association matrix based on the contrast scatter algorithm and the sigmoid function to learn the hidden probability distributions in the known microbe-drug associations (<xref ref-type="bibr" rid="ref5">Cheng et al., 2022</xref>).</p>
<p>Although above methods can achieve excellent prediction performance, there still exist some limitations. For instance, HMDAKATZ uses only simple metrics to evaluate the strength of microbe-drug associations, EGATMDA only randomly selects negative samples while ignores the specificity of different negative samples. Besides, recent studies have shown that the performance of Graph Convolutional Networks (GCN) and Graph Attention Networks (GAT) depend on the nature of selected datasets (<xref ref-type="bibr" rid="ref18">Knyazev et al., 2019</xref>; <xref ref-type="bibr" rid="ref4">Baranwal et al., 2021</xref>; <xref ref-type="bibr" rid="ref11">Fountoulakis et al., 2022</xref>), which means that the GCN-based GCNMDA cannot achieve satisfactory prediction on multiple different datasets at the same time, neither can the GAT-based GSAMDA. Therefore, in order to achieve better prediction performance, we need to choose between Graph Convolutional Networks (GCN) and Graph Attention Networks (GAT) through cross-validation. For this purpose, CAT (Graph Convolutional Attention Layer) is introduced to solve this problem. However, intensive experimental results have demonstrated that CAT can achieve better performance than both GAT and GCN at low noise levels in the dataset, but cannot improve the prediction performance significantly at higher noise levels, which means that there is no absolute difference between GCN, GAT and CAT, and their effectiveness is directly affected by the selected dataset. To solve this problem, Learnable Graph Convolutional Attention Networks (LCAT; <xref ref-type="bibr" rid="ref15">Javaloy et al., 2023</xref>) came into existence. Through efficiently combining the different GNN layers by adding two scalar parameters that are automatically interpolated in each layer of GCN, GAT and CAT, LCAT outperforms the methods of GCN, GAT and CAT in a wide range of datasets. Hence, it is obvious that, if we employ LCAT in the prediction model to infer possible microbial-drug associations, we can not only achieve better performance but also subtract the cross-validation requirement of choosing between the methods of GCN, GAT and CAT.</p>
<p>Moreover, in binary relationship prediction, how to select negative samples is important for model training, but selecting informative negative samples from the set of candidate negative samples is still an intractable problem (<xref ref-type="bibr" rid="ref19">Li et al., 2022</xref>). In link prediction problems, how to generate candidate negative samples has always been one of the challenges. Existing machine learning methods usually treat known associations between entities (labeled samples) as positive samples and unrecognized associations (unlabeled samples) as candidate negative samples (<xref ref-type="bibr" rid="ref48">Yang et al., 2012</xref>). However, since the number of known microbe-drug associations is very small in existing public datasets, the proportion of positive and negative samples will be extremely unbalanced in this case. Therefore, in order to avoid extreme imbalance in the proportion of positive and negative samples affecting the performance of the prediction model, we need further to perform a negative under-sampling strategy for candidate negative samples. But, as for the negative under-sampling strategies, the most common method is the random sampling, i.e., a subset of negative samples with the same number as the set of positive samples will be randomly selected from the candidate negative samples (<xref ref-type="bibr" rid="ref25">Lou et al., 2022</xref>). These random sampling-based strategies, while simple, tend to ignore informative negative samples and introduce less meaningful and noisy negative samples (<xref ref-type="bibr" rid="ref24">L&#x00F3;pez et al., 2013</xref>). Although there are some models that can improve the negative sampling strategy (<xref ref-type="bibr" rid="ref49">Zeng et al., 2020</xref>; <xref ref-type="bibr" rid="ref45">Wei et al., 2021</xref>; <xref ref-type="bibr" rid="ref7">Dai et al., 2022</xref>), but they do not focus on filtering out the most informative negative samples that play an important role for the classifier during the model training process, which may lead to under-training of the model, thus limiting the predictive power of the model.</p>
<p>Based on above analysis, in this study, we proposed a novel computational model LCASPMDA by integrating the Learnable Graph Convolutional Attention network and the Self-Paced iterative sampling ensemble strategy to identify potential Microbe-Drug Associations. In LCASPMDA, we will first construct a heterogeneous network of microbes and drugs based on these newly downloaded known microbe-drug associations and an integrated similarity of microbes and drugs. And then, we will employ LCAT to learn the hidden feature representations of nodes from heterogeneous networks. Subsequently, we will introduce the self-paced iterative sampling ensemble scheme to train the MLP classifier by selecting the most informative negative samples based on the prediction results after each training of the model, and finally input the feature representations extracted by the LCAT into the trained MLP (Multi-Layer Perceptron) classifier to infer potential associations between microbes and drugs. Intensive experimental results on two well-known public datasets showed that LCASPMDA significantly outperformed state-of-the-art competitive prediction methods in the prediction task of latent microbe-drug associations. And in addition, case studies of two common drugs further demonstrated the superiority of LCASPMDA in discovering new microbe-drug associations as well.</p>
</sec>
<sec sec-type="materials|methods" id="sec2">
<label>2</label>
<title>Materials and methods</title>
<p>As illustrated in <xref ref-type="fig" rid="fig1">Figure 1</xref>, LCASPMDA consists of four major steps. In the first step, we construct a heterogeneous network of microbes and drugs based on newly-downloaded known microbe-drug associations and an integrated similarity of microbes and drugs. In the second step, we adopt the LCAT to learn the feature representations of nodes in the heterogeneous network of microbes and drugs. In the third step, we introduce a Self-Paced Iterative Sampling Ensemble to select informative negative samples to train the MLP classifier. In the final step 4, we utilize the trained MLP to detect potential microbe-drug associations based on a novel loss function.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>The overall framework of LCASPMDA. Step1: a heterogeneous network of microbes and drugs is constructed based on newly-downloaded known microbe-drug associations and an integrated similarity of microbes and drugs. Step2: the heterogeneous network is inputted into the LCAT to learn the feature representations of nodes. Step 3: The Self-Paced Iterative Sampling Ensemble is adopted to select the most informative samples for training the MLP classifier while ensuring the balance of training samples. Step 4: potential associations between microbes and drugs are inferred by the trained MLP.</p>
</caption>
<graphic xlink:href="fmicb-15-1366272-g001.tif"/>
</fig>
<sec id="sec3">
<label>2.1</label>
<title>Datasets</title>
<p>In this section, we first downloaded known microbe-drug associations from the MDAD database, which were derived from 993 papers covering 1,388 drugs and 180 microorganisms (<xref ref-type="bibr" rid="ref43">Wang et al., 2022</xref>). After de-duplication, we finally obtained 2,470 known microbe-drug associations involving 173 microbes and 1,373 drugs. Subsequently, we further downloaded known microbe-drug associations for validation from the aBioflm database, which contained 2,884 known microbe-drug associations between 1,720 drugs and 140 microbes. Additionally, we downloaded the dataset of DrugVirus from the research (<xref ref-type="bibr" rid="ref22">Long et al., 2020a</xref>),<xref ref-type="fn" rid="fn0001"><sup>1</sup></xref> in which, there are 95 microbes and 175 drugs including 933 microbe&#x2013;drug associations between them. <xref ref-type="table" rid="tab1">Table 1</xref> illustrated the statistical information of these two kinds of newly downloaded datasets.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>The statistics of these two newly-downloaded databases.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Datasets</th>
<th align="center" valign="top">Microbes</th>
<th align="center" valign="top">Drugs</th>
<th align="center" valign="top">Associations</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">MDAD</td>
<td align="center" valign="top">173</td>
<td align="center" valign="top">1,373</td>
<td align="center" valign="top">2,470</td>
</tr>
<tr>
<td align="left" valign="top">aBiofilm</td>
<td align="center" valign="top">140</td>
<td align="center" valign="top">1,720</td>
<td align="center" valign="top">2,884</td>
</tr>
<tr>
<td align="left" valign="top">DrugVirus</td>
<td align="center" valign="top">95</td>
<td align="center" valign="top">175</td>
<td align="center" valign="top">933</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In each dataset, we used an adjacency matrix to represent the association relationship between microorganisms and drugs. Without loss of generality, the adjacency matrix can be represented as <inline-formula>
<mml:math id="M1">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mspace width="thickmathspace"/>
</mml:mrow>
</mml:msub>
<mml:mo>&#x00D7;</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math id="M2">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
</mml:mrow>
</mml:msub>
<mml:mspace width="thickmathspace"/>
</mml:mrow>
</mml:math>
</inline-formula>and <inline-formula>
<mml:math id="M3">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denote the number of microbes and drugs in the dataset, respectively. In the adjacency matrix <italic>A</italic>, for any given drug <inline-formula>
<mml:math id="M4">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and microbe <inline-formula>
<mml:math id="M5">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, if there is a known association between them, then the value of<inline-formula>
<mml:math id="M6">
<mml:mrow>
<mml:mspace width="thickmathspace"/>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> will be 1, otherwise the value of <inline-formula>
<mml:math id="M7">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> will be 0.</p>
</sec>
<sec id="sec4">
<label>2.2</label>
<title>Construction of the heterogeneous network of microbes and drugs</title>
<sec id="sec5">
<label>2.2.1</label>
<title>Calculation of the integrated similarity of microbe</title>
<p>In LCASPMDA, the similarity between microbes will be measured in two different ways. This first one is measured by the Gaussian interaction-profile-kernel similarity. Considering that drugs with similar therapeutic effects will be associated with similar microorganisms, let <italic>C</italic>(<italic>i</italic>) and <italic>C</italic>(&#x1D457;) denote the <italic>i</italic>th and <italic>j</italic>th column of the adjacency matrix <italic>A</italic> separately, then for any two given microorganisms <inline-formula>
<mml:math id="M8">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M9">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, the Gaussian interaction-profile-kernel similarity between them can be computed as follows:</p>
<disp-formula id="EQ1">
<label>(1)</label>
<mml:math id="M10">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>G</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x03BC;</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math id="M11">
<mml:mrow>
<mml:msub>
<mml:mi>&#x03BC;</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>is the normalized kernel bandwidth, which is calculated as:</p>
<disp-formula id="EQ2">
<label>(2)</label>
<mml:math id="M12">
<mml:mrow>
<mml:msub>
<mml:mi>&#x03BC;</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:msubsup>
<mml:mi>&#x03BC;</mml:mi>
<mml:mi>m</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Here <inline-formula>
<mml:math id="M13">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x03BC;</mml:mi>
<mml:mi>m</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>is the original bandwidth, which is usually set to 1. After determining the similarity of all microbial pairs according to above equations, then it is obvious that we can obtain a Microbe Gaussian Interaction-Profile-Kernel-based Similarity matrix <inline-formula>
<mml:math id="M14">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>G</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mspace width="thickmathspace"/>
</mml:mrow>
</mml:msub>
<mml:mo>&#x00D7;</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>In LCASPMDA, the second type of microbial similarity is measured by the microbial functional similarity in the following way: Firstly, we will construct a microbial protein&#x2013;protein functional association network and obtain genetic neighbor scores from the STRING database (<xref ref-type="bibr" rid="ref37">Szklarczyk et al., 2021</xref>). And then, for any two given microbes <inline-formula>
<mml:math id="M15">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M16">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, we will calculate the functional similarity <inline-formula>
<mml:math id="M17">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>(<inline-formula>
<mml:math id="M18">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) between them based on the method proposed by <xref ref-type="bibr" rid="ref16">Kamneva (2017)</xref>. Therefore, we can obtain a microbial functional similarity matrix <inline-formula>
<mml:math id="M19">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mspace width="thickmathspace"/>
</mml:mrow>
</mml:msub>
<mml:mo>&#x00D7;</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> as well.</p>
<p>Hence, for any two given microbes <inline-formula>
<mml:math id="M20">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M21">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, based on above two kinds of microbial similarities <inline-formula>
<mml:math id="M22">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>G</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>and <inline-formula>
<mml:math id="M23">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, it is easy to see that we can obtain an integrated microbial similarity <inline-formula>
<mml:math id="M24">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> according to the following <xref ref-type="disp-formula" rid="EQ3">equation (3)</xref>:</p>
<disp-formula id="EQ3">
<label>(3)</label>
<mml:math id="M25">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>G</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:mfrac>
<mml:mi mathvariant="normal"></mml:mi>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mspace width="thickmathspace"/>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2260;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mspace width="0.25em"/>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>G</mml:mi>
</mml:msub>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mi mathvariant="normal"></mml:mi>
<mml:mi mathvariant="normal"></mml:mi>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</sec>
<sec id="sec6">
<label>2.2.2</label>
<title>Calculation of the integrated similarity of drug</title>
<p>Let <italic>R</italic>(<italic>i</italic>) and <italic>R</italic>(&#x1D457;) denote the <italic>i</italic>th and <italic>j</italic>th rows of the adjacency matrix <italic>A</italic> separately, then in a manner similar to above <xref ref-type="disp-formula" rid="EQ1 EQ2">equations (1), (2)</xref>, for any two given drugs <inline-formula>
<mml:math id="M26">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M27">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, we can obviously obtain a drug Gaussian Interaction-Profile-Kernel-based Similarity matrix <inline-formula>
<mml:math id="M28">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>G</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mspace width="thickmathspace"/>
</mml:mrow>
</mml:msub>
<mml:mo>&#x00D7;</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> as well.</p>
<p>Besides, in LCASPMDA, the second type of drug similarity is measured by the drug structure similarity proposed by <xref ref-type="bibr" rid="ref14">Hattori et al. (2010)</xref>, and for convenience, for any two given drugs <inline-formula>
<mml:math id="M29">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M30">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, let the drug structure similarity between them be <inline-formula>
<mml:math id="M31">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, then it is obvious that we can obtain a drug structure similarity matrix <inline-formula>
<mml:math id="M32">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>&#x03F5;</mml:mi>
<mml:mspace width="thickmathspace"/>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mspace width="thickmathspace"/>
</mml:mrow>
</mml:msub>
<mml:mo>&#x00D7;</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> as well.</p>
<p>Hence, for any two given drugs <inline-formula>
<mml:math id="M33">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M34">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, based on above two kinds of drug similarities <inline-formula>
<mml:math id="M35">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>G</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M36">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, it is easy to see that we can obtain an integrated drug similarity <inline-formula>
<mml:math id="M37">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> according to the following <xref ref-type="disp-formula" rid="EQ4">equation (4)</xref>:</p>
<disp-formula id="EQ4">
<label>(4)</label>
<mml:math id="M38">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>G</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:mfrac>
<mml:mi mathvariant="normal"></mml:mi>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mspace width="thickmathspace"/>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2260;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mspace width="0.25em"/>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>G</mml:mi>
</mml:msub>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mi mathvariant="normal"></mml:mi>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</sec>
<sec id="sec7">
<label>2.2.3</label>
<title>Construction of the heterogeneous network</title>
<p>Through combining the adjacency matrix <italic>A</italic> with the integrated microbial similarity matrix <inline-formula>
<mml:math id="M39">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the integrated drug similarity matrix <inline-formula>
<mml:math id="M40">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, it is obvious that we can construct a heterogeneous network of microbes and drugs according to the following <xref ref-type="disp-formula" rid="EQ5">equation (5)</xref>:</p>
<disp-formula id="EQ5">
<label>(5)</label>
<mml:math id="M41">
<mml:mrow>
<mml:mi>Y</mml:mi>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mi>A</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msup>
<mml:mi>A</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mspace width="thickmathspace"/>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2217;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mspace width="thickmathspace"/>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
</sec>
</sec>
<sec id="sec8">
<label>2.3</label>
<title>Feature extraction for nodes in <italic>Y</italic> by LCAT</title>
<p>With the widespread use of GCN, GAT and CAT (Convolutional Attention Networks), researchers have gained some new insight into the limitations of these three kinds of Graph Neural Networks (GNN). For instance, <xref ref-type="bibr" rid="ref4">Baranwal et al. (2021)</xref> have demonstrated that GCN are significantly data separable when the graph data is neither sparse nor noisy. However, if the graph data is too noisy, the convolution essentially collapses the data to the same value and the GCN may fail. <xref ref-type="bibr" rid="ref11">Fountoulakis et al. (2022)</xref> have found that GAT exhibits strong differentiability even in noisy datasets. However, under this particular condition, <xref ref-type="bibr" rid="ref3">Anderson (2003)</xref> pointed out that simple classifiers can also show good results. Therefore, GCN are more beneficial in situations where the noise level is low, and GAT can perform better than GCN in other situations. That is, there is no way to conclude which network structure (GAT or GCN) is the optimal solution in all these two cases. In this research background, <xref ref-type="bibr" rid="ref15">Javaloy et al. (2023)</xref> have proposed the CAT, and experimentally demonstrated that CAT outperformed GAT with reasonable graph noise, however, it is also not always beneficial to perform convolution before computing attention, which is dependent on the datasets.</p>
<p>It is hard to know before the experiments of cross-validation which of GCN, GAT, or CAT works best. <xref ref-type="bibr" rid="ref15">Javaloy et al. (2023)</xref> believe that this problem can be solved by learning to interpolate between these three kinds of GNNs, and has proposed a new learnable graph convolutional attention network layer in the following way: for any given node <inline-formula>
<mml:math id="M42">
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mspace width="thickmathspace"/>
</mml:mrow>
</mml:math>
</inline-formula>in <inline-formula>
<mml:math id="M43">
<mml:mi>Y</mml:mi>
</mml:math>
</inline-formula>, let the feature of node <inline-formula>
<mml:math id="M44">
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
</mml:mrow>
</mml:math>
</inline-formula>be <inline-formula>
<mml:math id="M45">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> (i.e., the <italic>i</italic>th row of <italic>Y</italic>) and the set of neighboring nodes of <inline-formula>
<mml:math id="M46">
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mspace width="thickmathspace"/>
</mml:mrow>
</mml:math>
</inline-formula>in <inline-formula>
<mml:math id="M47">
<mml:mi>Y</mml:mi>
</mml:math>
</inline-formula> be <inline-formula>
<mml:math id="M48">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>&#x2009;then based on the following <xref ref-type="disp-formula" rid="EQ6">equations (6)</xref>, <xref ref-type="disp-formula" rid="EQ7">(7)</xref>, the learnable graph convolutional attention network layer can be represented as follows:</p>
<disp-formula id="EQ6"><label>(6)</label>
<mml:math id="M49">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>&#x03BB;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mi>L</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>k</mml:mi>
<mml:mi>y</mml:mi>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>u</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>a</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="true">&#x02DC;</mml:mo>
</mml:mover>
<mml:mspace width="thickmathspace"/>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="true">&#x02DC;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mrow>
<mml:mspace width="thickmathspace"/>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
</mml:mrow>
<mml:mi mathvariant="normal">where</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="true">&#x02DC;</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>&#x03BB;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:msub>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>&#x03BB;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:mspace width="0.25em"/>
</mml:mrow>
</mml:math></disp-formula>
<disp-formula id="EQ7">
<label>(7)</label>
<mml:math id="M51">
<mml:mrow>
<mml:msub>
<mml:mi>&#x03C1;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mi>s</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mspace width="thickmathspace"/>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mfrac>
<mml:mrow>
<mml:mi>exp</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mi>exp</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Here, <inline-formula>
<mml:math id="M52">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the attention score between nodes <inline-formula>
<mml:math id="M53">
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M54">
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M55">
<mml:mrow>
<mml:msub>
<mml:mi>&#x03C1;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the normalization of <inline-formula>
<mml:math id="M56">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M57">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mspace width="thickmathspace"/>
</mml:mrow>
</mml:math>
</inline-formula>is the learnable attention vector, and LeakyRelu is the commonly used activation function, <inline-formula>
<mml:math id="M58">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="true">&#x02DC;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M59">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="true">&#x02DC;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> denote the new node features after <inline-formula>
<mml:math id="M60">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M61">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> have been convolved, <inline-formula>
<mml:math id="M62">
<mml:mrow>
<mml:msub>
<mml:mi>&#x03BB;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M63">
<mml:mrow>
<mml:msub>
<mml:mi>&#x03BB;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are trainable values, and <inline-formula>
<mml:math id="M64">
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mspace width="thickmathspace"/>
</mml:mrow>
</mml:math>
</inline-formula>is a trainable weight matrix.</p>
<p>From observing <xref ref-type="fig" rid="fig2">Figure 2</xref>, we can understand the way that LACT works as a weighted average of the features obtained by GAT, GCN, and CAT, which enables that the weights of features obtained by GAT, GCN, and CAT can be dynamically adjusted to fit different data sets.</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>A new perspective on understanding the LCAT principle.</p>
</caption>
<graphic xlink:href="fmicb-15-1366272-g002.tif"/>
</fig>
<p>Attention mechanism is an indispensable and complex function of the human brain, as well as an important component of the LCAT. Through the attention mechanism, the human brain can consciously or unconsciously choose from a large number of input information to focus on a small number of useful information. This ensures that people can work in an organized way amid the information bombardment. GAT integrates the attention mechanism with graph neural networks, which has the ability to highlight important information and ignore irrelevant information. The core working principle of GAT is to compute the relationship between nodes by means of the attention mechanism. Among them, we need to clarify the attention vector: in graph neural networks, each node has a vector representing the features of that node. Attention vectors are computed on these feature vectors, which indicates how much attention each node pays to its neighboring nodes. Based on the calculated attention vector, the state of the node can be updated.</p>
<p>Then, after above operations, we can evidently obtain a new feature matrix <inline-formula>
<mml:math id="M65">
<mml:mi>E</mml:mi>
</mml:math>
</inline-formula>&#x2208;<inline-formula>
<mml:math id="M66">
<mml:mrow>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mspace width="thickmathspace"/>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mspace width="thickmathspace"/>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2217;</mml:mo>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, where each row in <italic>E</italic> represents the newly obtained deep features of nodes in the heterogeneous network <inline-formula>
<mml:math id="M67">
<mml:mi>Y</mml:mi>
</mml:math>
</inline-formula>, and <italic>F</italic> is the dimension of Embedding obtained by the LCAT.</p>
<p>Through analyses, it is easy to know that the above <xref ref-type="disp-formula" rid="EQ6">equation (6)</xref> can enable the LCAT to learn to interpolate between the GCN, the GAT and the CAT. For instance, when <inline-formula>
<mml:math id="M68">
<mml:mrow>
<mml:msub>
<mml:mi>&#x03BB;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is set to 0, node <inline-formula>
<mml:math id="M69">
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and its neighboring nodes will have the same<inline-formula>
<mml:math id="M70">
<mml:mrow>
<mml:mspace width="thickmathspace"/>
<mml:msub>
<mml:mi>&#x03C1;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and then the LCAT will turn to be a GCN. Additionally, when<inline-formula>
<mml:math id="M71">
<mml:mrow>
<mml:mspace width="thickmathspace"/>
<mml:msub>
<mml:mi>&#x03BB;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>=1 and <inline-formula>
<mml:math id="M72">
<mml:mrow>
<mml:msub>
<mml:mi>&#x03BB;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>=0, then the LCAT will be a GAT. Moreover, when <inline-formula>
<mml:math id="M73">
<mml:mrow>
<mml:msub>
<mml:mi>&#x03BB;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>and <inline-formula>
<mml:math id="M74">
<mml:mrow>
<mml:msub>
<mml:mi>&#x03BB;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are both set to 1, the LCAT will be a CAT. In this manuscript, as shown in <xref ref-type="fig" rid="fig3">Figure 3</xref>, we have also proved that LCAT is able to integrate the advantages of all these three kinds of GCNs and can achieve the best performance in different datasets.</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>Effects of different <italic>IR</italic> on LCASPMDA.</p>
</caption>
<graphic xlink:href="fmicb-15-1366272-g003.tif"/>
</fig>
</sec>
<sec id="sec9">
<label>2.4</label>
<title>Selection of negative samples for training the MLP classifier</title>
<sec id="sec10">
<label>2.4.1</label>
<title>The MLP classifier</title>
<p>MLP is a powerful tool for classification tasks, and its superiority has been proven in common binary classification tasks. In LCASPMDA, we will adopt the MLP classifier as the final decoder in the following way: firstly, the embedding of microorganisms and drugs obtained by LCAT will be taken as inputs of the MLP classifier. And then, the MLP classifier will implement the element-wise multiplication operation on these embeddings. Finally, the predicted score matrix of potential associations between microorganisms and drugs will be obtained after the processing of the activation function. For any given microorganism <inline-formula>
<mml:math id="M75">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and drug <inline-formula>
<mml:math id="M76">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, the final predicted score of potential association between them will be calculated according to the following <xref ref-type="disp-formula" rid="EQ8">equation (8)</xref>:</p>
<disp-formula id="EQ8">
<label>(8)</label>
<mml:math id="M77">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2299;</mml:mo>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math id="M78">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
<mml:mo>&#x00D7;</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the final predicted score matrix.<inline-formula>
<mml:math id="M79">
<mml:mrow>
<mml:mspace width="thickmathspace"/>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>and <inline-formula>
<mml:math id="M80">
<mml:mrow>
<mml:mspace width="thickmathspace"/>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>are embedding of <inline-formula>
<mml:math id="M81">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M82">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> obtained by LCAT respectively, <inline-formula>
<mml:math id="M83">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M84">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are trainable matrices, and the &#x2A00; operation is the element-wise multiplication. Let <italic>F</italic> be the dimension of Embedding obtained by the LCAT, then there is <inline-formula>
<mml:math id="M85">
<mml:mrow>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mi>F</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M86">
<mml:mrow>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mi>F</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M87">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M88">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. In addition, <italic>Rule</italic> and <italic>Sigmoid</italic> are activation functions adopted in the MLP classifier.</p>
</sec>
<sec id="sec11">
<label>2.4.2</label>
<title>Self-paced iterative sampling ensemble</title>
<p>Studies have demonstrated that the model performance decreases on datasets with imbalanced positive and negative samples (<xref ref-type="bibr" rid="ref20">Liu et al., 2020</xref>). The imbalance of positive and negative samples poses a considerable challenge to the training of classifiers. In simple terms, the unbalance of samples can cause serious deviations in the classification model, but it cannot be seen from some common metrics, for example, in the case where the number of positive samples is much larger than the number of negative samples, the trained model may already have a very high accuracy, but we can classify all the negative samples as false positive in such a case to also have a very high accuracy. In two well-known microbe-drug association databases such as the MDAD and the aBiofilm, all microbe-drug pairs with known associations are viewed as positive samples, while the remaining microbe-drug pairs are regarded as candidate negative samples. In all these two well-known databases, the number of candidate negative samples far exceeds the number of positive samples. In previous studies, many researchers have found this point, so they always choose the under-sampling method to balance the samples in order to ensure the balance of the dataset, and the commonly used method is the random under-sampling method. In this method, researchers will randomly draw the same number of negative samples as positive samples in the candidate negative sample set, thus ensuring that the ratio of positive to negative samples is 1:1. In this method, since the negative samples are selected randomly, the specificity of the negative samples is not fully considered, it may result in the loss of informative negative samples and introduction of meaningless samples at the same time. Selecting informative samples from the candidate negative samples is a challenging task, and in LCASPMDA, we will introduce the Self-Paced Iterative Sampling Ensemble strategy to pick out the informative negative samples (<xref ref-type="bibr" rid="ref20">Liu et al., 2020</xref>).</p>
<p>The Self-Paced Iterative Sampling Ensemble strategy proposed the concept of hardness function <italic>H</italic>, according to which the candidate negative samples were categorized into three categories such as the trivial samples, the noise samples and the edge samples, respectively. Among them, the noise samples have larger values of <italic>H</italic>, which means it may be a false negative sample. The trivial samples have smaller <italic>H</italic> values, which means it can be well classified by the prediction model. In LCASPMDA, we only need to keep a small portion of the trivial samples because they have been well learned. The remaining edge samples are the most informative samples in our training process, which symbolize the decision boundary of the prediction model. Obviously, expanding the proportion of edge samples on the dataset will benefit to improve the performance of the prediction model. In LCASPMDA, we will adopt the Self-Paced Iterative Sampling Ensemble strategy to pick out the informative negative samples according to the following steps:</p>
<p>Step 1: In LCASPMDA, the predicted values of all associations between microbes and drugs will be obtained by using the MLP classifier.</p>
<p>Step 2: the hardness function in the Self-Paced Iterative Sampling Ensemble strategy is defined by the following <xref ref-type="disp-formula" rid="EQ9">equation (9)</xref>:</p>
<disp-formula id="EQ9">
<label>(9)</label>
<mml:math id="M89">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:mspace width="0.25em"/>
<mml:mi>y</mml:mi>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:mspace width="0.25em"/>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Here, <inline-formula>
<mml:math id="M90">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the predicted score value obtained by the MLP classifier for the sample <italic>x</italic>, and <italic>y</italic> is the original label value of the sample <italic>x</italic>.</p>
<p>Step 3: All candidate negative samples are classified into <italic>k</italic> buckets based on the hardness function according to the following <xref ref-type="disp-formula" rid="EQ10">equation (10)</xref>:</p>
<disp-formula id="EQ10">
<label>(10)</label>
<mml:math id="M91">
<mml:mrow>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">x,</mml:mi>
<mml:mspace width="0.25em"/>
<mml:mi mathvariant="normal">y</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:mfrac>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:mspace width="0.25em"/>
<mml:mi>y</mml:mi>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:mspace width="0.25em"/>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2264;</mml:mo>
<mml:mfrac>
<mml:mi>l</mml:mi>
<mml:mi>k</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:mrow>
<mml:mspace width="thickmathspace"/>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">H</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:mspace width="0.25em"/>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Here, <italic>k</italic> is the number of buckets and is a hyperparameter. <inline-formula>
<mml:math id="M92">
<mml:mrow>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>represents the negative sample of the lth bucket.</p>
<p>Step 4: Adopting the Self-Paced Iterative Sampling Ensemble strategy to select different numbers of negative samples from <italic>k</italic> buckets to form the negative sample set for the next iteration of training. Let <inline-formula>
<mml:math id="M93">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the number of negative samples selected in the <inline-formula>
<mml:math id="M94">
<mml:mrow>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> bucket, then <inline-formula>
<mml:math id="M95">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> can be calculated according to the following <xref ref-type="disp-formula" rid="EQ11">equation (11)</xref>:</p>
<disp-formula id="EQ11">
<label>(11)</label>
<mml:math id="M96">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:munder>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x22C5;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>&#x03B2;</mml:mi>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mo>=</mml:mo>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>&#x03B2;</mml:mi>
<mml:mspace width="thickmathspace"/>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mspace width="0.25em"/>
<mml:mo>&#x22C5;</mml:mo>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mi>P</mml:mi>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mi mathvariant="normal"></mml:mi>
<mml:mi mathvariant="normal"></mml:mi>
<mml:mspace width="thickmathspace"/>
<mml:mi>t</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mspace width="0.25em"/>
<mml:mn>.....</mml:mn>
<mml:mo>,</mml:mo>
<mml:mspace width="0.25em"/>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Here, <inline-formula>
<mml:math id="M97">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>represents the average hardness value of the <italic>l</italic>th bucket, <italic>n</italic> is the number of epochs for which the model is ready to be trained and i is the current number of iterations, <inline-formula>
<mml:math id="M98">
<mml:mi>&#x03B2;</mml:mi>
</mml:math>
</inline-formula> is the self-paced factor, <inline-formula>
<mml:math id="M99">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the normalized sampling weight of the <italic>l</italic>th bucket, <inline-formula>
<mml:math id="M100">
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mi>P</mml:mi>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the number of positive samples, and <inline-formula>
<mml:math id="M101">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x00B7;</mml:mo>
<mml:mo>&#x00B7;</mml:mo>
<mml:mo>&#x22EF;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>Step 5: Randomly selecting <inline-formula>
<mml:math id="M102">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mspace width="thickmathspace"/>
</mml:mrow>
</mml:math>
</inline-formula>negative samples from the <italic>l</italic>th bucket, and gathering all the selected negative samples to form a new negative sample set. The set of negative samples selected by the Self-Paced Iterative Sampling Ensemble and all the positive samples are combined into a new training set to train the MLP classifier and proceed to the next iteration.</p>
<p>While implementing above strategy, we will update the hardness value at each iteration in order to generate the most informative samples. The self-paced factor <italic>&#x03B2;</italic> is the focus of the above strategy. The role of the self-paced factor <italic>&#x03B2;</italic> has been demonstrated experimentally in SPE (<xref ref-type="bibr" rid="ref20">Liu et al., 2020</xref>). Considering that as the training iterates, the number of trivial samples grows, then we need to reduce the weight value of the bucket with a large number of samples so that we can focus more on samples with higher hardness values. Therefore, in LCASPMDA, we will introduce a self-paced factor <italic>&#x03B2;</italic> growing from zero to infinity, and the growth of the self-paced factor <italic>&#x03B2;</italic> will be controlled by using a logarithmic function at the same time.</p>
</sec>
<sec id="sec12">
<label>2.4.3</label>
<title>Loss function</title>
<p>Since the association prediction problem belongs to the binary classification problem, in which the binary cross entropy has shown excellent performance, then in LCASPMDA, we will as well adopt the binary cross entropy function as the loss function of the MLP classifier, which is defined as follows:</p>
<disp-formula id="EQ12">
<label>(12)</label>
<mml:math id="M103">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>=</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:munder>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>z</mml:mi>
<mml:mo>+</mml:mo>
</mml:msup>
<mml:mo>&#x222A;</mml:mo>
<mml:msup>
<mml:mi>z</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msup>
<mml:mspace width="0.25em"/>
</mml:mrow>
</mml:munder>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>log</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mi>log</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mspace width="thickmathspace"/>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In LCASPMDA, we will consider each microbe-drug pair (<italic>i</italic>, <italic>j</italic>) as an independent microbe-drug sample. Besides, in above <xref ref-type="disp-formula" rid="EQ12">equation (12)</xref>, <inline-formula>
<mml:math id="M104">
<mml:mrow>
<mml:msup>
<mml:mi>z</mml:mi>
<mml:mo>+</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> denotes a subset of positive samples in the training sample and <inline-formula>
<mml:math id="M105">
<mml:mrow>
<mml:msup>
<mml:mi>z</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> represents a subset of negative samples, and for any given microbe-drug pair (<italic>i</italic>, <italic>j</italic>) belonging to <inline-formula>
<mml:math id="M106">
<mml:mrow>
<mml:msup>
<mml:mi>z</mml:mi>
<mml:mo>+</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, we will set its base truth value <inline-formula>
<mml:math id="M107">
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> to 1, while for any given microbe-drug pair (<italic>i</italic>, <italic>j</italic>) belonging to <inline-formula>
<mml:math id="M108">
<mml:mrow>
<mml:msup>
<mml:mi>z</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, we will set its base truth value <inline-formula>
<mml:math id="M109">
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="thickmathspace"/>
</mml:mrow>
</mml:math>
</inline-formula>to 0. Moreover, <italic>Score<sub>ij</sub></italic> represents the predicted score of the association between the <italic>i</italic>th microbe and the <italic>j</italic>th drug in the final score matrix obtained by the MLP classifier.</p>
<p>Finally, we will put the new features extracted by LCAT into the MLP classifier trained by the self-paced iterative sampling ensemble strategy to obtain the final output of our prediction model. Obviously, the MLP classifier will generate the predicted score of potential association between each pair of microbe and drug, which can help us discover the criticality of hidden microbe-drug associations.</p>
</sec>
</sec>
</sec>
<sec id="sec13">
<label>3</label>
<title>Experiments and results</title>
<p>In this section, we verified the prediction performance of LCASPMDA based on the framework of 5-fold cross-validation. In experiment, for any given newly-downloaded microbe-drug dataset, we will divide all microbe-drug pairs equally into five parts, and used one part at each time as the test set and the rest as the training set. Moreover, we will introduce the AUPR and the AUC values as the evaluation metrics to measure the performance of the model. In this section, to demonstrate the superiority of the LCASPMDA model, we will Comparison with baseline methods. Additionally, in order to improve the performance of LCASPMDA, we will first study the role of various parameters inside the model, and then, we will do ablation experiments to examine the contribution of the Self-Paced Iterative Sampling Ensemble strategy and the LCAT to the model. Finally, to prove the validity of our model, we will select some drugs from inside the database of MDAD to do case studies.</p>
<sec id="sec14">
<label>3.1</label>
<title>Comparison with baseline methods</title>
<p>In order to verify the prediction performance of LCASPMDA, in this section, we will compare it with the following five representative competing methods based on the databases of MDAD and aBiofilm respectively:</p><list list-type="bullet">
<list-item>
<p>GCNMDA (<xref ref-type="bibr" rid="ref22">Long et al., 2020a</xref>): in which, a graph convolutional network framework integrated with conditional random fields was proposed to infer potential associations between microbes and drugs.</p>
</list-item>
<list-item>
<p>GSAMDA (<xref ref-type="bibr" rid="ref38">Tan et al., 2022</xref>): which utilized graph attention networks and sparse auto-encoders to capture topological features and attribute features of nodes in a newly-constructed microbe-drug heterogeneous network first, and then, computed the likelihood of potential associations between microbe-drug pairs by leveraging these newly-captured features of microbes and drugs.</p>
</list-item>
<list-item>
<p>MDASAEA (<xref ref-type="bibr" rid="ref10">Fan et al., 2023</xref>): which predicted latent microbe-drug associations by combining the self-sparse encoders and the multi-head attention networks.</p>
</list-item>
<list-item>
<p>LRLSHMDA (<xref ref-type="bibr" rid="ref42">Wang et al., 2017</xref>): which employed the Laplace-regularized least squares classifier, a semi-supervised computational model, for predicting possible microbe-disease associations.</p>
</list-item>
<list-item>
<p>NTSHMDA (<xref ref-type="bibr" rid="ref26">Luo and Long, 2020</xref>): in which, an improved randomized wandering algorithm was used to infer potential microbe-disease associations by integrating topological similarities of nodes in a newly-constructed microbe-drug heterogeneous network.</p>
</list-item>
</list>
<p>The comparison results are shown in <xref ref-type="table" rid="tab2">Tables 2</xref>, <xref ref-type="table" rid="tab3">3</xref>. And in addition, we illustrate the optimal ROC curves and PR curves of these six competing methods, based on the databases of MDAD and aBiofilm respectively, in <xref ref-type="fig" rid="fig4">Figure 4</xref> to highlight the superiority of LCASPMDA. Finally, in order to better show the prediction performance of LCASPMDA, we further conducted intensive comparison experiments based on multiple metrics, in addition to the commonly used metrics such as the AUC and the AUPR, under the MDAD database and the aBiofilm database, respectively. And the comparison results were shown in <xref ref-type="table" rid="tab4">Table 4</xref>. Besides, we provided the experimental results of LCASPMDA based on the DrugVirus database in <xref ref-type="fig" rid="fig5">Figure 5</xref> as well.</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Performance comparison between baseline methods and LCASPMDA on MDAD under the framework of 5-fold cross-validation.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Methods</th>
<th align="center" valign="top">AUC</th>
<th align="center" valign="top">AUPR</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">GCNMDA</td>
<td align="center" valign="middle">0.9390<inline-formula>
<mml:math id="M110">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0014</td>
<td align="center" valign="middle">0.9346<inline-formula>
<mml:math id="M111">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0105</td>
</tr>
<tr>
<td align="left" valign="middle">GSAMDA</td>
<td align="center" valign="middle">0.9456<inline-formula>
<mml:math id="M112">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0007</td>
<td align="center" valign="middle">0.4513<inline-formula>
<mml:math id="M113">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0008</td>
</tr>
<tr>
<td align="left" valign="middle">MDASAEA</td>
<td align="center" valign="middle">0.9547<inline-formula>
<mml:math id="M114">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0015</td>
<td align="center" valign="middle">0.9406<inline-formula>
<mml:math id="M115">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0032</td>
</tr>
<tr>
<td align="left" valign="middle">LRLSHMDA</td>
<td align="center" valign="middle">0.9397<inline-formula>
<mml:math id="M116">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0031</td>
<td align="center" valign="middle">0.6347<inline-formula>
<mml:math id="M117">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0016</td>
</tr>
<tr>
<td align="left" valign="middle">NTSHMDA</td>
<td align="center" valign="middle">0.8683<inline-formula>
<mml:math id="M118">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0021</td>
<td align="center" valign="middle">0.1542<inline-formula>
<mml:math id="M119">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0115</td>
</tr>
<tr>
<td align="left" valign="middle">
<bold>LCASPMDA</bold>
</td>
<td align="center" valign="middle">
<bold>0.9703</bold>
<inline-formula>
<mml:math id="M120">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>
<bold>0.0109</bold>
</td>
<td align="center" valign="middle">
<bold>0.9674</bold>
<inline-formula>
<mml:math id="M121">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>
<bold>0.0117</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>The bold values are predicted scores achieved by LCASPMDA.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption>
<p>Performance comparison between baseline methods and LCASPMDA on aBiofilm under the framework of 5-fold cross-validation.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Methods</th>
<th align="center" valign="top">AUC</th>
<th align="center" valign="top">AUPR</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">GCNMDA</td>
<td align="center" valign="middle">0.9537<inline-formula>
<mml:math id="M122">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0022</td>
<td align="center" valign="middle">0.9376<inline-formula>
<mml:math id="M123">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0123</td>
</tr>
<tr>
<td align="left" valign="middle">GSAMDA</td>
<td align="center" valign="middle">0.9251<inline-formula>
<mml:math id="M124">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0101</td>
<td align="center" valign="middle">0.4649<inline-formula>
<mml:math id="M125">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0022</td>
</tr>
<tr>
<td align="left" valign="middle">MDASAEA</td>
<td align="center" valign="middle">0.9604<inline-formula>
<mml:math id="M126">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0007</td>
<td align="center" valign="middle">0.9534<inline-formula>
<mml:math id="M127">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0032</td>
</tr>
<tr>
<td align="left" valign="middle">LRLSHMDA</td>
<td align="center" valign="middle">0.9510<inline-formula>
<mml:math id="M128">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0007</td>
<td align="center" valign="middle">0.6747<inline-formula>
<mml:math id="M129">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0049</td>
</tr>
<tr>
<td align="left" valign="middle">NTSHMDA</td>
<td align="center" valign="middle">0.8811<inline-formula>
<mml:math id="M130">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0022</td>
<td align="center" valign="middle">0.1602<inline-formula>
<mml:math id="M131">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0204</td>
</tr>
<tr>
<td align="left" valign="middle">
<bold>LCASPMDA</bold>
</td>
<td align="center" valign="middle">
<bold>0.9742</bold>
<inline-formula>
<mml:math id="M132">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>
<bold>0.0121</bold>
</td>
<td align="center" valign="middle">
<bold>0.9720</bold>
<inline-formula>
<mml:math id="M133">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>
<bold>0.0120</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>The bold values are predicted scores achieved by LCASPMDA.</p>
</table-wrap-foot>
</table-wrap>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption>
<p>ROC and PR curves achieved by LCASPMDA and state-of-the-art methods based on MDAD and aBiofilm separately. <bold>(A)</bold> ROC curves based on MDAD, <bold>(B)</bold> PR curves based on MDAD, <bold>(C)</bold> ROC curves based on aBiofilm, <bold>(D)</bold> PR curves based on aBiofilm.</p>
</caption>
<graphic xlink:href="fmicb-15-1366272-g004.tif"/>
</fig>
<table-wrap position="float" id="tab4">
<label>Table 4</label>
<caption>
<p>Performance comparison between baseline methods and LCASPMDA base on the MDAD and the aBiofilm databases.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Datadabse</th>
<th align="left" valign="top">Performance Metrics</th>
<th align="center" valign="top">LCASPMDA</th>
<th align="center" valign="top">GCNMDA</th>
<th align="center" valign="top">GSAMDA</th>
<th align="center" valign="top">MDASAEA</th>
<th align="center" valign="top">NTSHMDA</th>
<th align="center" valign="top">LRLSHMDA</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">MDAD</td>
<td align="left" valign="top">ACC</td>
<td align="center" valign="top">0.9146<inline-formula>
<mml:math id="M134">
<mml:mrow>
<mml:mo>&#x2213;</mml:mo>
<mml:mn>0.0231</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula></td>
<td align="center" valign="top">0.7914<inline-formula>
<mml:math id="M135">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0149</td>
<td align="center" valign="top">0.5766<inline-formula>
<mml:math id="M136">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0156</td>
<td align="center" valign="top">0.9045<inline-formula>
<mml:math id="M137">
<mml:mrow>
<mml:mo>&#x2213;</mml:mo>
<mml:mn>0.0107</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula></td>
<td align="center" valign="top">0.9657<inline-formula>
<mml:math id="M138">
<mml:mrow>
<mml:mo>&#x2213;</mml:mo>
<mml:mn>0.0317</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula></td>
<td align="center" valign="top">0.9700<inline-formula>
<mml:math id="M139">
<mml:mrow>
<mml:mo>&#x2213;</mml:mo>
<mml:mn>0.0227</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula></td>
</tr>
<tr>
<td/>
<td align="left" valign="top">F1-SCORE</td>
<td align="center" valign="top">0.9144<inline-formula>
<mml:math id="M140">
<mml:mrow>
<mml:mo>&#x2213;</mml:mo>
<mml:mn>0.0159</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula></td>
<td align="center" valign="top">0.7500<inline-formula>
<mml:math id="M141">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0218</td>
<td align="center" valign="top">0.7707<inline-formula>
<mml:math id="M142">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0233</td>
<td align="center" valign="top">0.9270<inline-formula>
<mml:math id="M143">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0158</td>
<td align="center" valign="top">0.3205<inline-formula>
<mml:math id="M144">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0247</td>
<td align="center" valign="top">0.3009<inline-formula>
<mml:math id="M145">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0177</td>
</tr>
<tr>
<td/>
<td align="left" valign="top">MCC</td>
<td align="center" valign="top">0.8426<inline-formula>
<mml:math id="M146">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0478</td>
<td align="center" valign="top"><inline-formula>
<mml:math id="M147">
<mml:mrow>
<mml:mn>0.6101</mml:mn>
<mml:mo>&#x2213;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.0359</td>
<td align="center" valign="top">0.2881<inline-formula>
<mml:math id="M148">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0558</td>
<td align="center" valign="top">0.8415<inline-formula>
<mml:math id="M149">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0287</td>
<td align="center" valign="top">0.3918<inline-formula>
<mml:math id="M150">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0256</td>
<td align="center" valign="top">0.3344<inline-formula>
<mml:math id="M151">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0258</td>
</tr>
<tr>
<td align="left" valign="top">aBiofilm</td>
<td align="left" valign="top">ACC</td>
<td align="center" valign="top">0.8932<inline-formula>
<mml:math id="M152">
<mml:mrow>
<mml:mo>&#x2213;</mml:mo>
<mml:mn>0.0257</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula></td>
<td align="center" valign="top">0.8198<inline-formula>
<mml:math id="M153">
<mml:mrow>
<mml:mo>&#x2213;</mml:mo>
<mml:mn>0.0186</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula></td>
<td align="center" valign="top">0.5214<inline-formula>
<mml:math id="M154">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0378</td>
<td align="center" valign="top">0.8898<inline-formula>
<mml:math id="M155">
<mml:mrow>
<mml:mo>&#x2213;</mml:mo>
<mml:mn>0.0145</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula></td>
<td align="center" valign="top">0.9592<inline-formula>
<mml:math id="M156">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0309</td>
<td align="center" valign="top">0.9693<inline-formula>
<mml:math id="M157">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0108</td>
</tr>
<tr>
<td/>
<td align="left" valign="top">F1-SCORE</td>
<td align="center" valign="top">0.8954<inline-formula>
<mml:math id="M158">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0186</td>
<td align="center" valign="top">0.7634<inline-formula>
<mml:math id="M159">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0274</td>
<td align="center" valign="top">0.7554<inline-formula>
<mml:math id="M160">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0246</td>
<td align="center" valign="top">0.8845<inline-formula>
<mml:math id="M161">
<mml:mrow>
<mml:mo>&#x2213;</mml:mo>
<mml:mn>0.0203</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula></td>
<td align="center" valign="top">0.2262<inline-formula>
<mml:math id="M162">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0256</td>
<td align="center" valign="top">0.2367<inline-formula>
<mml:math id="M163">
<mml:mrow>
<mml:mo>&#x2213;</mml:mo>
<mml:mn>0.0214</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula></td>
</tr>
<tr>
<td/>
<td align="left" valign="top">MCC</td>
<td align="center" valign="top">0.7964<inline-formula>
<mml:math id="M164">
<mml:mrow>
<mml:mo>&#x2213;</mml:mo>
<mml:mn>0.0254</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula></td>
<td align="center" valign="top">0.6275<inline-formula>
<mml:math id="M165">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0478</td>
<td align="center" valign="top">0.2382<inline-formula>
<mml:math id="M166">
<mml:mrow>
<mml:mo>&#x2213;</mml:mo>
<mml:mn>0.0742</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula></td>
<td align="center" valign="top">0.7335<inline-formula>
<mml:math id="M167">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0288</td>
<td align="center" valign="top">0.3329<inline-formula>
<mml:math id="M168">
<mml:mrow>
<mml:mo>&#x2213;</mml:mo>
<mml:mn>0.0312</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula></td>
<td align="center" valign="top">0.3427<inline-formula>
<mml:math id="M169">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0212</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig position="float" id="fig5">
<label>Figure 5</label>
<caption>
<p>ROC and PR curves achieved by LCASPMDA based on the DrugVirus database. <bold>(A)</bold> ROC curves based on DrugVirus, <bold>(B)</bold> PR curves based on DrugVirus.</p>
</caption>
<graphic xlink:href="fmicb-15-1366272-g005.tif"/>
</fig>
<p>From above two tables, it can be seen that LCASPMDA can achieve the best prediction performance among these six competing methods. And the AUC and AUPR values of LCASPMDA on MDAD are 0.9703<inline-formula>
<mml:math id="M170">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0109 and 0.9674<inline-formula>
<mml:math id="M171">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0117, respectively.</p>
<p>From observing <xref ref-type="fig" rid="fig4">Figure 4</xref>, we can clearly see the superiority of LCASPMDA. Among these six competing methods, MDASAEA is only method that can achieve better performance than LCASPMDA in terms of ROC curve, but in terms of PR curve, the PR values of the GSAMDA on MDAD and aBiofilm are only 0.4513<inline-formula>
<mml:math id="M172">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0008 and 0.4649<inline-formula>
<mml:math id="M173">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0022, respectively, which are quite lower than that of LCASPMDA. Through analysis, the reason that why GSAMDA can only achieve such a lower PR value is that the extreme imbalance in the proportion of positive and negative samples causes the model to be biased toward making predictions with the majority class, i.e., the negative examples, resulting in lower precision and recall for positive examples. However, LCASPMDA selects information-rich negative samples by the Self-Paced Iterative Sampling Ensemble method while balancing the ratio of positive and negative samples, which ensures that satisfactory PR values can be obtained. Besides, the PR curve achieved by NTSHMDA is the strangest one, first of all, it seems to be relatively coarse, after analysis, we find the reason is that the model floats a lot in a time period and the time gap is very short. Secondly, the AUPR values achieved by NTSHMDA are only 0.1542<inline-formula>
<mml:math id="M174">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0115 and 0.1602<inline-formula>
<mml:math id="M175">
<mml:mo>&#x2213;</mml:mo>
</mml:math>
</inline-formula>0.0204 on MDAD and aBiofilm separately, after analysis, we find the reason is that it does not have a regularization operation to prevent the model from over-fitting, and at the same time, it does not use a deep learning algorithm with a loss function to point out a correct direction of learning. However, LCASPMDA uses the dropout method to prevent the model from over-fitting during model training, and at the same time, chooses the cross-entropy function, which is most suitable for binary classification tasks, as the loss function. Meanwhile, we also find that NTSHMDA achieved the lowest AUC and AUPR values compared to other models using deep learning algorithms. From observing <xref ref-type="table" rid="tab4">Table 4</xref>, it is easy to see that LCASPMDA significantly outperformed all these baseline models as a whole. Through analysis, we found that the main reason is due to the extreme imbalance in the proportion of positive and negative samples of other models (Except for GCNMDA and MDASAEA) as well as the excellent model architecture adopted by LCASPMDA. Moreover, it is well known that the metrics of AUC and ACC are insensitive to the proportions of positive and negative samples, however the rest of the metrics are very sensitive to their proportions, and in addition, MCC is a more comprehensive performance metric that only scores high when good results are obtained for these four metrics (True Positive, True Negative, False Positive, False Negative), that is the reason why NTSHMDA and LRLSHMDA performed well in the ACC metric but poorly in the rest of the metrics. Besides, LCASPMDA achieved better performance than GCNMDA, the reason is that although these two models can maintain the balance of positive and negative sample ratios, but LCASPMDA utilizes an innovative model LCAT to extract node features, which is more effective than the GCN adopted by GCNMDA.</p>
</sec>
<sec id="sec15">
<label>3.2</label>
<title>Parameter analysis</title>
<p>In this section, we will evaluate the effect of two important parameters, including the parameters <italic>IR</italic> and <italic>Out-dimension</italic> that denote the learning rate of our model and the number of embedding dimension of the LCAT separately, on LCASPMDA, based on the MDAD database. Through observing <xref ref-type="fig" rid="fig3">Figure 3</xref>, we can clearly know that LCASPMDA performs best when the learning rate <italic>IR</italic> is set to 0.0005. When we explore the impact of the parameter <italic>Out-dimension</italic>, we set the <italic>Out-dimension</italic> to {64, 128, 256, 512} respectively, and show the experimental results in <xref ref-type="fig" rid="fig6">Figure 6</xref>. It is obvious that the AUC value of LCASPMDA peaks when <italic>Out-dimension</italic> is set to 256, and the AUC value tends to decrease when it exceeds 256. Based on above analysis, we will finally set the parameter <italic>IR</italic> to 0.0005 and the parameter <italic>Out-dimension</italic> to 256 in experiments.</p>
<fig position="float" id="fig6">
<label>Figure 6</label>
<caption>
<p>Effects of different <italic>Out-dimension</italic> on LCASPMDA.</p>
</caption>
<graphic xlink:href="fmicb-15-1366272-g006.tif"/>
</fig>
</sec>
<sec id="sec16">
<label>3.3</label>
<title>Ablation study</title>
<p>The Self-Paced Iterative Sampling Ensemble strategy (hereinafter referred to as SPISE) is the core part of LCASPMDA, which focuses on how to obtain a balanced dataset in an unbalanced set of positive and negative samples through a special negative sampling method while ensuring that the negative samples have a large amount of information. In order to evaluate the impact of SPISE on the performance of LCASPMDA, we first conducted an ablation study in this section. And then, considering that SPISE is a pivotal component of the LCASPMDA framework, to fully ascertain the effectiveness of SPISE, we further conducted an additional evaluation by varying the proportions of negative samples selected by SPISE, and in the experiment, part of the negative samples were selected by SPIE, and the rest were selected by random. In addition, we also conducted experiments by replacing the LCAT in LCASPMDA with the GAT and the GCN, respectively. As shown in <xref ref-type="fig" rid="fig7">Figures 7</xref>&#x2013;<xref ref-type="fig" rid="fig9">9</xref>, it is easy to see that adopting the SPISE can improve the prediction performance of LCASPMDA observably, and simultaneously, adopting the LCAT can achieve better performance than adopting the GAT and the GCN in LCASPMDA as well.</p>
<fig position="float" id="fig7">
<label>Figure 7</label>
<caption>
<p>SPISE can improve the prediction performance of LCASPMDA. <bold>(A)</bold> ROC curves based on MDAD, <bold>(B)</bold> PR curves based on MDAD, <bold>(C)</bold> ROC curves based on aBiofilm, <bold>(D)</bold> PR curves based on aBiofilm.</p>
</caption>
<graphic xlink:href="fmicb-15-1366272-g007.tif"/>
</fig>
<fig position="float" id="fig8">
<label>Figure 8</label>
<caption>
<p>SPIE has an important effect on the overall model performance of LCASPMDA. <bold>(A)</bold> ROC curves, <bold>(B)</bold> PR curves.</p>
</caption>
<graphic xlink:href="fmicb-15-1366272-g008.tif"/>
</fig>
<fig position="float" id="fig9">
<label>Figure 9</label>
<caption>
<p>LCAT can improve the prediction performance of LCASPMDA. <bold>(A)</bold> ROC curves based on MDAD, <bold>(B)</bold> PR curves based on MDAD, <bold>(C)</bold> ROC curves based on aBiofilm, <bold>(D)</bold> PR curves based on aBiofilm.</p>
</caption>
<graphic xlink:href="fmicb-15-1366272-g009.tif"/>
</fig>
</sec>
<sec id="sec17">
<label>3.4</label>
<title>Case study</title>
<p>To further validate the ability of LCASPMDA in predicting unknown associations between microorganisms and drugs, we conducted case studies, respectively, based on two drugs, including the Ciprofloxacin and the Pefloxacin, which are commonly used in MDAD. We trained the models based on the MDAD database. Specifically, for each selected target drug, all known microbe-drug associations were set to be unknown, and then all candidate microbes were ranked in the descending order based on predicted scores obtained by LCASPMDA. In experiment, for any given drug, we would choose the top 20 related microorganisms predicted by LCASPMDA, and verified that whether these predicted microbes had already been reported to be associated with the given drug in the PubMed literatures.</p>
<p>As for the Ciprofloxacin, it is a fluoroquinolone-containing drug with a high potential for antibacterial activity, and commonly used in the treatment of joint infections, respiratory infections and other treatments. Ciprofloxacin has broad-spectrum antimicrobial activity, with strong bactericidal effect against <italic>Pseudomonas aeruginosa</italic>, <italic>Staphylococcus aureus</italic> and other common pathogenic bacteria. Numerous experiments have also confirmed the close relationship between Ciprofloxacin and human microorganisms. For example, <xref ref-type="bibr" rid="ref13">Hacioglu et al. (2019)</xref> discovered and validated Ciprofloxacin as an active drug against <italic>Candida albicans</italic>. Besides, <xref ref-type="bibr" rid="ref40">Trinh et al. (2017)</xref> demonstrated that the combination of ceftriaxone and Ciprofloxacin was the most effective treatment for foodborne Vibrio traumaticus. Moreover, <xref ref-type="bibr" rid="ref6">Cho et al. (2019)</xref> found that <italic>Mycobacterium avium</italic> was highly sensitive to Ciprofloxacin. <xref ref-type="table" rid="tab5">Table 5</xref> shows the predicted results of the top 20 microorganisms associated with Ciprofloxacin, 17 of which have been demonstrated in the PubMed literatures.</p>
<table-wrap position="float" id="tab5">
<label>Table 5</label>
<caption>
<p>The top 20 predict Ciprofloxacin-associated microbes.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Microbe</th>
<th align="center" valign="top">Rank</th>
<th align="left" valign="top">Evidence</th>
<th align="left" valign="top">Microbe</th>
<th align="center" valign="top">Rank</th>
<th align="left" valign="top">Evidence</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle"><italic>Escherichia coli</italic></td>
<td align="center" valign="middle">1</td>
<td align="left" valign="middle">PMID: 26607324</td>
<td align="left" valign="middle"><italic>Stenotrophomonas maltophilia</italic></td>
<td align="center" valign="middle">11</td>
<td align="left" valign="middle">PMID: 28488744</td>
</tr>
<tr>
<td align="left" valign="middle"><italic>Pseudomonas aeruginosa</italic></td>
<td align="center" valign="middle">2</td>
<td align="left" valign="middle">PMID: 30605076</td>
<td align="left" valign="middle"><italic>Bacillus subtilis</italic></td>
<td align="center" valign="middle">12</td>
<td align="left" valign="middle">PMID: 15194135</td>
</tr>
<tr>
<td align="left" valign="middle"><italic>Staphylococcus aureus</italic></td>
<td align="center" valign="middle">3</td>
<td align="left" valign="middle">PMID: 32488138</td>
<td align="left" valign="middle"><italic>Candida</italic> spp.</td>
<td align="center" valign="middle">13</td>
<td align="left" valign="middle">Unconfirmed</td>
</tr>
<tr>
<td align="left" valign="middle"><italic>Burkholderia cepacia</italic></td>
<td align="center" valign="middle">4</td>
<td align="left" valign="middle">PMID: 10091030</td>
<td align="left" valign="middle"><italic>Aeromonas hydrophila</italic></td>
<td align="center" valign="middle">14</td>
<td align="left" valign="middle">PMID: 24242249</td>
</tr>
<tr>
<td align="left" valign="middle"><italic>Klebsiella planticola</italic></td>
<td align="center" valign="middle">5</td>
<td align="left" valign="middle">PMID: 25465871</td>
<td align="left" valign="middle">
<italic>Burkholderia multivorans</italic>
</td>
<td align="center" valign="middle">15</td>
<td align="left" valign="middle">PMID:19633000</td>
</tr>
<tr>
<td align="left" valign="middle">
<italic>Burkholderia cenocepacia</italic>
</td>
<td align="center" valign="middle">6</td>
<td align="left" valign="middle">PMID:27799222</td>
<td align="left" valign="middle">
<italic>Streptococcus pneumoniae</italic>
</td>
<td align="center" valign="middle">16</td>
<td align="left" valign="middle">PMID: 15155208</td>
</tr>
<tr>
<td align="left" valign="middle">
<italic>Klebsiella pneumoniae</italic>
</td>
<td align="center" valign="middle">7</td>
<td align="left" valign="middle">PMID:27257956</td>
<td align="left" valign="middle">
<italic>Micrococcus luteus</italic>
</td>
<td align="center" valign="middle">17</td>
<td align="left" valign="middle">PMID:16340189</td>
</tr>
<tr>
<td align="left" valign="middle">
<italic>Listeria monocytogenes</italic>
</td>
<td align="center" valign="middle">8</td>
<td align="left" valign="middle">PMID:28355096</td>
<td align="left" valign="middle">Enteric bacteria and other eubacteria</td>
<td align="center" valign="middle">18</td>
<td align="left" valign="middle">PMID: 27436461</td>
</tr>
<tr>
<td align="left" valign="middle">
<italic>Vibrio harveyi</italic>
</td>
<td align="center" valign="middle">9</td>
<td align="left" valign="middle">PMID:27247095</td>
<td align="left" valign="middle">
<italic>Streptococcus epidermidis</italic>
</td>
<td align="center" valign="middle">19</td>
<td align="left" valign="middle">Unconfirmed</td>
</tr>
<tr>
<td align="left" valign="middle">
<italic>Acinetobacter baumannii</italic>
</td>
<td align="center" valign="middle">10</td>
<td align="left" valign="middle">PMID:25147676</td>
<td align="left" valign="middle">
<italic>Kocuria rhizophila</italic>
</td>
<td align="center" valign="middle">20</td>
<td align="left" valign="middle">Unconfirmed</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Pefloxacin is a broad-spectrum quinolone antibiotic with significant bactericidal effects against a wide range of Gram-negative and Gram-positive bacteria. For instance, <xref ref-type="bibr" rid="ref44">Wang et al. (2021)</xref> studied the resistance of <italic>Escherichia coli</italic> strains to Pefloxacin and found that the resistance rate was as high as 68.8%, which provided new clues for the study of genetic and epidemiological characterization of urinary tract infections after renal transplantation. <xref ref-type="bibr" rid="ref30">Moin et al. (2020)</xref> detected the susceptibility of <italic>Salmonella enterica</italic> thermophila to fluoroquinolones with the Pefloxacin disk diffusion test. <xref ref-type="bibr" rid="ref1">Abdoulaye et al. (2018)</xref> isolated 126 bacterial strains from patients with surgical site infections (ISO) at the National Hospital of Niamey and found them to be resistant to different fluoroquinolones (e.g., Pefloxacin and nalidixic acid) to varying degrees. <xref ref-type="table" rid="tab6">Table 6</xref> illustrates the predicted results for the top 20 Pefloxacin-associated microorganisms, 14 of which have been proved in the PubMed literatures.</p>
<table-wrap position="float" id="tab6">
<label>Table 6</label>
<caption>
<p>The top 20 predict Pefloxacin-associated microbes.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Microbe</th>
<th align="center" valign="top">Rank</th>
<th align="left" valign="top">Evidence</th>
<th align="left" valign="top">Microbe</th>
<th align="center" valign="top">Rank</th>
<th align="left" valign="top">Evidence</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">
<italic>Staphylococcus aureus</italic>
</td>
<td align="center" valign="middle">1</td>
<td align="left" valign="middle">PMID: 2258345</td>
<td align="left" valign="middle">Human Immunodeficiency Virus 1</td>
<td align="center" valign="middle">11</td>
<td align="left" valign="middle">PMID:9495677</td>
</tr>
<tr>
<td align="left" valign="middle">
<italic>Staphylococcus epidermis</italic>
</td>
<td align="center" valign="middle">2</td>
<td align="left" valign="middle">PMID: 2640275</td>
<td align="left" valign="middle">
<italic>Clostridium perfringens</italic>
</td>
<td align="center" valign="middle">12</td>
<td align="left" valign="middle">Unconfirmed</td>
</tr>
<tr>
<td align="left" valign="middle">
<italic>Vibrio harveyi</italic>
</td>
<td align="center" valign="middle">3</td>
<td align="left" valign="middle">Unconfirmed</td>
<td align="left" valign="middle">Enteric bacteria and other eubacteria</td>
<td align="center" valign="middle">13</td>
<td align="left" valign="middle">Unconfirmed</td>
</tr>
<tr>
<td align="left" valign="middle">
<italic>Bacillus subtilis</italic>
</td>
<td align="center" valign="middle">4</td>
<td align="left" valign="middle">PMID: 12024980</td>
<td align="left" valign="middle">
<italic>Pseudomonas aeruginosa</italic>
</td>
<td align="center" valign="middle">14</td>
<td align="left" valign="middle">PMID: 1645509</td>
</tr>
<tr>
<td align="left" valign="middle">
<italic>Staphylococcus epidermidis</italic>
</td>
<td align="center" valign="middle">5</td>
<td align="left" valign="middle">PMID: 26607324</td>
<td align="left" valign="middle">
<italic>Burkholderia cenocepacia</italic>
</td>
<td align="center" valign="middle">15</td>
<td align="left" valign="middle">Unconfirmed</td>
</tr>
<tr>
<td align="left" valign="middle">
<italic>Burkholderia thailandensis</italic>
</td>
<td align="center" valign="middle">6</td>
<td align="left" valign="middle">Unconfirmed</td>
<td align="left" valign="middle">
<italic>Mycobacterium smegmatis</italic>
</td>
<td align="center" valign="middle">16</td>
<td align="left" valign="middle">PMID: 25379514</td>
</tr>
<tr>
<td align="left" valign="middle">
<italic>Listeria monocytogenes</italic>
</td>
<td align="center" valign="middle">7</td>
<td align="left" valign="middle">PMID: 2504545</td>
<td align="left" valign="middle">Human immunodeficiency virus</td>
<td align="center" valign="middle">17</td>
<td align="left" valign="middle">PMID: 9495677</td>
</tr>
<tr>
<td align="left" valign="middle">
<italic>Enterococcus faecalis</italic>
</td>
<td align="center" valign="middle">8</td>
<td align="left" valign="middle">PMID: 2258345</td>
<td align="left" valign="middle">
<italic>Micrococcus luteus</italic>
</td>
<td align="center" valign="middle">18</td>
<td align="left" valign="middle">PMID:16340189</td>
</tr>
<tr>
<td align="left" valign="middle">
<italic>Burkholderia pseudomallei</italic>
</td>
<td align="center" valign="middle">9</td>
<td align="left" valign="middle">PMID: 19121669</td>
<td align="left" valign="middle">
<italic>Actinoplanes missouriensis</italic>
</td>
<td align="center" valign="middle">19</td>
<td align="left" valign="middle">Unconfirmed</td>
</tr>
<tr>
<td align="left" valign="middle">
<italic>Streptococcus pneumoniae</italic>
</td>
<td align="center" valign="middle">10</td>
<td align="left" valign="middle">PMID: 20384283</td>
<td align="left" valign="middle">
<italic>Salmonella enterica</italic>
</td>
<td align="center" valign="middle">20</td>
<td align="left" valign="middle">PMID: 28948961</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec sec-type="conclusions" id="sec18">
<label>4</label>
<title>Conclusion</title>
<p>We are committed to discovering more potential microbe-drug associations and making our contribution to the protection of human health. In this paper we have proposed a novel prediction model called LCASPMDA by combining a learnable graph convolution attention network, a Self-Paced Iterative Sampling Ensemble strategy and a multi-layer perceptron. Comparative experiments and case studies show that LCASPMDA can achieve excellent prediction performance. And at the same time, there are still areas where it can be improved, for example, LCASPMDA does not collect or use any actual negative samples. Secondly, using MLP to generate new microbe-drug association matrices may provide useless association information. Thirdly, the parameters used in MLP and LCAT may not be optimal and may even be biased, and the lack of negative samples may significantly affect the predictive performance of LCASPMDA. Therefore, on the one hand, it is crucial to obtain negative samples from biomedical databases and literature. On the other hand, developing computational methods to generate high-quality negative samples is another option to address this issue. In addition, it is noted that selected negative samples can achieve significant performance improvements in the area of protein-RNA interaction identification as well. Meanwhile, we can introduce some new mechanisms such as the attention mechanism, spatial convolution mechanism and so on to improve the performance of the model.</p>
</sec>
<sec sec-type="data-availability" id="sec19">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec sec-type="author-contributions" id="sec20">
<title>Author contributions</title>
<p>ZY: Data curation, Methodology, Resources, Software, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. LW: Conceptualization, Funding acquisition, Investigation, Methodology, Project administration, Supervision, Writing &#x2013; original draft. XZ: Data curation, Formal analysis, Validation, Writing &#x2013; original draft. BZ: Methodology, Supervision, Validation, Writing &#x2013; review &#x0026; editing. ZZ: Investigation, Methodology, Supervision, Visualization, Writing &#x2013; review &#x0026; editing. XL: Methodology, Software, Supervision, Validation, Writing &#x2013; review &#x0026; editing.</p>
</sec>
</body>
<back>
<sec sec-type="funding-information" id="sec21">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This work was partly sponsored by the National Natural Science Foundation of China (no. 62272064), and the Natural Science Foundation of Hunan Province (no. 2023JJ60185).</p>
</sec>
<ack>
<p>The authors thank the referees for suggestions that helped improve the paper substantially.</p>
</ack>
<sec sec-type="COI-statement" id="sec22">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="sec23">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="sec24">
<title>Supplementary material</title>
<p>The Supplementary material for this article can be found online at: <ext-link xlink:href="https://www.frontiersin.org/articles/10.3389/fmicb.2024.1366272/full#supplementary-material" ext-link-type="uri">https://www.frontiersin.org/articles/10.3389/fmicb.2024.1366272/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Table_1.XLS" id="SM1" mimetype="application/vnd.ms-excel" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_2.XLSX" id="SM2" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_3.XLSX" id="SM3" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_4.XLSX" id="SM4" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_5.XLSX" id="SM5" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_6.XLSX" id="SM6" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_7.XLSX" id="SM7" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_8.XLSX" id="SM8" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<fn-group>
<fn id="fn0001"><p><sup>1</sup><ext-link xlink:href="https://github.com/longyahui/GCNMDA" ext-link-type="uri">https://github.com/longyahui/GCNMDA</ext-link></p></fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="ref1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Abdoulaye</surname> <given-names>O.</given-names></name> <name><surname>Amadou</surname> <given-names>M. L. H.</given-names></name> <name><surname>Amadou</surname> <given-names>O.</given-names></name> <name><surname>Adakal</surname> <given-names>O.</given-names></name> <name><surname>Larwanou</surname> <given-names>H. M.</given-names></name> <name><surname>Boubou</surname> <given-names>L.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Epidemiological and bacteriological features of surgical site infections (ISO) in the division of surgery at the Niamey National Hospital (HNN)</article-title>. <source>Pan Afr. Med. J.</source> <volume>31</volume>:<fpage>33</fpage>. doi: <pub-id pub-id-type="doi">10.11604/pamj.2018.31.33.15921</pub-id></citation>
</ref>
<ref id="ref2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Andersen</surname> <given-names>P. I.</given-names></name> <name><surname>Ianevski</surname> <given-names>A.</given-names></name> <name><surname>Lysvand</surname> <given-names>H.</given-names></name> <name><surname>Vitkauskiene</surname> <given-names>A.</given-names></name> <name><surname>Oksenych</surname> <given-names>V.</given-names></name> <name><surname>Bj&#x00F8;r&#x00E5;s</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Discovery and development of safe-in-man broad-spectrum antiviral agents</article-title>. <source>Int. J. Infect. Dis.</source> <volume>93</volume>, <fpage>268</fpage>&#x2013;<lpage>276</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.ijid.2020.02.018</pub-id>, PMID: <pub-id pub-id-type="pmid">32081774</pub-id></citation>
</ref>
<ref id="ref3">
<citation citation-type="book"><person-group person-group-type="author">
<name><surname>Anderson</surname> <given-names>T. W.</given-names></name>
</person-group> (<year>2003</year>). <source>An introduction to multivariate statistical analysis</source>: <publisher-loc>United States</publisher-loc>: <publisher-name>John Wiley&#x0026;Sons</publisher-name>.</citation>
</ref>
<ref id="ref4">
<citation citation-type="other"><person-group person-group-type="author"><name><surname>Baranwal</surname> <given-names>A.</given-names></name> <name><surname>Fountoulakis</surname> <given-names>K.</given-names></name> <name><surname>Jagannath</surname> <given-names>A.</given-names></name></person-group> (<year>2021</year>). Graph convolution for semisupervised classification: Improved linear separability and out-of-distribution generalization. In: <italic>International conference on machine learning (ICML)</italic>. PMLR.</citation>
</ref>
<ref id="ref5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cheng</surname> <given-names>X.</given-names></name> <name><surname>Qu</surname> <given-names>J.</given-names></name> <name><surname>Song</surname> <given-names>S.</given-names></name> <name><surname>Bian</surname> <given-names>Z.</given-names></name></person-group> (<year>2022</year>). <article-title>Neighborhood-based inference and restricted Boltzmann machine for microbe and drug associations prediction</article-title>. <source>PeerJ</source> <volume>10</volume>:<fpage>e13848</fpage>. doi: <pub-id pub-id-type="doi">10.7717/peerj.13848</pub-id>, PMID: <pub-id pub-id-type="pmid">35990901</pub-id></citation>
</ref>
<ref id="ref6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cho</surname> <given-names>E. H.</given-names></name> <name><surname>Huh</surname> <given-names>H. J.</given-names></name> <name><surname>Song</surname> <given-names>D. J.</given-names></name> <name><surname>Lee</surname> <given-names>S. H.</given-names></name> <name><surname>Kim</surname> <given-names>C. K.</given-names></name> <name><surname>Shin</surname> <given-names>S. Y.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Drug susceptibility patterns of mycobacterium abscessus and <italic>Mycobacterium massiliense</italic> isolated from respiratory specimens</article-title>. <source>Diagn. Microbiol. Infect. Dis.</source> <volume>93</volume>, <fpage>107</fpage>&#x2013;<lpage>111</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.diagmicrobio.2018.08.008</pub-id>, PMID: <pub-id pub-id-type="pmid">30236529</pub-id></citation>
</ref>
<ref id="ref7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dai</surname> <given-names>Q.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Duan</surname> <given-names>X.</given-names></name> <name><surname>Song</surname> <given-names>J.</given-names></name> <name><surname>Guo</surname> <given-names>M.</given-names></name></person-group> (<year>2022</year>). <article-title>Predicting mirna-disease associations using an ensemble learning framework with resampling method</article-title>. <source>Brief. Bioinform.</source> <volume>23</volume>:<fpage>bbab543</fpage>. doi: <pub-id pub-id-type="doi">10.1093/bib/bbab543</pub-id>, PMID: <pub-id pub-id-type="pmid">34929742</pub-id></citation>
</ref>
<ref id="ref8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Deng</surname> <given-names>L.</given-names></name> <name><surname>Huang</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Liu</surname> <given-names>H.</given-names></name></person-group> (<year>2022</year>). <article-title>Graph2mda: a multi-modal variational graph embedding model for predicting microbe&#x2013;drug associations</article-title>. <source>Bioinformatics</source> <volume>38</volume>, <fpage>1118</fpage>&#x2013;<lpage>1125</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btab792</pub-id>, PMID: <pub-id pub-id-type="pmid">34864873</pub-id></citation>
</ref>
<ref id="ref9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Durack</surname> <given-names>J.</given-names></name> <name><surname>Lynch</surname> <given-names>S. V.</given-names></name></person-group> (<year>2019</year>). <article-title>The gut microbiome: relationships with disease and opportunities for therapy</article-title>. <source>J. Exp. Med.</source> <volume>216</volume>, <fpage>20</fpage>&#x2013;<lpage>40</lpage>. doi: <pub-id pub-id-type="doi">10.1084/jem.20180448</pub-id>, PMID: <pub-id pub-id-type="pmid">30322864</pub-id></citation>
</ref>
<ref id="ref10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fan</surname> <given-names>L.</given-names></name> <name><surname>Wang</surname> <given-names>L.</given-names></name> <name><surname>Zhu</surname> <given-names>X.</given-names></name></person-group> (<year>2023</year>). <article-title>A novel microbe-drug association prediction model based on stacked autoencoder with multi-head attention mechanism</article-title>. <source>Sci. Rep.</source> <volume>13</volume>:<fpage>7396</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41598-023-34438-8</pub-id>, PMID: <pub-id pub-id-type="pmid">37149692</pub-id></citation>
</ref>
<ref id="ref11">
<citation citation-type="other"><person-group person-group-type="author"><name><surname>Fountoulakis</surname> <given-names>K.</given-names></name> <name><surname>Levi</surname> <given-names>A.</given-names></name> <name><surname>Yang</surname> <given-names>S.</given-names></name> <name><surname>Baranwal</surname> <given-names>A.</given-names></name> <name><surname>Jagannath</surname> <given-names>A.</given-names></name></person-group> (<year>2022</year>). Graph Attention Retrospective. arXiv [Preprint].</citation>
</ref>
<ref id="ref12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gill</surname> <given-names>S. R.</given-names></name> <name><surname>Pop</surname> <given-names>M.</given-names></name> <name><surname>DeBoy</surname> <given-names>R. T.</given-names></name> <name><surname>Eckburg</surname> <given-names>P. B.</given-names></name> <name><surname>Turnbaugh</surname> <given-names>P. J.</given-names></name> <name><surname>Samuel</surname> <given-names>B. S.</given-names></name> <etal/></person-group>. (<year>2006</year>). <article-title>Metagenomic analysis of the human distal gut microbiome</article-title>. <source>Science</source> <volume>312</volume>, <fpage>1355</fpage>&#x2013;<lpage>1359</lpage>. doi: <pub-id pub-id-type="doi">10.1126/science.1124234</pub-id>, PMID: <pub-id pub-id-type="pmid">16741115</pub-id></citation>
</ref>
<ref id="ref13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hacioglu</surname> <given-names>M.</given-names></name> <name><surname>Haciosmanoglu</surname> <given-names>E.</given-names></name> <name><surname>Birteksoz-Tan</surname> <given-names>A. S.</given-names></name> <name><surname>Bozkurt-Guzel</surname> <given-names>C.</given-names></name> <name><surname>Savage</surname> <given-names>P. B.</given-names></name></person-group> (<year>2019</year>). <article-title>Effects of ceragenins and conventional antimicrobials on Candida albicans and <italic>Staphylococcus aureus</italic> mono and multispecies biofilms</article-title>. <source>Diagn. Microbiol. Infect. Dis.</source> <volume>95</volume>:<fpage>114863</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.diagmicrobio.2019.06.014</pub-id>, PMID: <pub-id pub-id-type="pmid">31471074</pub-id></citation>
</ref>
<ref id="ref14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hattori</surname> <given-names>M.</given-names></name> <name><surname>Tanaka</surname> <given-names>N.</given-names></name> <name><surname>Kanehisa</surname> <given-names>M.</given-names></name> <name><surname>Goto</surname> <given-names>S.</given-names></name></person-group> (<year>2010</year>). <article-title>Simcomp/subcomp:chemical structure search servers for network analyses</article-title>. <source>Nucleic Acids Res.</source> <volume>38</volume>, <fpage>W652</fpage>&#x2013;<lpage>W656</lpage>. doi: <pub-id pub-id-type="doi">10.1093/nar/gkq367</pub-id>, PMID: <pub-id pub-id-type="pmid">20460463</pub-id></citation>
</ref>
<ref id="ref15">
<citation citation-type="other"><person-group person-group-type="author"><name><surname>Javaloy</surname> <given-names>A.</given-names></name> <name><surname>Sanchez-Mart&#x0131;n</surname> <given-names>P.</given-names></name> <name><surname>Levi</surname> <given-names>A.</given-names></name> <name><surname>Valera</surname> <given-names>I.</given-names></name></person-group> Learnable graph convolutional attention networks. ICLR. arXiv [Preprint]. (<year>2023</year>).</citation>
</ref>
<ref id="ref16">
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Kamneva</surname> <given-names>O. K.</given-names></name>
</person-group> (<year>2017</year>). <article-title>Genome composition and phylogeny of microbes predict their co-occurrence in the environment</article-title>. <source>PLoS Comput. Biol.</source> <volume>13</volume>:<fpage>e1005366</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pcbi.1005366</pub-id>, PMID: <pub-id pub-id-type="pmid">28152007</pub-id></citation>
</ref>
<ref id="ref17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kau</surname> <given-names>A. L.</given-names></name> <name><surname>Ahern</surname> <given-names>P. P.</given-names></name> <name><surname>Griffin</surname> <given-names>N. W.</given-names></name> <name><surname>Goodman</surname> <given-names>A. L.</given-names></name> <name><surname>Gordon</surname> <given-names>J. I.</given-names></name></person-group> (<year>2011</year>). <article-title>Human nutrition, the gut microbiome and the immune system</article-title>. <source>Nature</source> <volume>474</volume>, <fpage>327</fpage>&#x2013;<lpage>336</lpage>. doi: <pub-id pub-id-type="doi">10.1038/nature10213</pub-id>, PMID: <pub-id pub-id-type="pmid">21677749</pub-id></citation>
</ref>
<ref id="ref18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Knyazev</surname> <given-names>B.</given-names></name> <name><surname>Taylor</surname> <given-names>G. W.</given-names></name> <name><surname>Amer</surname> <given-names>M.</given-names></name></person-group> (<year>2019</year>). <article-title>Understanding attention and generalization in graph neural networks</article-title>. <source>Adv. Neural Inf. Proces. Syst.</source> <volume>32</volume>, <fpage>4204</fpage>&#x2013;<lpage>4214</lpage>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.1905.02850</pub-id></citation>
</ref>
<ref id="ref19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>F.</given-names></name> <name><surname>Dong</surname> <given-names>S.</given-names></name> <name><surname>Leier</surname> <given-names>A.</given-names></name> <name><surname>Han</surname> <given-names>M.</given-names></name> <name><surname>Guo</surname> <given-names>X.</given-names></name> <name><surname>Xu</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Positive-unlabeled learning in bioinformatics and computational biology: a brief review</article-title>. <source>Brief. Bioinform.</source> <volume>23</volume>:<fpage>bbab461</fpage>. doi: <pub-id pub-id-type="doi">10.1093/bib/bbab461</pub-id>, PMID: <pub-id pub-id-type="pmid">34729589</pub-id></citation>
</ref>
<ref id="ref20">
<citation citation-type="other"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Cao</surname> <given-names>W.</given-names></name> <name><surname>Gao</surname> <given-names>Z.</given-names></name> <name><surname>Bian</surname> <given-names>J.</given-names></name> <name><surname>Chen</surname> <given-names>H.</given-names></name> <name><surname>Chang</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2020</year>). Self-paced Ensemble for Highly Imbalanced Massive Data Classification. In: <italic>2020 IEEE 36th International Conference on Data Engineering (ICDE)</italic>; 841&#x2013;852.</citation>
</ref>
<ref id="ref21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Long</surname> <given-names>Y.</given-names></name> <name><surname>Min</surname> <given-names>W.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Fang</surname> <given-names>Y.</given-names></name> <name><surname>Kwoh</surname> <given-names>C. K.</given-names></name> <name><surname>Chen</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Pre-training graph neural networks for link prediction in biomedical networks</article-title>. <source>Bioinformatics</source> <volume>38</volume>, <fpage>2254</fpage>&#x2013;<lpage>2262</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btac100</pub-id>, PMID: <pub-id pub-id-type="pmid">35171981</pub-id></citation>
</ref>
<ref id="ref22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Long</surname> <given-names>Y.</given-names></name> <name><surname>Wu</surname> <given-names>M.</given-names></name> <name><surname>Kwoh</surname> <given-names>C.-K.</given-names></name> <name><surname>Luo</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name></person-group> (<year>2020a</year>). <article-title>Predicting human microbe-drug associations via graph convolutional network with conditional random field</article-title>. <source>Bioinformatics</source> <volume>36</volume>, <fpage>4918</fpage>&#x2013;<lpage>4927</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa598</pub-id>, PMID: <pub-id pub-id-type="pmid">32597948</pub-id></citation>
</ref>
<ref id="ref23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Long</surname> <given-names>Y.</given-names></name> <name><surname>Wu</surname> <given-names>M.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Kwoh</surname> <given-names>C. K.</given-names></name> <name><surname>Luo</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name></person-group> (<year>2020b</year>). <article-title>Ensembling graph attention networks for human microbe&#x2013;drug association prediction</article-title>. <source>Bioinformatics</source> <volume>36</volume>, <fpage>i779</fpage>&#x2013;<lpage>i786</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa891</pub-id>, PMID: <pub-id pub-id-type="pmid">33381844</pub-id></citation>
</ref>
<ref id="ref24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>L&#x00F3;pez</surname> <given-names>V.</given-names></name> <name><surname>Fern&#x00E1;ndez</surname> <given-names>A.</given-names></name> <name><surname>Garc&#x00ED;a</surname> <given-names>S.</given-names></name> <name><surname>Palade</surname> <given-names>V.</given-names></name> <name><surname>Herrera</surname> <given-names>F.</given-names></name></person-group> (<year>2013</year>). <article-title>An insight into classification with imbalanced data: empirical results and current trends on using data intrinsic characteristics</article-title>. <source>Inform. Sci.</source> <volume>250</volume>, <fpage>113</fpage>&#x2013;<lpage>141</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.ins.2013.07.007</pub-id></citation>
</ref>
<ref id="ref25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lou</surname> <given-names>Z.</given-names></name> <name><surname>Cheng</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Teng</surname> <given-names>Z.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Tian</surname> <given-names>Z.</given-names></name></person-group> (<year>2022</year>). <article-title>Predicting miRNA-disease associations via learning multimodal networks and fusing mixed neighborhood information</article-title>. <source>Brief. Bioinform.</source> <volume>23</volume>:<fpage>bbac159</fpage>. doi: <pub-id pub-id-type="doi">10.1093/bib/bbac159</pub-id></citation>
</ref>
<ref id="ref26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Luo</surname> <given-names>J.</given-names></name> <name><surname>Long</surname> <given-names>Y.</given-names></name></person-group> (<year>2020</year>). <article-title>NTSHMDA: prediction of human microbe-disease association based on random walk by integrating network topological similarity</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinform.</source> <volume>17</volume>, <fpage>1341</fpage>&#x2013;<lpage>1351</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TCBB.2018.2883041</pub-id>, PMID: <pub-id pub-id-type="pmid">30489271</pub-id></citation>
</ref>
<ref id="ref27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ma</surname> <given-names>Q.</given-names></name> <name><surname>Tan</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>L.</given-names></name></person-group> (<year>2023</year>). <article-title>GACNNMDA: a computational model for predicting potential human microbe-drug associations based on graph attention network and CNN-based classifier</article-title>. <source>BMC Bioinformatics</source> <volume>24</volume>:<fpage>35</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s12859-023-05158-7</pub-id>, PMID: <pub-id pub-id-type="pmid">36732704</pub-id></citation>
</ref>
<ref id="ref28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Macpherson</surname> <given-names>A. J.</given-names></name> <name><surname>Harris</surname> <given-names>N. L.</given-names></name></person-group> (<year>2004</year>). <article-title>Interactions between commensal intestinal bacteria and the immune system</article-title>. <source>Nat. Rev. Immunol.</source> <volume>4</volume>, <fpage>478</fpage>&#x2013;<lpage>485</lpage>. doi: <pub-id pub-id-type="doi">10.1038/nri1373</pub-id></citation>
</ref>
<ref id="ref29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>McCoubrey</surname> <given-names>L. E.</given-names></name> <name><surname>Gaisford</surname> <given-names>S.</given-names></name> <name><surname>Orlu</surname> <given-names>M.</given-names></name> <name><surname>Basit</surname> <given-names>A. W.</given-names></name></person-group> (<year>2022</year>). <article-title>Predicting drug-microbiome interactions with machine learning</article-title>. <source>Biotechnol. Adv.</source> <volume>54</volume>:<fpage>107797</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.biotechadv.2021.107797</pub-id>, PMID: <pub-id pub-id-type="pmid">34260950</pub-id></citation>
</ref>
<ref id="ref30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Moin</surname> <given-names>S.</given-names></name> <name><surname>Zeeshan</surname> <given-names>M.</given-names></name> <name><surname>Laiq</surname> <given-names>S.</given-names></name> <name><surname>Raheem</surname> <given-names>A.</given-names></name> <name><surname>Zafar</surname> <given-names>A.</given-names></name></person-group> (<year>2020</year>). <article-title>Use of pefloxacin as a surrogate marker to detect ciprofloxacin susceptibility in <italic>Salmonella enterica</italic> serotypes Typhi and Paratyphi a</article-title>. <source>J. Pak. Med. Assoc.</source> <volume>70</volume>, <fpage>96</fpage>&#x2013;<lpage>99</lpage>. doi: <pub-id pub-id-type="doi">10.5455/JPMA.8635</pub-id>, PMID: <pub-id pub-id-type="pmid">31954032</pub-id></citation>
</ref>
<ref id="ref31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Panebianco</surname> <given-names>C.</given-names></name> <name><surname>Andriulli</surname> <given-names>A.</given-names></name> <name><surname>Pazienza</surname> <given-names>V.</given-names></name></person-group> (<year>2018</year>). <article-title>Pharmacomicrobiomics: exploiting the drug-microbiota interactions in anticancer therapies</article-title>. <source>Microbiome</source> <volume>6</volume>:<fpage>92</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s40168-018-0483-7</pub-id>, PMID: <pub-id pub-id-type="pmid">29789015</pub-id></citation>
</ref>
<ref id="ref32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qu</surname> <given-names>J.</given-names></name> <name><surname>Song</surname> <given-names>Z.</given-names></name> <name><surname>Cheng</surname> <given-names>X.</given-names></name> <name><surname>Jiang</surname> <given-names>Z.</given-names></name> <name><surname>Zhou</surname> <given-names>J.</given-names></name></person-group> (<year>2023</year>). <article-title>A new integrated framework for the identification of potential virus&#x2013;drug associations</article-title>. <source>Front. Microbiol.</source> <volume>14</volume>:<fpage>414</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fmicb.2023.1179414</pub-id>, PMID: <pub-id pub-id-type="pmid">37675432</pub-id></citation>
</ref>
<ref id="ref33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rajput</surname> <given-names>A.</given-names></name> <name><surname>Thakur</surname> <given-names>A.</given-names></name> <name><surname>Sharma</surname> <given-names>S.</given-names></name> <name><surname>Kumar</surname> <given-names>M.</given-names></name></person-group> (<year>2018</year>). <article-title>aBioflm: a resource of anti-bioflm agents and their potential implications in targeting antibiotic drug resistance</article-title>. <source>Nucleic Acids Res.</source> <volume>46</volume>, <fpage>D894</fpage>&#x2013;<lpage>D900</lpage>. doi: <pub-id pub-id-type="doi">10.1093/nar/gkx1157</pub-id>, PMID: <pub-id pub-id-type="pmid">29156005</pub-id></citation>
</ref>
<ref id="ref34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schwabe</surname> <given-names>R. F.</given-names></name> <name><surname>Jobin</surname> <given-names>C.</given-names></name></person-group> (<year>2013</year>). <article-title>The microbiome and cancer</article-title>. <source>Nat. Rev. Cancer</source> <volume>13</volume>, <fpage>800</fpage>&#x2013;<lpage>812</lpage>. doi: <pub-id pub-id-type="doi">10.1038/nrc3610</pub-id>, PMID: <pub-id pub-id-type="pmid">24132111</pub-id></citation>
</ref>
<ref id="ref35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sommer</surname> <given-names>F.</given-names></name> <name><surname>B&#x00E4;ckhed</surname> <given-names>F.</given-names></name></person-group> (<year>2013</year>). <article-title>The gut microbiota &#x2014; masters of host development and physiology</article-title>. <source>Nat. Rev. Microbiol.</source> <volume>11</volume>, <fpage>227</fpage>&#x2013;<lpage>238</lpage>. doi: <pub-id pub-id-type="doi">10.1038/nrmicro2974</pub-id>, PMID: <pub-id pub-id-type="pmid">23435359</pub-id></citation>
</ref>
<ref id="ref36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>Y.-Z.</given-names></name> <name><surname>Zhang</surname> <given-names>D.-H.</given-names></name> <name><surname>Cai</surname> <given-names>S.-B.</given-names></name> <name><surname>Ming</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>J. Q.</given-names></name> <name><surname>Chen</surname> <given-names>X.</given-names></name></person-group> (<year>2018</year>). <article-title>MDAD: a special resource for microbe-drug associations</article-title>. <source>Microbiology</source> <volume>8</volume>:<fpage>424</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fcimb.2018.00424</pub-id></citation>
</ref>
<ref id="ref37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Szklarczyk</surname> <given-names>D.</given-names></name> <name><surname>Gable</surname> <given-names>A. L.</given-names></name> <name><surname>Nastou</surname> <given-names>K. C.</given-names></name> <name><surname>Lyon</surname> <given-names>D.</given-names></name> <name><surname>Kirsch</surname> <given-names>R.</given-names></name> <name><surname>Pyysalo</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Correction to &#x2018;the STRING database in 2021: customizable protein&#x2013;protein networks, and functional characterization of user-uploaded gene/measurement sets&#x2019;</article-title>. <source>Nucleic Acids Res.</source> <volume>49</volume>:<fpage>10800</fpage>. doi: <pub-id pub-id-type="doi">10.1093/nar/gkab835</pub-id>, PMID: <pub-id pub-id-type="pmid">34530444</pub-id></citation>
</ref>
<ref id="ref38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tan</surname> <given-names>Y.</given-names></name> <name><surname>Zou</surname> <given-names>J.</given-names></name> <name><surname>Kuang</surname> <given-names>L.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Zeng</surname> <given-names>B.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>GSAMDA: a computational model for predicting potential microbe&#x2013;drug associations based on graph attention network and sparse autoencoder</article-title>. <source>BMC Bioinformatics</source> <volume>23</volume>:<fpage>492</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s12859-022-05053-7</pub-id>, PMID: <pub-id pub-id-type="pmid">36401174</pub-id></citation>
</ref>
<ref id="ref39">
<citation citation-type="journal"><person-group person-group-type="author">
<collab id="coll1">The Human Microbiome Project Consortium</collab>
</person-group> (<year>2012</year>). <article-title>Structure, function and diversity of the healthy human microbiome</article-title>. <source>Nature</source> <volume>486</volume>, <fpage>207</fpage>&#x2013;<lpage>214</lpage>. doi: <pub-id pub-id-type="doi">10.1038/nature11234</pub-id>, PMID: <pub-id pub-id-type="pmid">22699609</pub-id></citation>
</ref>
<ref id="ref40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Trinh</surname> <given-names>S. A.</given-names></name> <name><surname>Gavin</surname> <given-names>H. E.</given-names></name> <name><surname>Satchell</surname> <given-names>K. J. F.</given-names></name></person-group> (<year>2017</year>). <article-title>Efficacy of ceftriaxone, cefepime, doxycycline, ciprofloxacin, and combination therapy for <italic>Vibrio vulnificus</italic> foodborne septicemia</article-title>. <source>Antimicrob. Agents Chemother.</source> <volume>61</volume>, <fpage>e01106</fpage>&#x2013;<lpage>e01117</lpage>. doi: <pub-id pub-id-type="doi">10.1128/AAC.01106-17</pub-id></citation>
</ref>
<ref id="ref41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ventura</surname> <given-names>M.</given-names></name> <name><surname>O'Flaherty</surname> <given-names>S.</given-names></name> <name><surname>Claesson</surname> <given-names>M. J.</given-names></name> <name><surname>Turroni</surname> <given-names>F.</given-names></name> <name><surname>Klaenhammer</surname> <given-names>T. R.</given-names></name> <name><surname>van Sinderen</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2009</year>). <article-title>Genome-scale analyses of health-promoting bacteria: probiogenomics</article-title>. <source>Nat. Rev. Microbiol.</source> <volume>7</volume>, <fpage>61</fpage>&#x2013;<lpage>71</lpage>. doi: <pub-id pub-id-type="doi">10.1038/nrmicro2047</pub-id>, PMID: <pub-id pub-id-type="pmid">19029955</pub-id></citation>
</ref>
<ref id="ref42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>F.</given-names></name> <name><surname>Huang</surname> <given-names>Z.-A.</given-names></name> <name><surname>Chen</surname> <given-names>X.</given-names></name> <name><surname>Zhu</surname> <given-names>Z.</given-names></name> <name><surname>Wen</surname> <given-names>Z.</given-names></name> <name><surname>Zhao</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>LRLSHMDA: Laplacian regularized least squares for human microbe-disease association prediction</article-title>. <source>Sci. Rep.</source> <volume>7</volume>:<fpage>7601</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41598-017-08127-2</pub-id>, PMID: <pub-id pub-id-type="pmid">28790448</pub-id></citation>
</ref>
<ref id="ref43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>L.</given-names></name> <name><surname>Tan</surname> <given-names>Y.</given-names></name> <name><surname>Yang</surname> <given-names>X.</given-names></name> <name><surname>Kuang</surname> <given-names>L.</given-names></name> <name><surname>Ping</surname> <given-names>P.</given-names></name></person-group> (<year>2022</year>). <article-title>Review on predicting pairwise relationships between human microbes, drugs and diseases: from biological data to computational models</article-title>. <source>Brief. Bioinform.</source> <volume>23</volume>:<fpage>bbac080</fpage>. doi: <pub-id pub-id-type="doi">10.1093/bib/bbac080</pub-id>, PMID: <pub-id pub-id-type="pmid">35325024</pub-id></citation>
</ref>
<ref id="ref44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Q.</given-names></name> <name><surname>Zhao</surname> <given-names>K.</given-names></name> <name><surname>Guo</surname> <given-names>C.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Huang</surname> <given-names>T.</given-names></name> <name><surname>Ji</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Antibiotic resistance and virulence genes of <italic>Escherichia coli</italic> isolated from patients with urinary tract infections after kidney transplantation from deceased donors</article-title>. <source>Infect Drug Resist</source> <volume>14</volume>, <fpage>4039</fpage>&#x2013;<lpage>4046</lpage>. doi: <pub-id pub-id-type="doi">10.2147/IDR.S332897</pub-id>, PMID: <pub-id pub-id-type="pmid">34616161</pub-id></citation>
</ref>
<ref id="ref45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wei</surname> <given-names>H.</given-names></name> <name><surname>Xu</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>B.</given-names></name></person-group> (<year>2021</year>). <article-title>Ipidi-pul: identifying piwi-interacting rnadisease associations based on positive unlabeled learning</article-title>. <source>Brief. Bioinform.</source> <volume>22</volume>:<fpage>bbaa058</fpage>. doi: <pub-id pub-id-type="doi">10.1093/bib/bbaa058</pub-id>, PMID: <pub-id pub-id-type="pmid">32393982</pub-id></citation>
</ref>
<ref id="ref46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wen</surname> <given-names>L.</given-names></name> <name><surname>Ley</surname> <given-names>R. E.</given-names></name> <name><surname>Volchkov</surname> <given-names>P. Y.</given-names></name> <name><surname>Stranges</surname> <given-names>P. B.</given-names></name> <name><surname>Avanesyan</surname> <given-names>L.</given-names></name> <name><surname>Stonebraker</surname> <given-names>A. C.</given-names></name> <etal/></person-group>. (<year>2008</year>). <article-title>Innate immunity and intestinal microbiota in the development of type 1 diabetes</article-title>. <source>Nature</source> <volume>455</volume>, <fpage>1109</fpage>&#x2013;<lpage>1113</lpage>. doi: <pub-id pub-id-type="doi">10.1038/nature07336</pub-id>, PMID: <pub-id pub-id-type="pmid">18806780</pub-id></citation>
</ref>
<ref id="ref47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xiang</surname> <given-names>Y.-T.</given-names></name> <name><surname>Li</surname> <given-names>W.</given-names></name> <name><surname>Zhang</surname> <given-names>Q.</given-names></name> <name><surname>Jin</surname> <given-names>Y.</given-names></name> <name><surname>Rao</surname> <given-names>W. W.</given-names></name> <name><surname>Zeng</surname> <given-names>L. N.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Timely research papers about COVID-19 in China</article-title>. <source>Lancet</source> <volume>395</volume>, <fpage>684</fpage>&#x2013;<lpage>685</lpage>. doi: <pub-id pub-id-type="doi">10.1016/S0140-6736(20)30375-5</pub-id>, PMID: <pub-id pub-id-type="pmid">32078803</pub-id></citation>
</ref>
<ref id="ref48">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>P.</given-names></name> <name><surname>Li</surname> <given-names>X.-L.</given-names></name> <name><surname>Mei</surname> <given-names>J.-P.</given-names></name> <name><surname>Kwoh</surname> <given-names>C. K.</given-names></name> <name><surname>Ng</surname> <given-names>S. K.</given-names></name></person-group> (<year>2012</year>). <article-title>Positive-unlabeled learning for disease gene identification</article-title>. <source>Bioinformatics</source> <volume>28</volume>, <fpage>2640</fpage>&#x2013;<lpage>2647</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/bts504</pub-id>, PMID: <pub-id pub-id-type="pmid">22923290</pub-id></citation>
</ref>
<ref id="ref49">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zeng</surname> <given-names>X.</given-names></name> <name><surname>Zhong</surname> <given-names>Y.</given-names></name> <name><surname>Lin</surname> <given-names>W.</given-names></name> <name><surname>Zou</surname> <given-names>Q.</given-names></name></person-group> (<year>2020</year>). <article-title>Predicting disease-associated circular rnas using deep forests combined with positive-unlabeled learning methods</article-title>. <source>Brief. Bioinform.</source> <volume>21</volume>, <fpage>1425</fpage>&#x2013;<lpage>1436</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bib/bbz080</pub-id>, PMID: <pub-id pub-id-type="pmid">31612203</pub-id></citation>
</ref>
<ref id="ref50">
<citation citation-type="other"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>L.</given-names></name> <name><surname>Duan</surname> <given-names>G.</given-names></name> <name><surname>Yan</surname> <given-names>C.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name></person-group> (<year>2019</year>). Prediction of microbe-drug associations based on KATZ measure. In: <italic>IEEE International Conference on Bioinformatics and Biomedicine (BIBM) 2019</italic>; 183&#x2013;187.</citation>
</ref>
<ref id="ref51">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zimmermann</surname> <given-names>M.</given-names></name> <name><surname>Zimmermann-Kogadeeva</surname> <given-names>M.</given-names></name> <name><surname>Wegmann</surname> <given-names>R.</given-names></name> <name><surname>Goodman</surname> <given-names>A. L.</given-names></name></person-group> (<year>2019</year>). <article-title>Mapping human microbiome drug metabolism by gut bacteria and their genes</article-title>. <source>Nature</source> <volume>570</volume>, <fpage>462</fpage>&#x2013;<lpage>467</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41586-019-1291-3</pub-id>, PMID: <pub-id pub-id-type="pmid">31158845</pub-id></citation>
</ref>
</ref-list>
</back>
</article>