<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1535279</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2025.1535279</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>CLMT: graph contrastive learning model for microbe-drug associations prediction with transformer</article-title>
<alt-title alt-title-type="left-running-head">Xiao et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2025.1535279">10.3389/fgene.2025.1535279</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Xiao</surname>
<given-names>Liqi</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2906226/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wu</surname>
<given-names>Junlong</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2868279/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Fan</surname>
<given-names>Liu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Wang</surname>
<given-names>Lei</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/664933/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Zhu</surname>
<given-names>Xianyou</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>College of Computer Science and Technology</institution>, <institution>Hengyang Normal University</institution>, <addr-line>Hengyang</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Technology Innovation Center of Changsha</institution>, <institution>Changsha University</institution>, <addr-line>Changsha</addr-line>, <country>China</country>
</aff> <aff id="aff3">
<sup>3</sup>
<institution>Hunan Engineering Research Center of Cyberspace Security Technology and Applications</institution>, <institution>Hengyang Normal University</institution>, <addr-line>Hengyang</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/522202/overview">Andrei Rodin</ext-link>, City of Hope National Medical Center, United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/381999/overview">Akanksha Rajput</ext-link>, University of California, San Diego, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2328461/overview">Weiqiang Jin</ext-link>, Xi&#x2019;an Jiaotong University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2937005/overview">Mythili R.</ext-link>, SRM Institute of Science and Technology, India</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Lei Wang, <email>wanglei@xtu.edu.cn</email>; Xianyou Zhu, <email>zxy@hynu.edu.cn</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>12</day>
<month>03</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1535279</elocation-id>
<history>
<date date-type="received">
<day>27</day>
<month>11</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>21</day>
<month>02</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Xiao, Wu, Fan, Wang and Zhu.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Xiao, Wu, Fan, Wang and Zhu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Accurate prediction of microbe-drug associations is essential for drug development and disease diagnosis. However, existing methods often struggle to capture complex nonlinear relationships, effectively model long-range dependencies, and distinguish subtle similarities between microbes and drugs. To address these challenges, this paper introduces a new model for microbe-drug association prediction, CLMT. The proposed model differs from previous approaches in three key ways. Firstly, unlike conventional GCN-based models, CLMT leverages a Graph Transformer network with an attention mechanism to model high-order dependencies in the microbe-drug interaction graph, enhancing its ability to capture long-range associations. Then, we introduce graph contrastive learning, generating multiple augmented views through node perturbation and edge dropout. By optimizing a contrastive loss, CLMT distinguishes subtle structural variations, making the learned embeddings more robust and generalizable. By integrating multi-view contrastive learning and Transformer-based encoding, CLMT effectively mitigates data sparsity issues, significantly outperforming existing methods. Experimental results on three publicly available datasets demonstrate that CLMT achieves state-of-the-art performance, particularly in handling sparse data and nonlinear microbe-drug interactions, confirming its effectiveness for real-world biomedical applications. On the MDAD, aBiofilm, and Drug Virus datasets, CLMT outperforms the previously best model in terms of Accuracy by 4.3%, 3.5%, and 2.8%, respectively.</p>
</abstract>
<kwd-group>
<kwd>microbe-drug association</kwd>
<kwd>graph transformer</kwd>
<kwd>similarity matrices</kwd>
<kwd>contrastive learning</kwd>
<kwd>nonlinear relationships</kwd>
<kwd>prediction accuracy</kwd>
<kwd>graph augmentation</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Genomics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>The human body hosts trillions of microorganisms, including bacteria, archaea, fungi, protozoa, and viruses, collectively forming the human microbiota, which interacts closely with its host (<xref ref-type="bibr" rid="B12">Gevers et al., 2012</xref>; <xref ref-type="bibr" rid="B42">Sommer and B&#xe4;ckhed, 2013</xref>). These microorganisms inhabit various regions such as the skin, oral and nasal cavities, gastrointestinal tract, and genitourinary system, exerting profound effects on health. For instance, they regulate gastrointestinal function, support internal balance, and facilitate metabolic activities (<xref ref-type="bibr" rid="B13">Gill et al., 2006</xref>; <xref ref-type="bibr" rid="B46">Ventura et al., 2009</xref>). Additionally, the microbiota collaborates with mucosal barriers to prevent pathogen invasion (<xref ref-type="bibr" rid="B31">Macpherson and Harris, 2004</xref>). Microbes also contribute to processes like sugar metabolism and vitamin synthesis, both critical for T-cell response (<xref ref-type="bibr" rid="B21">Kau et al., 2011</xref>). However, an imbalance in microbial populations, or dysbiosis, can lead to conditions such as diabetes (<xref ref-type="bibr" rid="B51">Wen et al., 2008</xref>), inflammatory bowel disease (<xref ref-type="bibr" rid="B10">Durack and Lynch, 2019</xref>), and even cancer (<xref ref-type="bibr" rid="B41">Schwabe and Jobin, 2013</xref>). Furthermore, pathogens like certain bacteria and viruses are linked to numerous infectious diseases, including pneumococcal pneumonia, with evidence suggesting involvement in up to 27 conditions (<xref ref-type="bibr" rid="B47">Wang D. et al., 2020</xref>). The overuse and misuse of medications in recent years have accelerated microbial resistance, creating significant obstacles for clinical treatments and drug development. Microbial metabolism also influences drug efficacy, absorption, and toxicity, highlighting its critical role in pharmacology (<xref ref-type="bibr" rid="B59">Zimmermann and Curtis, 2019</xref>; <xref ref-type="bibr" rid="B33">McCoubrey et al., 2023</xref>). For example, interactions between intestinal flora and anticancer drugs can alter therapeutic outcomes and side effects. Strategies such as probiotics, prebiotics, synbiotics, biologics, and antibiotics have been proposed to manage microbial populations and enhance treatment effectiveness (<xref ref-type="bibr" rid="B36">Panebianco et al., 2018</xref>). Consequently, identifying microbe-drug relationships is a vital challenge in precision medicine, underscoring the urgent need for advanced computational models to explore these interactions.</p>
<p>In recent years, the rise of microbial resistance has paralleled the increasing diversity of drug candidates explored by the medical community (<xref ref-type="bibr" rid="B20">Jiang et al., 2024</xref>). Traditional pharmaceutical research often relied on cultivating specific microbial populations under controlled conditions before integrating them into drugs, a process that is both time-intensive and laborious. This challenge underscores the pressing need for advanced computational methods to identify potential microbe-drug relationships, which could revolutionize drug discovery and disease diagnosis (<xref ref-type="bibr" rid="B19">Jiang et al., 2023</xref>; <xref ref-type="bibr" rid="B18">Jiang et al., 2025</xref>). The advent of bioinformatics has facilitated the establishment of several databases documenting experimentally validated microbe-drug associations, including MDAD (<xref ref-type="bibr" rid="B43">Sun et al., 2018</xref>), aBiofilm (<xref ref-type="bibr" rid="B38">Rajput et al., 2018</xref>), and DrugVirus (<xref ref-type="bibr" rid="B3">Andersen et al., 2020</xref>).</p>
<p>To complement these resources, numerous computational approaches have emerged. For instance, HMDAKATZ, developed by (<xref ref-type="bibr" rid="B58">Zhu et al., 2019</xref>), utilizes KATZ metrics within a heterogeneous network to predict microbial-drug correlations. However, its applicability is limited for novel drugs without known microbial associations or isolated microbes lacking disease links. Similarly (<xref ref-type="bibr" rid="B25">Long et al., 2020a</xref>), introduced EGATMDA, a graph attention network-based framework with hierarchical attention mechanisms for analyzing microbial-drug interactions. Despite its innovation, this model&#x2019;s accuracy is constrained by its reliance on pre-existing association data for similarity computation.</p>
<p>Another approach, WHGMF, proposed by <xref ref-type="bibr" rid="B30">Ma and Liu (2022)</xref>, employs weighted hypergraph learning with generalized matrix decomposition to estimate potential microbe-drug interactions. Yet, it overlooks critical biological details, such as microbial sequences and drug side effect-based similarities, which diminishes prediction accuracy. GCNMDA, introduced by <xref ref-type="bibr" rid="B25">Long et al. (2020a)</xref>, combines graph convolutional networks and conditional random fields with an attention mechanism to predict microbial-drug associations. Nevertheless, its performance is hindered by noise within extracted similarity features.</p>
<p>
<xref ref-type="bibr" rid="B9">Deng et al. (2022)</xref> presented Graph2MDA, which uses multimodal attribute graphs and a variogram self-encoder to analyze node-level information and infer potential interactions. In contrast (<xref ref-type="bibr" rid="B44">Tan et al., 2022</xref>), proposed GSAMDA, a model integrating graph attention networks with sparse self-encoders to compute microbe-drug correlations. However, GSAMDA struggles with sparse data matrices, limiting its effectiveness. Although these computational models exhibit strengths in certain areas, each faces distinct challenges, emphasizing the need for continued innovation in this field.</p>
<p>In binary relation prediction, selecting appropriate negative samples is critical for effective model training. However, identifying informative negative samples from a pool of candidate negatives remains a significant challenge (<xref ref-type="bibr" rid="B23">Li et al., 2022</xref>). This issue is particularly evident in link prediction tasks, where generating meaningful negative samples has long been a persistent problem. Conventional machine learning methods typically classify known associations between entities (labeled samples) as positive samples, while unrecognized or unlabeled associations are treated as candidate negatives (<xref ref-type="bibr" rid="B53">Yang et al., 2012</xref>). Yet, due to the scarcity of known microbe-drug associations in publicly available datasets, the imbalance between positive and negative samples becomes a critical issue. To mitigate this imbalance and preserve model performance, advanced negative sampling strategies are essential.</p>
<p>The most widely used approach, random sampling, involves selecting a subset of negative samples equal in number to the positive samples (<xref ref-type="bibr" rid="B28">Lou et al., 2022</xref>). While straightforward, this method often fails to prioritize informative negatives and may include irrelevant or noisy examples (<xref ref-type="bibr" rid="B27">L&#xf3;pez et al., 2013</xref>). Efforts to enhance negative sampling strategies (<xref ref-type="bibr" rid="B57">Zeng et al., 2020</xref>; <xref ref-type="bibr" rid="B50">Wei et al., 2021</xref>; <xref ref-type="bibr" rid="B8">Dai et al., 2022</xref>) have achieved limited success, as they do not sufficiently focus on identifying the most valuable negatives critical for effective classifier training. This oversight can result in undertraining and reduced predictive performance.</p>
<p>To address these limitations, we developed a novel microbe-drug association prediction model, CLMT. This model leverages a Graph Transformer network to identify potential associations between graph nodes. It incorporates contrastive learning and employs a four-phase approach with diverse augmented views as positive samples, significantly enhancing prediction accuracy. The key contributions of our work are as follows:<list list-type="simple">
<list-item>
<p>(1) We develop a novel heterogeneous graph-based model that employs a Graph Transformer network to effectively capture complex interactions between microbes and drugs. This allows the model to leverage long-range dependencies within the network structure, surpassing traditional GCN-based methods.</p>
</list-item>
<list-item>
<p>(2) We introduce contrastive learning into microbe-drug association prediction, a technique previously underexplored in this domain. The model generates multiple augmented graph views through node perturbation, treating them as positive samples, while negative samples are selected from different graphs. This contrastive loss mechanism significantly enhances the model&#x2019;s ability to learn discriminative and generalizable embeddings.</p>
</list-item>
<list-item>
<p>(3) We conduct extensive experiments on three widely used public datasets (MDAD, aBiofilm and Drug Virus), demonstrating that CLMT significantly outperforms state-of-the-art prediction methods. We further validate CLMT&#x2019;s ability to uncover novel microbe-drug associations through case studies on two common drugs, reinforcing the model&#x2019;s practical value in biomedical research.</p>
</list-item>
</list>
</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<sec id="s2-1">
<title>2.1 Datasets</title>
<p>In this study, we used three publicly available datasets for model training and validation: the Microbe-Drug Association Database (MDAD), the aBiofilm database, and the Drug Virus database.</p>
<p>MDAD is a comprehensive resource specializing in known associations between microbes and drugs, integrating data from authoritative sources such as DrugBank, the Human Microbiome Project (HMP), KEGG, and PubChem. Specifically, the MDAD database includes 2,470 clinically or experimentally validated associations between 1,373 drugs and 173 microorganisms. Each association is backed by high-quality data and confirmed through rigorous experimental validation or clinical trials.</p>
<p>The aBiofilm database contains 2,884 associations between 1,720 drugs and 140 microorganisms, focusing on biofilm-associated microbial-drug interactions. It collects a substantial amount of experimental data, particularly on drug-microbe associations related to biofilm formation and inhibition.</p>
<p>The Drug Virus database provides an extensive collection of drug-virus interactions, which are critical for understanding the potential antiviral effects of drugs. This dataset integrates data from multiple biomedical resources, including DrugBank, CTD, and literature-reported associations, and contains over 3,000 drug-virus interactions covering a wide range of viral pathogens. The inclusion of this dataset allows us to assess the model&#x2019;s ability to handle a broader spectrum of drug-target interactions, particularly in the context of antiviral drug discovery and drug repurposing.</p>
<p>To ensure the reliability of the analyzed results and the biological significance of the associations, we further incorporated drug-disease and microbe/virus-disease association data. The results of the analyses of the MDAD, aBiofilm, and Drug Virus datasets are presented in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Results of MDAD and aBiofilm dataset analysis.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Data set</th>
<th align="left">Number of drugs</th>
<th align="left">Microbial population</th>
<th align="left">Number of diseases</th>
<th align="left">Number of associations</th>
<th align="left">Number of drug-disease associations</th>
<th align="left">Number of microbe-disease associations</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">MDAD</td>
<td align="left">1,373</td>
<td align="left">173</td>
<td align="left">109</td>
<td align="left">2,470</td>
<td align="left">1,121</td>
<td align="left">402</td>
</tr>
<tr>
<td align="left">aBiofilm</td>
<td align="left">1,720</td>
<td align="left">140</td>
<td align="left">72</td>
<td align="left">2,884</td>
<td align="left">435</td>
<td align="left">254</td>
</tr>
<tr>
<td align="left">Drug Virus</td>
<td align="left">1,950</td>
<td align="left">200&#x2b;</td>
<td align="left">85</td>
<td align="left">3,050&#x2b;</td>
<td align="left">720</td>
<td align="left">580</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Consistent with the methodology described by <xref ref-type="bibr" rid="B44">Tan et al. (2022)</xref>, we implemented the following data screening strategy. First, we selected diseases that were associated with at least one drug and one microorganism in the MDAD dataset. This screening step yielded 109 diseases linked to both drugs and microorganisms. From these, we further extracted 1,121 drug-disease associations and 402 microbe-disease associations.</p>
<p>Similarly, we screened the aBiofilm dataset for diseases associated with at least one drug and one microorganism. This process identified 72 diseases, from which we extracted 435 drug-disease associations and 254 microbe-disease associations.</p>
<p>For the Drug Virus dataset, we applied the same screening criteria, selecting diseases associated with at least one drug and one virus. This step identified 85 diseases, from which we extracted 720 drug-disease associations and 580 virus-disease associations. The inclusion of the Drug Virus dataset allows us to evaluate the model&#x2019;s performance on a larger and more diverse dataset, particularly in the context of antiviral drug discovery and cross-domain generalization.</p>
<p>By integrating the Drug Virus dataset into our study, we aim to assess the model&#x2019;s scalability and robustness when applied to a broader range of biomedical problems. Additionally, given the growing need for antiviral drug repurposing&#x2014;particularly in response to emerging viral diseases&#x2014;this dataset provides an important benchmark for evaluating the model&#x2019;s ability to predict drug-virus associations with potential clinical relevance.</p>
</sec>
<sec id="s2-2">
<title>2.2 Overview</title>
<p>
<xref ref-type="fig" rid="F1">Figure 1</xref> shows the detailed architecture of the Graph Contrastive Learning Model with Transformer proposed in this study for Microbe-Drug Associations Prediction (abbreviated as CLMT). The model aims to capture underlying structural relationships in microbe-drug graphs and enhance the robustness and discriminative power of the representation through contrastive learning. The CLMT model consists of four main modules: the Input Microbe-Drug Graph, the Graph Transformer Module, the Graph Contrastive Learning Module, and the Association Prediction Network.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Structural diagram of the model proposed in this study for the microbe-drug association prediction task.</p>
</caption>
<graphic xlink:href="fgene-16-1535279-g001.tif"/>
</fig>
<p>First, the model constructs a heterogeneous graph structure composed of microbes and drugs as input. This graph is then processed by the Graph Transformer Network to capture potential association relationships between the nodes (<xref ref-type="bibr" rid="B56">Yun et al., 2019</xref>). Next, the Graph Transformer encoder further refines these association relationships within the microbe-drug graph structure. The model also incorporates contrastive learning (<xref ref-type="bibr" rid="B54">You et al., 2020</xref>), generating multiple augmented views of the graph as positive samples, while negative samples are derived from different graphs. By calculating the contrastive loss between the original graph and its augmented views, the model learns a more robust representation. Finally, the Association Prediction Network formalizes this task as a binary classification problem to compute the potential association information between microbes and drugs.</p>
</sec>
<sec id="s2-3">
<title>2.3 Input microbe-drug graph</title>
<p>The Microbe-Drug Graph Representation Layer is the base module of the CLMT model. The layer receives raw microbe and drug data, transforms them into heterogeneous graph structures and computes the similarity matrices of drugs and microbes, and finally generates microbe-drug embedding representations. These embeddings representations are used as input vectors for the subsequent graph Transformer learning module.</p>
<p>The initial inputs to this layer are raw microbial data and drug data. First, based on known microbe-drug associations, we construct a heterogeneous network structure by combining drug similarity and microbe similarity networks. We define the microbe-drug neighbor matrix <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> , where <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denote the number of microbes and drugs, respectively. If the first <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> the drug <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is associated with the number of <inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> microorganism <inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> there is an association between them, then the element at the corresponding position in the adjacency matrix <inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> takes the value of 1, otherwise it takes the value of <inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>To calculate the similarity of microbial nodes as well as drug nodes, we introduce exponential similarity. Exponential similarity is a method commonly used to calculate similarity between nodes (<xref ref-type="bibr" rid="B14">Goodall, 1966</xref>). Setting <inline-formula id="inf10">
<mml:math id="m10">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf11">
<mml:math id="m11">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denote the adjacency matrix respectively <inline-formula id="inf12">
<mml:math id="m12">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> the rows of <inline-formula id="inf13">
<mml:math id="m13">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> rows and <inline-formula id="inf14">
<mml:math id="m14">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> columns, then the drugs <inline-formula id="inf15">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf16">
<mml:math id="m16">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> The exponential similarity between is calculated as follows:<disp-formula id="equ1">
<mml:math id="m17">
<mml:mrow>
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi mathvariant="italic">exp</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="italic">exp</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x2225;</mml:mo>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msup>
<mml:mo>&#x2225;</mml:mo>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ2">
<mml:math id="m18">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:msup>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
</mml:munderover>
</mml:mstyle>
<mml:mtext>&#x200a;</mml:mtext>
<mml:mrow>
<mml:mo>&#x2225;</mml:mo>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msup>
<mml:mo>&#x2225;</mml:mo>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Where <inline-formula id="inf17">
<mml:math id="m19">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the tuning parameter and takes the value of 1. <inline-formula id="inf18">
<mml:math id="m20">
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> denotes the Frobenius norm. Similarly, the exponential similarity matrix of microorganisms <inline-formula id="inf19">
<mml:math id="m21">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi mathvariant="italic">exp</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is calculated similarly:<disp-formula id="equ3">
<mml:math id="m22">
<mml:mrow>
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi mathvariant="italic">exp</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="italic">exp</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x2225;</mml:mo>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msup>
<mml:mo>&#x2225;</mml:mo>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ4">
<mml:math id="m23">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:msup>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:munderover>
</mml:mstyle>
<mml:mtext>&#x200a;</mml:mtext>
<mml:mrow>
<mml:mo>&#x2225;</mml:mo>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msup>
<mml:mo>&#x2225;</mml:mo>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Next, we use Jaccard Similarity (<xref ref-type="bibr" rid="B5">Bag S et al., 2019</xref>) to measure the similarity between nodes. Jaccard similarity measures similarity based on the ratio of intersection to concatenation. The Jaccard similarity between drug node pairs is defined as follows:<disp-formula id="equ5">
<mml:math id="m24">
<mml:mrow>
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mi>J</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mfenced open="" close="|" separators="&#x7c;">
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="" separators="&#x7c;">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2229;</mml:mo>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="" close="|" separators="&#x7c;">
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="" separators="&#x7c;">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x222a;</mml:mo>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf20">
<mml:math id="m25">
<mml:mrow>
<mml:mfenced open="|" close="|" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> denotes the number of elements in the set. Similarly, we have calculated the Jaccard similarity of microorganisms:<disp-formula id="equ6">
<mml:math id="m26">
<mml:mrow>
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mi>J</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mfenced open="" close="|" separators="&#x7c;">
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="" separators="&#x7c;">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2229;</mml:mo>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="" close="|" separators="&#x7c;">
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="" separators="&#x7c;">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x222a;</mml:mo>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Further, we combine the index similarity of drugs <inline-formula id="inf21">
<mml:math id="m27">
<mml:mrow>
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi mathvariant="italic">exp</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and Jaccard Similarity <inline-formula id="inf22">
<mml:math id="m28">
<mml:mrow>
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mi>J</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> , to get the integrated drug similarity matrix <inline-formula id="inf23">
<mml:math id="m29">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> :<disp-formula id="equ7">
<mml:math id="m30">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi mathvariant="italic">exp</mml:mi>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mi>J</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Similarly, the integrated microbial similarity matrix <inline-formula id="inf24">
<mml:math id="m31">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is calculated as follows:<disp-formula id="equ8">
<mml:math id="m32">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi mathvariant="italic">exp</mml:mi>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mi>J</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Ultimately, we construct graph networks based on these integrated similarity and adjacency matrices:<disp-formula id="equ9">
<mml:math id="m33">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
</mml:mtd>
<mml:mtd>
<mml:mi>A</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msup>
<mml:mi>A</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
</mml:mtd>
<mml:mtd>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>The graph network constructed in this way <inline-formula id="inf25">
<mml:math id="m34">
<mml:mrow>
<mml:mi mathvariant="normal">N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> that not only retains the similarity information of drugs and microorganisms, but also incorporates the interactions between them, providing a rich feature representation for subsequent graph neural network models.</p>
</sec>
<sec id="s2-4">
<title>2.4 Graph transformer module</title>
<p>In this study, the Graph Transformer module is the core component, which is designed to capture potential microbe-drug association features by learning the deep representation of nodes in the microbe-drug graph structure.</p>
<p>The graph attention mechanism is the key mechanism of the Graph Transformer module, which allows nodes to dynamically adjust their own representations in microbial-drug networks by taking into account the information of neighboring nodes (<xref ref-type="bibr" rid="B48">Wang X et al., 2019</xref>). Specifically, in the first <inline-formula id="inf26">
<mml:math id="m35">
<mml:mrow>
<mml:mi mathvariant="normal">l</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> layer, the feature representation of each node is <inline-formula id="inf27">
<mml:math id="m36">
<mml:mrow>
<mml:msup>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> , where <inline-formula id="inf28">
<mml:math id="m37">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the total number of nodes, i.e., the sum of the number of microbes and drugs, and <inline-formula id="inf29">
<mml:math id="m38">
<mml:mrow>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the number of nodes in the first <inline-formula id="inf30">
<mml:math id="m39">
<mml:mrow>
<mml:mi mathvariant="normal">l</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> number of hidden units in the layer. And the node features can be obtained by linear transformation:<disp-formula id="equ10">
<mml:math id="m40">
<mml:mrow>
<mml:msup>
<mml:mi>Z</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf31">
<mml:math id="m41">
<mml:mrow>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mrow>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the learnable weight matrix. Multihead attention allows the model to learn multiple sets of different attention weights in parallel to capture the relationships of different feature subspaces in a heterogeneous network. For the first <inline-formula id="inf32">
<mml:math id="m42">
<mml:mrow>
<mml:mi mathvariant="normal">l</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> layer of multi-head attention, node <inline-formula id="inf33">
<mml:math id="m43">
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and its neighbor nodes <inline-formula id="inf34">
<mml:math id="m44">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> The attention weights between the node and its neighbor nodes <inline-formula id="inf35">
<mml:math id="m45">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> can be computed in the following way:<disp-formula id="equ11">
<mml:math id="m46">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="italic">exp</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>k</mml:mi>
<mml:mi>y</mml:mi>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msup>
<mml:mi>a</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msup>
</mml:msup>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mi>Z</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2225;</mml:mo>
<mml:msubsup>
<mml:mi>Z</mml:mi>
<mml:mi>j</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>j</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mrow>
<mml:mtext>&#x200a;</mml:mtext>
<mml:mi mathvariant="italic">exp</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>k</mml:mi>
<mml:mi>y</mml:mi>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msup>
<mml:mi>a</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msup>
</mml:msup>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mi>Z</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2225;</mml:mo>
<mml:msubsup>
<mml:mi>Z</mml:mi>
<mml:msup>
<mml:mi>j</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf36">
<mml:math id="m47">
<mml:mrow>
<mml:msup>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the learned attention parameter, and <inline-formula id="inf37">
<mml:math id="m48">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>k</mml:mi>
<mml:mi>y</mml:mi>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the activation function, and <inline-formula id="inf38">
<mml:math id="m49">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the node <inline-formula id="inf39">
<mml:math id="m50">
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> the set of neighbors of the node, and <inline-formula id="inf40">
<mml:math id="m51">
<mml:mrow>
<mml:mo>&#x2225;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> denotes the vector splicing operation. Further, the Graph Transformer module introduces a multi-head attention mechanism, where the outputs of multiple attention heads undergo a splicing operation to obtain the final node representation:<disp-formula id="equ12">
<mml:math id="m52">
<mml:mrow>
<mml:msubsup>
<mml:mi>H</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mi>H</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>H</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf41">
<mml:math id="m53">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the number of attention heads. In addition, in order to promote the stability of information transfer and model training, the output of each layer of Graph Transformer is subjected to residual concatenation and layer normalization, a process that can be formally described as:<disp-formula id="equ13">
<mml:math id="m54">
<mml:mrow>
<mml:msubsup>
<mml:mi>H</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>L</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>y</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mi>H</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mi>H</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf42">
<mml:math id="m55">
<mml:mrow>
<mml:mtext>LayerNorm</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the layer normalization operation.</p>
<p>After iterative updating by the multi-layer graph attention mechanism, the feature representation of each node in the final layer of the microbe-drug network <inline-formula id="inf43">
<mml:math id="m56">
<mml:mrow>
<mml:msup>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> contains rich feature information of drugs and microbes, which can effectively capture the potential interaction patterns and association laws between them.</p>
</sec>
<sec id="s2-5">
<title>2.5 Graph contrastive learning module</title>
<p>The Graph Contrastive Learning module is designed to enhance the model&#x2019;s ability to extract informative and discriminative representations of nodes in the heterogeneous microbe-drug interaction network. By leveraging contrastive learning, our model learns to maximize the agreement between positive samples (different augmented views of the same node) while minimizing the similarity with negative samples (nodes from different distributions).</p>
<p>In order to enhance the model&#x2019;s understanding of the structure of the microbe-drug graph and to improve the generalization ability of the overall model, we employed graph data augmentation techniques to generate multiple augmented views of the original graph, thereby enriching the data sample space for model training. Specifically, we used the node perturbation method to generate augmented graphs (<xref ref-type="bibr" rid="B16">Hiratani N et al., 2022</xref>). For each node in the microbe-drug graph structure, we randomly perturbed its feature vector to simulate the variation and uncertainty of node features. Let the node in the original graph <inline-formula id="inf44">
<mml:math id="m57">
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of the original graph be represented by the features of <inline-formula id="inf45">
<mml:math id="m58">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">h</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> , the feature representation of the node after node perturbation is <inline-formula id="inf46">
<mml:math id="m59">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi mathvariant="normal">h</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> , the node perturbation process can be formalized as:<disp-formula id="equ14">
<mml:math id="m60">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>h</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3f5;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>&#x3f5;</mml:mi>
<mml:mo>&#x223c;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf47">
<mml:math id="m61">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3f5;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is a random perturbation added to the node features, usually obeying some predefined distribution such as Gaussian or uniform. In this way, with the node perturbations, we can generate multiple augmented views with a slightly different structure from the original graph.</p>
<p>The purpose of the node perturbation operation is to introduce enough randomness to increase the diversity of the data and thus help the model learn a representation that is robust to noise and variation in the input data. In the graph contrast learning framework, these augmented views are used as the basis for the generation of positive sample pairs for optimizing the contrast learning process of the model. For the multiple augmented views generated, further inputs are provided to learn the deep feature representation of the nodes in Graph Transformer. For the output of Graph Transformer, the high-dimensional node representations are mapped to a low-dimensional space suitable for comparative learning through Feature Transformation. The goal of Feature Transformation is to reduce the dimensionality of the representations and to enhance their expressive power, typically using a fully connected layer, a process that can be formalized as:<disp-formula id="equ15">
<mml:math id="m62">
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mi>H</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf48">
<mml:math id="m63">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the first <inline-formula id="inf49">
<mml:math id="m64">
<mml:mrow>
<mml:mi mathvariant="normal">L</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> node representation of the layer, and <inline-formula id="inf50">
<mml:math id="m65">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">z</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the node representation after projection.</p>
<p>After generating augmented views, we apply a contrastive loss function to maximize agreement between the original and augmented representations while ensuring separation from negative samples. Specifically, for each node in the microbe-drug graph structure <inline-formula id="inf51">
<mml:math id="m66">
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> , we define its contrast loss as:<disp-formula id="equ16">
<mml:math id="m67">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="italic">log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mo>&#x2061;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="italic">exp</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="italic">sin</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mtext>&#x200a;</mml:mtext>
<mml:msub>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2260;</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi mathvariant="italic">exp</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="italic">sin</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ17">
<mml:math id="m68">
<mml:mrow>
<mml:mi mathvariant="italic">sin</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2225;</mml:mo>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>&#x2225;</mml:mo>
<mml:mo>&#x2225;</mml:mo>
</mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2225;</mml:mo>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf52">
<mml:math id="m69">
<mml:mrow>
<mml:mi mathvariant="italic">sin</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> are the nodes <inline-formula id="inf53">
<mml:math id="m70">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">z</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf54">
<mml:math id="m71">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">z</mml:mi>
<mml:mi mathvariant="normal">j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> the similarity between the nodes, and <inline-formula id="inf55">
<mml:math id="m72">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3c4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the temperature parameter, which controls the sharpness of the similarity distribution. <inline-formula id="inf56">
<mml:math id="m73">
<mml:mrow>
<mml:msub>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2260;</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the indicator function that ensures the sum normalization of pairs of samples other than positive samples. One critical aspect of contrastive learning is the selection of negative samples, as poorly chosen negatives can lead to suboptimal representations. We employ semi-hard negative mining, where negative samples are selected based on their similarity scores. Nodes with extremely low similarity are ignored, as they contribute little useful information. Nodes with moderate similarity are prioritized, as they force the model to learn more discriminative features.</p>
<p>The loss function optimizes the embeddings such that positive pairs (nodes representing the same entity in different augmentations) are pulled closer together, while negative pairs (nodes from different distributions) are pushed apart.</p>
<p>The graph contrast learning module combines graph data enhancement and unsupervised contrast learning ideas to effectively optimize the representation learning process of microbial-drug graphs, and experimental results show that the module can improve the model&#x2019;s prediction accuracy and generalization ability of microbial-drug associations.</p>
</sec>
<sec id="s2-6">
<title>2.6 Association prediction network</title>
<p>In the association prediction layer, we will utilize the microbial and drug graph structure representations obtained from the prelude steps for association prediction. Since the output of the model is still the node representations learned by Graph Transformer and Graph Comparison Learning Module, we first map these high-dimensional node representations to the final association prediction results. Specifically, we reduce the set of nodes by a linear transformation <inline-formula id="inf57">
<mml:math id="m74">
<mml:mrow>
<mml:mi mathvariant="normal">Z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> for dimensionality reduction, and the linear transformation can be expressed as:<disp-formula id="equ18">
<mml:math id="m75">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>Z</mml:mi>
<mml:mi>W</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf58">
<mml:math id="m76">
<mml:mrow>
<mml:mi mathvariant="normal">H</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the node representation matrix after linear transformation. Then, the sigmoid activation function is utilized to map the linearly transformed node representations into the <inline-formula id="inf59">
<mml:math id="m77">
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> the probability space for predicting the association probability between microorganisms and drugs:<disp-formula id="equ19">
<mml:math id="m78">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>Y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Where. <inline-formula id="inf60">
<mml:math id="m79">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>Y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mn>1</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> denotes the probability of potential association between microorganism and drug.</p>
</sec>
<sec id="s2-7">
<title>2.7 Loss function</title>
<p>At this point, we complete the inference process of the CLMT model, and the pseudo-code corresponding to this process is shown in <xref ref-type="fig" rid="F2">Figure 2</xref>. In order to measure the difference between the predicted and true values of the model, we use the cross-entropy loss function to evaluate the effect of microbe-drug association prediction (<xref ref-type="bibr" rid="B32">Mao A et al., 2023</xref>). The cross-entropy loss function is a commonly used loss function in classification problems, and in microbe-drug association prediction, we modeled the problem as a binary classification task, i.e., predicting whether a certain pair of microbes and drugs are associated. The cross-entropy loss function is defined as follows:<disp-formula id="equ20">
<mml:math id="m80">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi mathvariant="italic">log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi mathvariant="italic">log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf61">
<mml:math id="m81">
<mml:mrow>
<mml:mi mathvariant="normal">N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the sample size; <inline-formula id="inf62">
<mml:math id="m82">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">y</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the first <inline-formula id="inf63">
<mml:math id="m83">
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> true label of the first sample, which indicates the presence of association, and 0 indicates the absence of association. <inline-formula id="inf64">
<mml:math id="m84">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the model&#x2019;s predicted probability for the <inline-formula id="inf65">
<mml:math id="m85">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> sample, indicating the probability of an association between the microbe and the disease. The cross-entropy loss function improves the accuracy of the prediction by penalizing the wrong prediction of the model so that the model continuously adjusts the parameters during the training process.</p>
<p>To prevent model overfitting, we add a regularization term to the loss function. The regularization term improves the generalization ability of the model by adding a penalty to the model complexity in the loss function, encouraging the model to choose simpler parameter configurations (<xref ref-type="bibr" rid="B22">Kuka&#x10d;ka J et al., 2017</xref>). In the CLMT model, we use L2 regularization, i.e., weight decay. It is defined as follows:<disp-formula id="equ21">
<mml:math id="m86">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mrow>
<mml:mtext>reg</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>k</mml:mi>
</mml:munder>
</mml:mstyle>
<mml:msubsup>
<mml:mrow>
<mml:mo>&#x2225;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>&#x2225;</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Where <inline-formula id="inf66">
<mml:math id="m87">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the number, which controls the weight of the regularization term; <inline-formula id="inf67">
<mml:math id="m88">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the model&#x2019;s first <inline-formula id="inf68">
<mml:math id="m89">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> weight matrix; <inline-formula id="inf69">
<mml:math id="m90">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mo>&#x2225;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>&#x2225;</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the number of <inline-formula id="inf70">
<mml:math id="m91">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> the L2 paradigm of the sum of squares of the weight matrices. The regularization term prevents the model from overfitting the training data by penalizing excessively large values of the weights, thus improving the model&#x2019;s performance on the test data.</p>
<p>Ultimately, the combined loss function of the CLMT model consists of an unsupervised graph-contrast learning loss, a cross-entropy loss, and a regularization term of the following form:<disp-formula id="equ22">
<mml:math id="m92">
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mrow>
<mml:mtext>reg</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>This comprehensive loss function optimizes the node representation in the microbe-drug graph structure on the one hand, and takes into account the accuracy of the model prediction and the complexity of the model to ensure that the model not only can accurately fit the training data during the training process, but also has good generalization ability.</p>
</sec>
</sec>
<sec id="s3">
<title>3 Experiments and results</title>
<p>This section provides a comprehensive description of the experimental setup, evaluation metrics, and baseline methods used to assess the performance of the CLMT model. We also present the results of the experiments along with a detailed analysis. The effectiveness and superiority of the CLMT model in predicting microbe-disease associations are demonstrated through comparisons with several baseline methods. Additionally, <xref ref-type="fig" rid="F2">Figure 2</xref> presents the corresponding pseudo-code of the CLMT model.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Pseudocode of the CLMT model proposed in this study.</p>
</caption>
<graphic xlink:href="fgene-16-1535279-g002.tif"/>
</fig>
<sec id="s3-1">
<title>3.1 Experimental setup</title>
<p>In this study, we extracted drug features, microbial characteristics, and microbe-drug association matrices from the MDAD and aBiofilm databases. These feature matrices were subsequently used to construct heterogeneous networks that represent the interactions between drugs and microbes.</p>
<p>For the CLMT model, we set the number of epochs to 1,000 and the learning rate for the optimization algorithm to 0.001. The Graph Transformer model was configured with 3 layers, while the Multihead Self-Attention module contained 6 heads. Specifically, we tested different configurations of the Graph Transformer layer (ranging from 1 to 5 layers) and found that using 3 layers achieved the best balance between model complexity and performance. A smaller number of layers (e.g., 1 or 2) led to insufficient representation learning, while a larger snumber of layers (e.g., 4 or 5) caused overfitting and increased computational costs without significant performance improvements. For the multi-head attention mechanism, we experimented with different head numbers (ranging from 2 to 8). We found that 6 heads provided the most effective feature aggregation, allowing the model to capture diverse interaction patterns between microbes and drugs. Using fewer heads (e.g., 2 or 4) limited the model&#x2019;s ability to focus on multiple aspects of the relationships, while using more heads (e.g., 8) led to increased computational overhead without notable gains in predictive accuracy. In the contrastive learning module, we set the temperature parameter <inline-formula id="inf71">
<mml:math id="m93">
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> in the contrastive loss function to 0.5, following extensive empirical analysis. The temperature parameter controls the sharpness of the similarity distribution, affecting how the model distinguishes positive and negative pairs. To determine the optimal value of, we tested values in the range [0.1,1.0] with a step size of 0.1. We observed that smaller values (<inline-formula id="inf72">
<mml:math id="m94">
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3c;0.3) led to over-concentration of representations, where the model assigned overly confident similarity scores, reducing the discriminative ability of learned embeddings. Larger values (<inline-formula id="inf73">
<mml:math id="m95">
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3e;0.7) resulted in overly smooth embeddings, making it harder for the model to effectively separate positive and negative pairs. Setting <inline-formula id="inf74">
<mml:math id="m96">
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 0.5 achieved the best balance between representation compactness and separability, ensuring that positive pairs remained close while maintaining sufficient distinction from negative pairs.</p>
<p>To improve model generalization, we incorporated a stochastic deactivation strategy in the association prediction module of the linear layer, with a dropout rate of 50%.</p>
<p>The model was trained using the Adam optimizer with a weight decay prevent overfitting. We applied an early stopping criterion with a patience of 20 epochs, monitoring the validation loss to avoid unnecessary training cycles. During both training and evaluation, we performed multiple rounds of cross-validation. Specifically, 5-fold cross-validation was applied, where the dataset was randomly split into five subsets. In each fold, one subset was used as the test set, and the remaining four were used for training. To ensure the reliability and robustness of the results, the entire experiment was repeated five times, and the average performance metrics were reported. All experiments were conducted on a NVIDIA 2080Ti GPU (11GB VRAM). The GPU acceleration significantly improved the efficiency of graph-based operations, particularly in the Graph Transformer module and contrastive learning calculations.</p>
</sec>
<sec id="s3-2">
<title>3.2 Evaluation indicators</title>
<p>In order to evaluate the methodology proposed in this paper, we employ a series of evaluation metrics to comprehensively measure the performance of the model, including AUC, AUPR and Accuracy. The following are the formal definitions and calculations of each evaluation metric:</p>
<p>AUC (Area Under the ROC Curve) represents the area under the receiver operating characteristic curve (ROC Curve), which is used to measure the classification performance of the model. The ROC Curve plots the True Positive Rate (TPR) and False Positive Rate (FPR) through different thresholds. TPR and FPR are defined as follows:<disp-formula id="equ23">
<mml:math id="m97">
<mml:mrow>
<mml:mtext>TPR</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mtext>Recall</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ24">
<mml:math id="m98">
<mml:mrow>
<mml:mtext>FPR</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf75">
<mml:math id="m99">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes true positives (True Positives) and <inline-formula id="inf76">
<mml:math id="m100">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes False Positives, and <inline-formula id="inf77">
<mml:math id="m101">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes False Negatives, and <inline-formula id="inf78">
<mml:math id="m102">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes True Negatives. AUC is a threshold-independent metric, meaning it evaluates model performance across all possible decision thresholds rather than a single threshold. It measures the model&#x2019;s discrimination ability&#x2014;the probability that a randomly chosen positive sample is ranked higher than a randomly chosen negative sample. In tasks like microbe-drug association prediction, where both false positives (misidentifying non-associations as associations) and false negatives (failing to identify true associations) are critical, AUC provides a balanced view of the model&#x2019;s classification performance.</p>
<p>AUPR (Area Under the Precision-Recall Curve) denotes the area under the Precision-Recall Curve, which is used to measure the classification performance of the model on unbalanced datasets. The Precision-Recall Curve plots Precision and Recall through different thresholds. Precision and Recall are defined as follows:<disp-formula id="equ25">
<mml:math id="m103">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ26">
<mml:math id="m104">
<mml:mrow>
<mml:mtext>Recall</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf79">
<mml:math id="m105">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes true positives (True Positives) and <inline-formula id="inf80">
<mml:math id="m106">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes False Positives, and <inline-formula id="inf81">
<mml:math id="m107">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes False Negatives. Precision reflects the proportion of samples predicted to be positive by the model that are actually positive, while recall reflects the proportion of samples that are actually positive that are correctly predicted to be positive. AUPR has a value between 0 and 1, with larger values indicating better model performance. Since microbe-drug association datasets often contain significantly more negative samples than positive ones, AUC may overestimate model performance by giving equal weight to both classes. AUPR, on the other hand, focuses on the positive class and better reflects the model&#x2019;s ability to identify meaningful associations.</p>
<p>In addition to AUC and AUPR, we also report Accuracy as a standard evaluation metric to measure the overall correctness of the model&#x2019;s predictions. Accuracy is defined as:<disp-formula id="equ27">
<mml:math id="m108">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Accuracy provides a simple and intuitive measure of the model&#x2019;s classification ability. It is useful when the dataset is relatively balanced, as it evaluates both positive and negative class predictions equally.</p>
</sec>
<sec id="s3-3">
<title>3.3 Methods of comparison</title>
<p>To evaluate the performance of our proposed method, we compared it with five existing microbe-drug association prediction approaches. A brief overview of each method and their limitations is provided below:</p>
<p>HMDAKATZ (<xref ref-type="bibr" rid="B58">Zhu et al., 2019</xref>): This method predicts microbe-drug associations using the KATZ metric. However, it primarily relies on traditional graph metrics, which are unable to capture complex long-range dependencies. It also fails to account for important biological features of microbes and drugs, such as drug side effects, which limits its applicability to novel drugs or microbes with unknown associations.</p>
<p>GCNMDA (<xref ref-type="bibr" rid="B26">Long et al., 2020b</xref>): This method is based on Graph Convolutional Networks (GCNs) and conditional random fields to predict associations between microbes and drugs. While GCNs can capture local interactions, they struggle to model complex heterogeneous network structures and long-range dependencies, which affects their performance in handling noisy data and unknown associations.</p>
<p>GSAMDA (<xref ref-type="bibr" rid="B44">Tan et al., 2022</xref>): GSAMDA uses graph attention networks and sparse autoencoders to model both topological and attribute features within a microbe-drug heterogeneous network. However, its performance is limited by data sparsity, especially when there is insufficient labeled data, and it does not adequately model the intricate biological interactions between microbes and drugs.</p>
<p>LAGCN (<xref ref-type="bibr" rid="B55">Yu et al., 2021</xref>): LAGCN applies graph convolution to learn drug and disease embeddings, using an attentional mechanism to integrate embeddings from multiple layers for drug-disease association prediction. However, it is optimized for drug-disease predictions and does not specifically target microbe-drug associations, limiting its effectiveness for the task at hand.</p>
<p>NTSHMDA (<xref ref-type="bibr" rid="B29">Luo and Long, 2018</xref>): This method uses an improved randomized roaming algorithm to infer microbe-disease associations by integrating topological similarities within a microbe-drug network. However, it overlooks important biological features such as microbial genome information and drug side effects, which reduces its predictive power, especially for microbe-drug interactions.</p>
<p>These methods were evaluated on the MDAD, aBiofilm and Drug Virus datasets, using their default configurations and tuning their hyperparameters. All methods underwent 5-fold cross-validation, with known microbe-drug associations serving as positive samples and randomly generated negative samples for the training and test sets. To minimize sampling bias, each comparison was repeated five times, and the final AUC score was reported as the average of these iterations.</p>
<p>In contrast to these methods, our CLMT model introduces several innovations:<list list-type="simple">
<list-item>
<p>1. Graph Transformer Network: CLMT uses a Graph Transformer network to capture complex, long-range dependencies within the microbe-drug interaction network, surpassing the limitations of GCN-based approaches.</p>
</list-item>
<list-item>
<p>2. Contrastive Learning: By leveraging contrastive learning and generating multiple augmented views of the graph, CLMT significantly improves the model&#x2019;s ability to learn discriminative and generalizable embeddings, even with sparse data.</p>
</list-item>
<list-item>
<p>3. Prediction of Novel Interactions: CLMT excels at predicting not only known associations but also novel microbe-drug interactions, making it more versatile and applicable in real-world scenarios where data may be limited or incomplete.</p>
</list-item>
</list>
</p>
<p>Our extensive experiments on the MDAD, aBiofilm and Drug Virus datasets demonstrate that CLMT outperforms these existing methods, offering superior predictive accuracy and uncovering novel microbe-drug associations with greater reliability.</p>
</sec>
<sec id="s3-4">
<title>3.4 Experimental results and analysis</title>
<p>
<xref ref-type="table" rid="T2">Tables 2</xref>, <xref ref-type="table" rid="T3">3</xref> present the AUC, AUPR, and Accuracy scores of the CLMT model proposed in this paper, along with those of the compared methods on the MDAD and aBiofilm datasets. As shown in the tables, the CLMT method achieved the highest AUC (0.9735 and 0.9742), AUPR (0.9720 and 0.9714), and Accuracy (0.9045 and 0.9121) scores on both datasets, significantly outperforming the other five methods.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>5-fold cv results on MDAD dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Model</th>
<th align="left">AUC</th>
<th align="left">AUPR</th>
<th align="left">Accuracy</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">HMDAKATZ</td>
<td align="left">0.8712 &#xb1; 0.0010</td>
<td align="left">0.8798 &#xb1; 0.0068</td>
<td align="left">0.7691 &#xb1; 0.0167</td>
</tr>
<tr>
<td align="left">GCNMDA</td>
<td align="left">0.9365 &#xb1; 0.0001</td>
<td align="left">0.9300 &#xb1; 0.0002</td>
<td align="left">0.8617 &#xb1; 0.0011</td>
</tr>
<tr>
<td align="left">GSAMDA</td>
<td align="left">0.9460 &#xb1; 0.0197</td>
<td align="left">0.9223 &#xb1; 0.0164</td>
<td align="left">0.7979 &#xb1; 0.0279</td>
</tr>
<tr>
<td align="left">LAGCN</td>
<td align="left">0.8974 &#xb1; 0.0056</td>
<td align="left">0.9062 &#xb1; 0.0050</td>
<td align="left">0.8572 &#xb1; 0.0067</td>
</tr>
<tr>
<td align="left">NTSHMDA</td>
<td align="left">0.8512 &#xb1; 0.0043</td>
<td align="left">0.8094 &#xb1; 0.0055</td>
<td align="left">0.7820 &#xb1; 0.0137</td>
</tr>
<tr>
<td align="left">CLMT</td>
<td align="left">0.9735 &#xb1; 0.0014</td>
<td align="left">0.9720 &#xb1; 0.0025</td>
<td align="left">0.9045 &#xb1; 0.0031</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>5-fold cv results on aBiofilm dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Model</th>
<th align="center">AUC</th>
<th align="center">AUPR</th>
<th align="center">Accuracy</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">HMDAKATZ</td>
<td align="left">0.8982 &#xb1; 0.0042</td>
<td align="left">0.9018 &#xb1; 0.0037</td>
<td align="left">0.7811 &#xb1; 0.0083</td>
</tr>
<tr>
<td align="left">GCNMDA</td>
<td align="left">0.9465 &#xb1; 0.0073</td>
<td align="left">0.9376 &#xb1; 0.0026</td>
<td align="left">0.8772 &#xb1; 0.0012</td>
</tr>
<tr>
<td align="left">GSAMDA</td>
<td align="left">0.8955 &#xb1; 0.0020</td>
<td align="left">0.9073 &#xb1; 0.0033</td>
<td align="left">0.8345 &#xb1; 0.0001</td>
</tr>
<tr>
<td align="left">LAGCN</td>
<td align="left">0.8991 &#xb1; 0.0047</td>
<td align="left">0.9084 &#xb1; 0.0028</td>
<td align="left">0.8710 &#xb1; 0.0011</td>
</tr>
<tr>
<td align="left">NTSHMDA</td>
<td align="left">0.8633 &#xb1; 0.0065</td>
<td align="left">0.8204 &#xb1; 0.0045</td>
<td align="left">0.8073 &#xb1; 0.0038</td>
</tr>
<tr>
<td align="left">CLMT</td>
<td align="left">0.9742 &#xb1; 0.0024</td>
<td align="left">0.9714 &#xb1; 0.0011</td>
<td align="left">0.9121 &#xb1; 0.0005</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>First, several comparative methods have demonstrated effectiveness in microbe-disease association tasks. For instance, the GSAMDA model, which utilizes graph attention networks and sparse autoencoders, achieved AUC scores of 0.9460 and 0.8955, and AUPR scores of 0.9223 and 0.9073 on the MDAD and aBiofilm datasets, respectively. These results indicate that GSAMDA effectively captures the topological and attribute features of nodes in the newly constructed microbial-drug heterogeneous network. Specifically, when dealing with graph data involving complex relationships, the graph attention network (GAT) can effectively focus on important node features through the attention mechanism, while the sparse autoencoder (SAE) can capture the data&#x2019;s sparse structure. These characteristics enable GSAMDA to perform well in this task, demonstrating the feasibility of using graph neural networks and autoencoders for microbe-drug association prediction.</p>
<p>However, despite the satisfactory performance of many methods on this task, their AUC, AUPR, and Accuracy metrics still have room for improvement. Taking the GSAMDA model as an example, its AUC on the aBiofilm dataset is 0.8955, and its AUPR is 0.9073, which represents a gap of 5.05% and 1.5%, respectively, compared to its performance on the MDAD dataset. This gap suggests that GSAMDA has limitations, particularly when handling different datasets, indicating its potential shortcomings in capturing features and modeling relationships. Therefore, the microbe-drug association prediction task requires further exploration, and more powerful and robust methods are necessary to enhance prediction performance.</p>
<p>In comparison, the CLMT method proposed in this paper significantly outperforms all other methods on both datasets. The three evaluation metrics on the MDAD dataset are 0.9735, 0.9720, and 0.9045, respectively, while on the aBiofilm dataset, the corresponding metrics are 0.9742, 0.9714, and 0.9121. We attribute this superior performance to the unique model structure and design principles of CLMT.</p>
<p>To further verify whether the experimental results of CLMT are statistically significant, we conducted a statistical analysis on the AUC scores from 5-fold cross-validation and compared them with a baseline method. Since some existing methods do not have publicly available implementations, we reproduced GSAMDA, one of the best-performing models on the MDAD dataset, as a comparison model and computed the p-value to assess whether CLMT provides a statistically significant improvement. On the MDAD test set, the AUC scores from 5-fold cross-validation for GSAMDA were [0.9497, 0.9277, 0.9389, 0.9539, 0.9581], while our proposed CLMT achieved [0.9735, 0.9730, 0.9737, 0.9748, 0.9729] under the same conditions. To quantify whether the performance gain of CLMT over GSAMDA is statistically significant, we applied a paired t-test, obtaining a p-value of 0.0010 (p &#x3c; 0.05). This result confirms that the improvement of CLMT over GSAMDA is not due to random variations but represents a statistically significant performance enhancement driven by the methodological improvements introduced in CLMT.</p>
<p>CLMT employs data enhancement techniques such as node perturbation, which enriches the training data by generating a multi-view graph structure. This technique helps the model better learn the diversity of nodes and edges within the graph, thereby improving its generalization ability. More importantly, in the graph contrastive learning module, CLMT utilizes a projection head to map the node representations output by the graph encoder to a space suitable for contrastive learning. By calculating the contrastive loss, this mechanism maximizes the consistency between different views of the same graph structure and minimizes the similarity between different graph structures. This contrastive learning mechanism effectively enhances the model&#x2019;s ability to capture graph structure features, enabling it to make more accurate association predictions when faced with different graph structures.</p>
<p>Additionally, CLMT incorporates a Transformer model based on a multi-head self-attention mechanism within the graph encoder. This approach enhances the model&#x2019;s representational capacity by capturing various relationships and feature interactions between nodes through multiple attention heads. The multi-head self-attention mechanism not only focuses on globally important features but also mitigates the overfitting problem that can arise from relying on a single attention head.</p>
<p>
<xref ref-type="table" rid="T4">Table 4</xref> presents the AUC, AUPR, and Accuracy scores of the CLMT model proposed in this paper, along with those of the compared methods on the Drug Virus dataset. As shown in the table, the CLMT method achieved the highest AUC (0.9727), AUPR (0.9699), and Accuracy (0.9235), significantly outperforming the other five methods.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>5-fold cv results on Drug Virus dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Model</th>
<th align="center">AUC</th>
<th align="center">AUPR</th>
<th align="center">Accuracy</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">HMDAKATZ</td>
<td align="left">0.8523 &#xb1; 0.0074</td>
<td align="left">0.8617 &#xb1; 0.0045</td>
<td align="left">0.7245 &#xb1; 0.0074</td>
</tr>
<tr>
<td align="left">GCNMDA</td>
<td align="left">0.9214 &#xb1; 0.0052</td>
<td align="left">0.8965 &#xb1; 0.0037</td>
<td align="left">0.8674 &#xb1; 0.0069</td>
</tr>
<tr>
<td align="left">GSAMDA</td>
<td align="left">0.8754 &#xb1; 0.0024</td>
<td align="left">0.8868 &#xb1; 0.0064</td>
<td align="left">0.8745 &#xb1; 0.0068</td>
</tr>
<tr>
<td align="left">LAGCN</td>
<td align="left">0.9214 &#xb1; 0.0036</td>
<td align="left">0.9247 &#xb1; 0.0029</td>
<td align="left">0.8958 &#xb1; 0.0036</td>
</tr>
<tr>
<td align="left">NTSHMDA</td>
<td align="left">0.8354 &#xb1; 0.0085</td>
<td align="left">0.8004 &#xb1; 0.0074</td>
<td align="left">0.7954 &#xb1; 0.0023</td>
</tr>
<tr>
<td align="left">CLMT</td>
<td align="left">0.9727 &#xb1; 0.0012</td>
<td align="left">0.9699 &#xb1; 0.0014</td>
<td align="left">0.9235 &#xb1; 0.0007</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Despite the reasonable performance of many existing methods, their AUC, AUPR, and Accuracy metrics still have room for improvement. For example, the GSAMDA model, which utilizes graph attention networks and sparse autoencoders, achieved an AUC of 0.8754 and an AUPR of 0.8868 on the Drug Virus dataset. While GSAMDA successfully captures node attributes and sparse structures, its performance lags behind that of CLMT, highlighting potential limitations in generalizing to diverse datasets. Similarly, the HMDAKATZ model, based on heterogeneous graph diffusion, showed the lowest performance, with an AUC of 0.8523 and an Accuracy of 0.7245, indicating its struggles in capturing complex relationships in Virus-drug interactions.</p>
<p>In comparison, the CLMT method proposed in this paper significantly outperforms all other methods across all evaluation metrics. The three evaluation metrics on the Drug Virus dataset are 0.9727, 0.9699, and 0.9235, respectively. We attribute this superior performance to the unique model structure and design principles of CLMT.</p>
<p>Overall, the results on the Drug Virus dataset further validate the effectiveness of CLMT in microbial-drug association prediction. By leveraging contrastive learning, self-attention mechanisms, and data augmentation techniques, CLMT demonstrates superior adaptability and generalization capabilities, setting a new benchmark for future research in this domain.</p>
</sec>
<sec id="s3-5">
<title>3.5 Ablation experiment</title>
<p>To further validate the effectiveness of the individual modules in our proposed CLMT method, we conducted ablation experiments on the MDAD, aBiofilm, and DrugVirus datasets. The results are shown in <xref ref-type="table" rid="T5">Table 5</xref> and <xref ref-type="fig" rid="F3">Figure 3</xref>.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Performance comparison of CLMT, CLMT-CL, CLMT-Transformer, CLMT-node perturbation on MDAD, aBiofilm, and DrugVirus datasets.</p>
</caption>
<graphic xlink:href="fgene-16-1535279-g003.tif"/>
</fig>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Results of ablation experiments on MDAD, aBiofilm, and DrugVirus datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left">Model</th>
<th colspan="3" align="left">AUC</th>
</tr>
<tr>
<th align="left">MDAD</th>
<th align="left">aBiofilm</th>
<th align="left">Drug virus</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">CLMT</td>
<td align="left">0.9735 &#xb1; 0.0014</td>
<td align="left">0.9742 &#xb1; 0.0024</td>
<td align="left">0.9727 &#xb1; 0.0012</td>
</tr>
<tr>
<td align="left">-CL</td>
<td align="left">0.9629 &#xb1; 0.0018</td>
<td align="left">0.9576 &#xb1; 0.0015</td>
<td align="left">0.9514 &#xb1; 0.0025</td>
</tr>
<tr>
<td align="left">-Transformer</td>
<td align="left">0.9481 &#xb1; 0.0017</td>
<td align="left">0.9482 &#xb1; 0.0016</td>
<td align="left">0.9493 &#xb1; 0.0014</td>
</tr>
<tr>
<td align="left">-node perturbation</td>
<td align="left">0.9727 &#xb1; 0.0011</td>
<td align="left">0.9719 &#xb1; 0.0014</td>
<td align="left">0.9701 &#xb1; 0.0003</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>When the Graph Contrastive Learning Module was removed, the model exhibited consistent performance degradation across all datasets. Specifically, the AUC decreased from 0.9735 to 0.9629 on MDAD, 0.9742 to 0.9576 on aBiofilm, and 0.9727 to 0.9514 on DrugVirus. These results highlight the critical role of contrastive learning in enhancing the model&#x2019;s discriminative ability by maximizing consistency between augmented graph views. The significant performance drop (average 1.9% across datasets) underscores its contribution to generalization. Additionally, to qualitatively analyze the effectiveness of the contrastive learning module, we present the embedding distribution of the MDAD dataset&#x2019;s test data. Specifically, we obtained the high-dimensional embeddings of the test data both &#x201c;before&#x201d; and &#x201c;after&#x201d; contrastive learning, performed clustering, and visualized the results using t-SNE, as shown in <xref ref-type="fig" rid="F4">Figure 4</xref>. The left image shows the embeddings before contrastive learning, where clusters are present but may overlap due to large variance. The right image shows the embeddings after contrastive learning, where the clusters are more compact and distinctly separated, indicating improved feature discrimination.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Effect of contrastive learning on embedding distributions.</p>
</caption>
<graphic xlink:href="fgene-16-1535279-g004.tif"/>
</fig>
<p>Removing the Graph Transformer Module led to the most pronounced performance decline, with AUC values dropping to 0.9481 (MDAD), 0.9482 (aBiofilm), and 0.9493 (DrugVirus). This demonstrates the Transformer&#x2019;s irreplaceable capability in modeling complex global dependencies and feature interactions within the graph structure. The multi-head self-attention mechanism effectively captures long-range relationships, which is particularly crucial for sparse biological networks like DrugVirus.</p>
<p>Replacing node perturbation with random edge deletion caused minor but consistent performance degradation across all datasets: AUC decreased to 0.9727 (MDAD), 0.9719 (aBiofilm), and 0.9701 (DrugVirus). While edge deletion remains a viable augmentation strategy, node perturbation&#x2019;s superior performance (average 0.3% improvement) suggests its advantage in preserving critical structural information during view generation. This effect is especially notable on DrugVirus, where biological interaction sparsity demands more nuanced augmentation.</p>
<p>The ablation experiments confirm that each module uniquely enhances CLMT&#x2019;s performance:Contrastive learning mitigates overfitting through view invariance. Graph Transformer enables global relational reasoning. Node perturbation optimizes augmentation for biological graph characteristics. Their combined effect achieves state-of-the-art AUC values (&#x3e;0.97 on all datasets), validating CLMT&#x2019;s robustness in diverse microbe-drug-virus association prediction scenarios.</p>
</sec>
<sec id="s3-6">
<title>3.6 Case study</title>
<p>In this case study, we aimed to validate the practical effectiveness of the CLMT model in identifying new microbe-drug associations by selecting three commonly used drugs-Cloxacillin, Carvacrol, and Ciprofloxacin-and the microorganism <italic>Mycobacterium tuberculosis</italic> from the MDAD dataset. For each drug, we cross-checked the top 20 predicted microorganisms by searching for their synonyms in the MeSH and DrugBank databases. Additionally, we verified whether the predicted microbe-drug associations had been reported in the scientific literature through PubMed searches.</p>
<p>Cloxacillin, a semi-synthetic penicillin antibiotic, is widely used to treat infections caused by beta-hemolytic streptococci, pneumococci, and staphylococci (<xref ref-type="bibr" rid="B15">Grillo et al., 2023</xref>). It is particularly effective against penicillinase-producing strains of <italic>Staphylococcus aureus</italic> and <italic>Staphylococcus</italic> epidermidis, which are resistant to other antibiotics (<xref ref-type="bibr" rid="B2">Aldman et al., 2022</xref>). Research has demonstrated that cloxacillin inhibits up to 50% of the activity of <italic>S. aureus</italic>, S. haematobium, and <italic>Salmonella typhi</italic> (<xref ref-type="bibr" rid="B35">Orogade and Akuse, 2004</xref>). In our study, 15 of the top 20 microorganisms predicted to be associated with cloxacillin (75%) were confirmed in the literature, as shown in <xref ref-type="table" rid="T6">Table 6</xref>.</p>
<table-wrap id="T6" position="float">
<label>TABLE 6</label>
<caption>
<p>The top 20 Cloxacillin-related microbes predicted by CLMT and the related publications.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Rank</th>
<th align="left">Microbe</th>
<th align="left">Evidence</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">1</td>
<td align="left">
<italic>Enterobacter</italic> aerogenes</td>
<td align="left">PMID22001269</td>
</tr>
<tr>
<td align="left">2</td>
<td align="left">
<italic>Clostridium</italic> pasteurianum</td>
<td align="left">Unconfirmeda</td>
</tr>
<tr>
<td align="left">3</td>
<td align="left">
<italic>Streptomyces</italic> sp. nov.</td>
<td align="left">PMID6970744</td>
</tr>
<tr>
<td align="left">4</td>
<td align="left" style="color:#000000">
<italic>Staphylococcus aureus</italic>
</td>
<td align="left">PMID15490798</td>
</tr>
<tr>
<td align="left">5</td>
<td align="left">Burkholderia cepacia</td>
<td align="left">Unconfirmeda</td>
</tr>
<tr>
<td align="left">6</td>
<td align="left" style="color:#000000">
<italic>Klebsiella pneumoniae</italic>
</td>
<td align="left">PMID20597925</td>
</tr>
<tr>
<td align="left">7</td>
<td align="left">Thermus thermophilus</td>
<td align="left">Unconfirmeda</td>
</tr>
<tr>
<td align="left">8</td>
<td align="left" style="color:#000000">
<italic>Bacillus subtilis</italic>
</td>
<td align="left">PMID25945113</td>
</tr>
<tr>
<td align="left">9</td>
<td align="left" style="color:#000000">
<italic>Salmonella typhi</italic>
</td>
<td align="left">PMID15490798</td>
</tr>
<tr>
<td align="left">10</td>
<td align="left" style="color:#000000">
<italic>Helicobacter pylori</italic>
</td>
<td align="left">PMID10748053</td>
</tr>
<tr>
<td align="left">11</td>
<td align="left">Schistosoma</td>
<td align="left">PMID15490798</td>
</tr>
<tr>
<td align="left">12</td>
<td align="left">
<italic>Candida</italic> albicans</td>
<td align="left">PMID2713774</td>
</tr>
<tr>
<td align="left">13</td>
<td align="left">
<italic>Micrococcus</italic> luteus</td>
<td align="left">PMID7771695</td>
</tr>
<tr>
<td align="left">14</td>
<td align="left">
<italic>Bacillus</italic> cereusereus</td>
<td align="left">PMID24876650</td>
</tr>
<tr>
<td align="left">15</td>
<td align="left">Francisella novicida</td>
<td align="left">Unconfirmeda</td>
</tr>
<tr>
<td align="left">16</td>
<td align="left">Pantoea agglomerans</td>
<td align="left">PMID33666040</td>
</tr>
<tr>
<td align="left">17</td>
<td align="left">
<italic>Candida</italic> dubliniensis</td>
<td align="left">PMID316353125</td>
</tr>
<tr>
<td align="left">18</td>
<td align="left">
<italic>Candida</italic> spp.</td>
<td align="left">PMID21496537</td>
</tr>
<tr>
<td align="left">19</td>
<td align="left">Baker&#x2019;s yeast</td>
<td align="left">PMID25945113</td>
</tr>
<tr>
<td align="left">20</td>
<td align="left">
<italic>Klebsiella</italic> planticola</td>
<td align="left">Unconfirmeda</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Carvacrol, a naturally occurring phenolic monoterpene found in aromatic plants, has demonstrated a wide range of bioactivities in both <italic>in vivo</italic> and <italic>in vitro</italic> studies. These include antioxidant (<xref ref-type="bibr" rid="B7">Churklam et al., 2020</xref>), diabetes prevention (<xref ref-type="bibr" rid="B4">Arkali et al., 2021</xref>), hepatoprotective (<xref ref-type="bibr" rid="B11">Elbe et al., 2020</xref>), reproductive (<xref ref-type="bibr" rid="B40">Saghrouchni et al., 2023</xref>), antimicrobial, and immunomodulatory properties (<xref ref-type="bibr" rid="B6">Chraibi et al., 2020</xref>). Additionally, carvacrol is used as a food preservative due to its flavoring properties (<xref ref-type="bibr" rid="B37">Patel, 2015</xref>). Previous research has highlighted its association with various microorganisms. For instance (<xref ref-type="bibr" rid="B1">Abdelhamid and Yousef, 2021</xref>), described how carvacrol counteracts desiccation-resistant <italic>Salmonella</italic> nacionalis, suggesting its potential as an additive against desiccation-adapted <italic>Enterococcus faecalis</italic> in low-moisture foods (<xref ref-type="bibr" rid="B17">Javed et al., 2021</xref>). demonstrated that carvacrol and its metabolites have beneficial effects on immune dysfunction and infection related to COVID-19. Moreover (<xref ref-type="bibr" rid="B49">Wang Y. et al., 2020</xref>), found that carvacrol reduced biofilm formation and extracellular polysaccharide secretion by <italic>Pseudomonas</italic> fluorescens and <italic>S. aureus</italic>, without affecting cell viability. Of the top 20 microorganisms predicted to be associated with carvacrol, 17 were confirmed by the literature, as shown in <xref ref-type="table" rid="T7">Table 7</xref>.</p>
<table-wrap id="T7" position="float">
<label>TABLE 7</label>
<caption>
<p>The top 20 Carvacrol-related microbes predicted by CLMT and the related publications.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Rank</th>
<th align="left">Microbe</th>
<th align="left">Evidence</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">1</td>
<td align="left">
<italic>Streptococcus</italic> mutans</td>
<td align="left">PMID: 28233286</td>
</tr>
<tr>
<td align="left">2</td>
<td align="left">Enteric bacteria and other eubacteria</td>
<td align="left">PMID: 16355827</td>
</tr>
<tr>
<td align="left">3</td>
<td align="left">
<italic>Streptomyces</italic> sp. nov.</td>
<td align="left">Unconfirmeda</td>
</tr>
<tr>
<td align="left">4</td>
<td align="left">
<italic>Vibrio</italic> campbellii</td>
<td align="left">Unconfirmeda</td>
</tr>
<tr>
<td align="left">5</td>
<td align="left">
<italic>Micrococcus</italic> luteus</td>
<td align="left">PMID: 33240953</td>
</tr>
<tr>
<td align="left">6</td>
<td align="left" style="color:#000000">
<italic>Salmonella enterica</italic>
</td>
<td align="left">PMID: 20132667</td>
</tr>
<tr>
<td align="left">7</td>
<td align="left" style="color:#000000">
<italic>Staphylococcus aureus</italic>
</td>
<td align="left">PMID: 34730626</td>
</tr>
<tr>
<td align="left">8</td>
<td align="left">Stenotrophomonas maltophilia</td>
<td align="left">PMID: 14659660</td>
</tr>
<tr>
<td align="left">9</td>
<td align="left" style="color:#000000">
<italic>Enterococcus faecalis</italic>
</td>
<td align="left">PMID: 29877104</td>
</tr>
<tr>
<td align="left">10</td>
<td align="left" style="color:#000000">
<italic>Bacillus anthracis</italic>
</td>
<td align="left">Unconfirmeda</td>
</tr>
<tr>
<td align="left">11</td>
<td align="left">Kocuria rhizophila</td>
<td align="left">PMID: 37481932</td>
</tr>
<tr>
<td align="left">12</td>
<td align="left" style="color:#000000">
<italic>Mycobacterium tuberculosis</italic>
</td>
<td align="left">PMID: 31552700</td>
</tr>
<tr>
<td align="left">13</td>
<td align="left" style="color:#000000">
<italic>Klebsiella pneumoniae</italic>
</td>
<td align="left">PMID: 34729712</td>
</tr>
<tr>
<td align="left">14</td>
<td align="left">Edwardsiella tarda</td>
<td align="left">PMID: 37476823</td>
</tr>
<tr>
<td align="left">15</td>
<td align="left" style="color:#000000">
<italic>Pseudomonas aeruginosa</italic>
</td>
<td align="left">PMID: 35776742</td>
</tr>
<tr>
<td align="left">16</td>
<td align="left">
<italic>Klebsiella</italic> planticola</td>
<td align="left">PMID: 23030501</td>
</tr>
<tr>
<td align="left">17</td>
<td align="left">
<italic>Staphylococcus</italic> epidermidis</td>
<td align="left">PMID: 37508194</td>
</tr>
<tr>
<td align="left">18</td>
<td align="left">Burkholderia cenocepacia</td>
<td align="left">PMID: 26946055</td>
</tr>
<tr>
<td align="left">19</td>
<td align="left" style="color:#000000">
<italic>Salmonella Typhi</italic>
</td>
<td align="left">PMID: 16355827</td>
</tr>
<tr>
<td align="left">20</td>
<td align="left">
<italic>Acinetobacter</italic> baumannii</td>
<td align="left">PMID: 25177730</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Ciprofloxacin, a fluoroquinolone antibiotic, is widely used for treating a variety of infections, including pneumonia, typhoid fever, and skin and soft tissue infections (<xref ref-type="bibr" rid="B34">McCurdy et al., 2017</xref>). Numerous studies have confirmed its effectiveness against various human microorganisms. For example (<xref ref-type="bibr" rid="B39">Rehaman et al., 2019</xref>), demonstrated ciprofloxacin&#x2019;s efficacy against <italic>Pseudomonas aeruginosa</italic>, an opportunistic pathogen (<xref ref-type="bibr" rid="B24">Liu et al., 2021</xref>). reported reduced lung inflammation in pneumonia patients treated with ciprofloxacin, while (<xref ref-type="bibr" rid="B45">Trinh et al., 2017</xref>) found that combining ciprofloxacin with ceftriaxone provided the most effective treatment for foodborne <italic>Vibrio</italic> traumaticus. In our study, all 20 of the top microorganisms predicted to be associated with ciprofloxacin were validated by the literature, as shown in <xref ref-type="table" rid="T8">Table 8</xref>.</p>
<table-wrap id="T8" position="float">
<label>TABLE 8</label>
<caption>
<p>The top 20 Ciprofloxacin-related microbes predicted by CLMT and the related publications.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Rank</th>
<th align="left">Microbe</th>
<th align="left">Evidence</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">1</td>
<td align="left" style="color:#000000">
<italic>Bacillus subtilis</italic>
</td>
<td align="left">PMID: 33218776</td>
</tr>
<tr>
<td align="left">2</td>
<td align="left" style="color:#000000">
<italic>Mycobacterium tuberculosis</italic>
</td>
<td align="left">PMID: 22421328</td>
</tr>
<tr>
<td align="left">3</td>
<td align="left">
<italic>Listeria</italic> monocytogenes</td>
<td align="left">PMID: 34068252</td>
</tr>
<tr>
<td align="left">4</td>
<td align="left" style="color:#000000">
<italic>Enterobacter cloacae</italic>
</td>
<td align="left">PMID: 11909836</td>
</tr>
<tr>
<td align="left">5</td>
<td align="left">
<italic>Proteus</italic> vulgaris</td>
<td align="left">PMID: 34638966</td>
</tr>
<tr>
<td align="left">6</td>
<td align="left">Enteric bacteria and other eubacteri</td>
<td align="left">PMID: 27436461</td>
</tr>
<tr>
<td align="left">7</td>
<td align="left" style="color:#000000">
<italic>Salmonella Typhi</italic>
</td>
<td align="left">PMID: 31877141</td>
</tr>
<tr>
<td align="left">8</td>
<td align="left">Actinobacillus actinomycetemcomitans</td>
<td align="left">PMID: 12019120</td>
</tr>
<tr>
<td align="left">9</td>
<td align="left" style="color:#000000">
<italic>Pseudomonas aeruginosa</italic>
</td>
<td align="left">PMID: 30605076</td>
</tr>
<tr>
<td align="left">10</td>
<td align="left">
<italic>Micrococcus</italic> luteus</td>
<td align="left">PMID: 3010848</td>
</tr>
<tr>
<td align="left">11</td>
<td align="left">
<italic>Haemophilus</italic> influenzae</td>
<td align="left">PMID: 8453168</td>
</tr>
<tr>
<td align="left">12</td>
<td align="left">
<italic>Streptococcus</italic> epidermidis</td>
<td align="left">PMID: 27579011</td>
</tr>
<tr>
<td align="left">13</td>
<td align="left" style="color:#000000">
<italic>Staphylococcus aureus</italic>
</td>
<td align="left">PMID: 35301951</td>
</tr>
<tr>
<td align="left">14</td>
<td align="left">
<italic>Klebsiella</italic> planticola</td>
<td align="left">PMID: 25465871</td>
</tr>
<tr>
<td align="left">15</td>
<td align="left">Providencia stuartii</td>
<td align="left">PMID: 15528892</td>
</tr>
<tr>
<td align="left">16</td>
<td align="left">Stenotrophomonas maltophilia</td>
<td align="left">PMID: 14982788</td>
</tr>
<tr>
<td align="left">17</td>
<td align="left" style="color:#000000">
<italic>Bacillus anthracis</italic>
</td>
<td align="left">PMID: 22064542</td>
</tr>
<tr>
<td align="left">18</td>
<td align="left" style="color:#000000">
<italic>Escherichia coli</italic>
</td>
<td align="left">PMID: 35091053</td>
</tr>
<tr>
<td align="left">19</td>
<td align="left">Porphyromonas gingivalis</td>
<td align="left">PMID: 15231772</td>
</tr>
<tr>
<td align="left">20</td>
<td align="left" style="color:#000000">
<italic>Helicobacter pylori</italic>
</td>
<td align="left">PMID: 25721770</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In addition, <italic>M. tuberculosis</italic> was selected for our case study. This Gram-positive, aerobic bacterium is the causative agent of tuberculosis, one of the deadliest diseases worldwide. According to the 2019 Global Tuberculosis Report (<xref ref-type="bibr" rid="B52">WHO Global, 2019</xref>), tuberculosis resulted in 1.5 million deaths in 2018. As shown in <xref ref-type="table" rid="T9">Table 9</xref>, 17 of the top 20 predicted drugs for <italic>M. tuberculosis</italic> have been supported by prior studies. This underscores the CLMT model&#x2019;s strong predictive ability in case studies involving drugs and microorganisms.</p>
<table-wrap id="T9" position="float">
<label>TABLE 9</label>
<caption>
<p>The top 20 <italic>Mycobacterium</italic> tuberculosis-associated drugs predicted by CLMT and the related publications.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Rank</th>
<th align="left">Drug</th>
<th align="left">Evidence</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">1</td>
<td align="left">Calanolide A</td>
<td align="left">PMID: 14980631</td>
</tr>
<tr>
<td align="left">2</td>
<td align="left">Hydrogen peroxide</td>
<td align="left">PMID: 30551469</td>
</tr>
<tr>
<td align="left">3</td>
<td align="left">Ciprofloxacin</td>
<td align="left">PMID: 16154314</td>
</tr>
<tr>
<td align="left">4</td>
<td align="left">beta-Pinene</td>
<td align="left">PMID: 19753839</td>
</tr>
<tr>
<td align="left">5</td>
<td align="left">Pyrazinamide</td>
<td align="left">PMID: 26521205</td>
</tr>
<tr>
<td align="left">6</td>
<td align="left">Vitamin C</td>
<td align="left">PMID: 23695675</td>
</tr>
<tr>
<td align="left">7</td>
<td align="left">Gentamicin</td>
<td align="left">PMID: 22143521</td>
</tr>
<tr>
<td align="left">8</td>
<td align="left">Rilpivirine</td>
<td align="left">Unconfirmeda</td>
</tr>
<tr>
<td align="left">9</td>
<td align="left">Ceforanide</td>
<td align="left">PMID: 7624446</td>
</tr>
<tr>
<td align="left">10</td>
<td align="left">Zidovudine</td>
<td align="left">PMID: 16154314</td>
</tr>
<tr>
<td align="left">11</td>
<td align="left">Polysorbate 80</td>
<td align="left">Unconfirmeda</td>
</tr>
<tr>
<td align="left">12</td>
<td align="left">Amikacin</td>
<td align="left">PMID: 29311078</td>
</tr>
<tr>
<td align="left">13</td>
<td align="left">Zinc oxide</td>
<td align="left">PMID: 33845951</td>
</tr>
<tr>
<td align="left">14</td>
<td align="left">Vanillylacetone</td>
<td align="left">Unconfirmeda</td>
</tr>
<tr>
<td align="left">15</td>
<td align="left">Vitamin E</td>
<td align="left">PMID: 26491981</td>
</tr>
<tr>
<td align="left">16</td>
<td align="left">Darunavir</td>
<td align="left">PMID: 28193650</td>
</tr>
<tr>
<td align="left">17</td>
<td align="left">Saquinavir</td>
<td align="left">PMID: 33841429</td>
</tr>
<tr>
<td align="left">18</td>
<td align="left">Lopinavir</td>
<td align="left">PMID: 21442799</td>
</tr>
<tr>
<td align="left">19</td>
<td align="left">Tobramycin</td>
<td align="left">PMID: 19723387</td>
</tr>
<tr>
<td align="left">20</td>
<td align="left">Minocycline</td>
<td align="left">PMID: 30597040</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s4">
<title>4 Discussion and conclusion</title>
<p>The microbe-drug association prediction task seeks to identify potential associations between microbes and drugs, which can support drug development and disease treatment. In this study, we propose the CLMT model for this task. The CLMT model improves learning capabilities by integrating a Graph Transformer network with contrastive learning techniques. Specifically, we utilize a multilayer Graph Convolutional Network (GCN) to capture the complex relationships between microbes and drugs. The contrastive learning module further enhances the model&#x2019;s discriminative ability, thereby improving prediction accuracy.</p>
<p>By effectively modeling complex interactions and overcoming data sparsity, CLMT can serve as a valuable tool in early-stage drug screening, ultimately reducing experimental costs and speeding up the development pipeline. Its robust performance on public datasets suggests that CLMT has the potential to be integrated into clinical decision-making frameworks, offering insights that could lead to more personalized and effective treatment strategies. The findings of this study have notable biological implications. By elucidating previously unknown associations between microbes and drugs, CLMT can contribute to a deeper understanding of the molecular mechanisms underlying drug efficacy and resistance. These insights are particularly relevant in the context of rising antimicrobial resistance and the need for precision medicine. Furthermore, the ability of CLMT to highlight subtle, yet biologically meaningful patterns in microbe-drug interactions may inform future research on microbial metabolism, host-microbe interactions, and the role of the microbiome in disease progression. In this way, the model not only advances computational methodology but also holds promise for driving novel biological discoveries.</p>
<p>While our experimental results on two publicly available datasets demonstrate the effectiveness of CLMT, it is important to acknowledge several limitations and failure cases. In certain instances, the model&#x2019;s performance was less robust. For example, in cases where the microbe-drug association data is extremely sparse, CLMT sometimes struggled to capture weaker or less obvious associations. This may be due to insufficient signal in the available data or limitations in the current data augmentation strategy. When the relationships between certain microbes and drugs are subtle or not well-characterized by the provided features, the model occasionally misclassified these associations. This suggests that additional biological information (e.g., gene expression profiles or metabolic pathways) might be needed to fully capture the underlying mechanisms. Although CLMT performs well on the MDAD and aBiofilm datasets, its scalability and effectiveness on larger or more heterogeneous datasets remain to be thoroughly evaluated. Future work is needed to optimize the model structure for such scenarios.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement </title>
<p>The original contributions presented in the study are publicly available. This data can be found here: GitHub repository, <ext-link ext-link-type="uri" xlink:href="https://github.com/qimou-515/CLMT">https://github.com/qimou-515/CLMT</ext-link>.</p>
</sec>
<sec sec-type="author-contributions" id="s6">
<title>Author contributions</title>
<p>LX: Conceptualization, Data curation, Formal Analysis, Writing&#x2013;original draft. JW: Conceptualization, Data curation, Formal Analysis, Writing&#x2013;original draft. LF: Conceptualization, Data curation, Formal Analysis, Writing&#x2013;original draft. LW: Methodology, Project administration, Resources, Writing&#x2013;review and editing. XZ: Methodology, Project administration, Resources, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s7">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This work was partly sponsored by the National Natural Science Foundation of China (No. 62272064), the Scientific Research Program of Education Department of Hunan Province (23A0514), the Natural Science Foundation of Hunan Province (No. 2023JJ60185), the Natural Science Foundation of Hunan Province Program (2022JJ50138),the Application-oriented Special Disciplines, Double First-Class University Project of Hunan Province (Xiangjiaotong [2018] 469) and the Hunan Provincial Education Department Scientific Research Project (No. 20B080).</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abdelhamid</surname>
<given-names>A. G.</given-names>
</name>
<name>
<surname>Yousef</surname>
<given-names>A. E.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Carvacrol and thymol combat desiccation resistance mechanisms in <italic>Salmonella enterica</italic> serovar Tennessee</article-title>. <source>Microorganisms</source> <volume>10</volume> (<issue>1</issue>), <fpage>44</fpage>. <pub-id pub-id-type="doi">10.3390/microorganisms10010044</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Aldman</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Kavyani</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Kahn</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>P&#xe5;hlman</surname>
<given-names>L. I.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Treatment outcome with penicillin G or cloxacillin in penicillin-susceptible <italic>Staphylococcus aureus</italic> bacteraemia: a retrospective cohort study</article-title>. <source>Int. J. Antimicrob. Agents</source> <volume>59</volume> (<issue>4</issue>), <fpage>106567</fpage>. <pub-id pub-id-type="doi">10.1016/j.ijantimicag.2022.106567</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Andersen</surname>
<given-names>P. I.</given-names>
</name>
<name>
<surname>Ianevski</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lysvand</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Vitkauskiene</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Oksenych</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Bj&#xf8;r&#xe5;s</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Discovery and development of safe-in-man broad-spectrum antiviral agents</article-title>. <source>Int. J. Infect. Dis.</source> <volume>93</volume>, <fpage>268</fpage>&#x2013;<lpage>276</lpage>. <pub-id pub-id-type="doi">10.1016/j.ijid.2020.02.018</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arkali</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Aksakal</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kaya</surname>
<given-names>&#x15e;. &#xd6;.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Protective effects of carvacrol against diabetes-induced reproductive damage in male rats: modulation of Nrf2/HO-1 signalling pathway and inhibition of Nf-kB-mediated testicular apoptosis and inflammation</article-title>. <source>Andrologia</source> <volume>53</volume> (<issue>2</issue>), <fpage>e13899</fpage>. <pub-id pub-id-type="doi">10.1111/and.13899</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bag</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>S. K.</given-names>
</name>
<name>
<surname>Tiwari</surname>
<given-names>M. K.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>An efficient recommendation generation using relevant Jaccard similarity</article-title>. <source>Inf. Sci.</source> <volume>483</volume>, <fpage>53</fpage>&#x2013;<lpage>64</lpage>. <pub-id pub-id-type="doi">10.1016/j.ins.2019.01.023</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chraibi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Farah</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Elamin</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Iraqui</surname>
<given-names>H. M.</given-names>
</name>
<name>
<surname>Fikri-Benbrahim</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Characterization, antioxidant, antimycobacterial, antimicrobial effcts of Moroccan rosemary essential oil, and its synergistic antimicrobial potential with carvacrol</article-title>. <source>J. Adv. Pharm. Technol. and Res.</source> <volume>11</volume> (<issue>1</issue>), <fpage>25</fpage>&#x2013;<lpage>29</lpage>. <pub-id pub-id-type="doi">10.4103/japtr.JAPTR_74_19</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Churklam</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chaturongakul</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ngamwongsatit</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Aunpad</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>The mechanisms of action of carvacrol and its synergism with nisin against Listeria monocytogenes on sliced bologna sausage</article-title>. <source>Food control.</source> <volume>108</volume>, <fpage>106864</fpage>. <pub-id pub-id-type="doi">10.1016/j.foodcont.2019.106864</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dai</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Duan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Predicting miRNA-disease associations using an ensemble learning framework with resampling method</article-title>. <source>Briefings Bioinforma.</source> <volume>23</volume> (<issue>1</issue>), <fpage>bbab543</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbab543</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Deng</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Graph2MDA: a multi-modal variational graph embedding model for predicting microbe-drug associations</article-title>. <source>Bioinformatics</source> <volume>38</volume> (<issue>4</issue>), <fpage>1118</fpage>&#x2013;<lpage>1125</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btab792</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Durack</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lynch</surname>
<given-names>S. V.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>The gut microbiome: relationships with disease and opportunities for therapy</article-title>. <source>J. Exp. Med.</source> <volume>216</volume> (<issue>1</issue>), <fpage>20</fpage>&#x2013;<lpage>40</lpage>. <pub-id pub-id-type="doi">10.1084/jem.20180448</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Elbe</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yigitturk</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Cavusoglu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Baygar</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Ozgul Onal</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ozturk</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Comparison of ultrastructural changes and the anticarcinogenic effects of thymol and carvacrol on ovarian cancer cells: which is more effective?</article-title> <source>Ultrastruct. Pathol.</source> <volume>44</volume> (<issue>2</issue>), <fpage>193</fpage>&#x2013;<lpage>202</lpage>. <pub-id pub-id-type="doi">10.1080/01913123.2020.1740366</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gevers</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Knight</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Petrosino</surname>
<given-names>J. F.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>McGuire</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Birren</surname>
<given-names>B. W.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>The Human Microbiome Project: a community resource for the healthy human microbiome</article-title>. <source>PLoS Biol.</source> <volume>10</volume>, <fpage>e1001377</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pbio.1001377</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gill</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Pop</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>DeBoy</surname>
<given-names>R. T.</given-names>
</name>
<name>
<surname>Eckburg</surname>
<given-names>P. B.</given-names>
</name>
<name>
<surname>Turnbaugh</surname>
<given-names>P. J.</given-names>
</name>
<name>
<surname>Samuel</surname>
<given-names>B. S.</given-names>
</name>
<etal/>
</person-group> (<year>2006</year>). <article-title>Metagenomic analysis of the human distal gut microbiome</article-title>. <source>science</source> <volume>312</volume> (<issue>5778</issue>), <fpage>1355</fpage>&#x2013;<lpage>1359</lpage>. <pub-id pub-id-type="doi">10.1126/science.1124234</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goodall</surname>
<given-names>D. W.</given-names>
</name>
</person-group> (<year>1966</year>). <article-title>A new similarity index based on probability</article-title>. <source>Biometrics</source> <volume>22</volume>, <fpage>882</fpage>&#x2013;<lpage>907</lpage>. <pub-id pub-id-type="doi">10.2307/2528080</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Grillo</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Pujol</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mir&#xf3;</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>L&#xf3;pez-Contreras</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Euba</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Gasch</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Cloxacillin plus fosfomycin versus cloxacillin alone for methicillin-susceptible <italic>Staphylococcus aureus</italic> bacteremia: a randomized trial</article-title>. <source>Nat. Med.</source> <volume>29</volume> (<issue>10</issue>), <fpage>2518</fpage>&#x2013;<lpage>2525</lpage>. <pub-id pub-id-type="doi">10.1038/s41591-023-02569-0</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hiratani</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Mehta</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lillicrap</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Latham</surname>
<given-names>P. E.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>On the stability and scalability of node perturbation learning</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>35</volume>, <fpage>31929</fpage>&#x2013;<lpage>31941</lpage>.</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Javed</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Meeran</surname>
<given-names>M. F. N.</given-names>
</name>
<name>
<surname>Jha</surname>
<given-names>N. K.</given-names>
</name>
<name>
<surname>Ojha</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Carvacrol, a plant metabolite targeting viral protease (Mpro) and ACE2 in host cells can be a possible candidate for COVID-19</article-title>. <source>Front. Plant Sci.</source> <volume>11</volume>, <fpage>2237</fpage>. <pub-id pub-id-type="doi">10.3389/fpls.2020.601335</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Hierarchical multi-relational graph representation learning for large-scale prediction of drug-drug interactions</article-title>. <source>IEEE Trans. Big Data</source>, <fpage>1</fpage>&#x2013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.1109/tbdata.2025.3536924</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Relation-aware subgraph embedding with co-contrastive learning for drug-drug interaction prediction</article-title>. <source>arXiv Prepr. arXiv:2307.01507</source>.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Relation-aware graph structure embedding with co-contrastive learning for drug&#x2013;drug interaction prediction</article-title>. <source>Neurocomputing</source> <volume>572</volume>, <fpage>127203</fpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2023.127203</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kau</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Ahern</surname>
<given-names>P. P.</given-names>
</name>
<name>
<surname>Griffin</surname>
<given-names>N. W.</given-names>
</name>
<name>
<surname>Goodman</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Gordon</surname>
<given-names>J. I.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Human nutrition, the gut microbiome and the immune system</article-title>. <source>Nature</source> <volume>474</volume> (<issue>7351</issue>), <fpage>327</fpage>&#x2013;<lpage>336</lpage>. <pub-id pub-id-type="doi">10.1038/nature10213</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kuka&#x10d;ka</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Golkov</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Cremers</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Regularization for deep learning: a taxonomy</article-title>. <source>arXiv Prepr. arXiv:1710.10686</source>.</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Leier</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Positive-unlabeled learning in bioinformatics and computational biology: a brief review</article-title>. <source>Briefings Bioinforma.</source> <volume>23</volume> (<issue>1</issue>), <fpage>bbab461</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbab461</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xiang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Qu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Pneumonia caused by Pseudomonas fluorescens: a case report</article-title>. <source>BMC Pulm. Med.</source> <volume>21</volume> (<issue>1</issue>), <fpage>212</fpage>&#x2013;<lpage>216</lpage>. <pub-id pub-id-type="doi">10.1186/s12890-021-01573-9</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Long</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kwoh</surname>
<given-names>C. K.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2020a</year>). <article-title>Predicting human microbe-drug associations via graph convolutional network with conditional random field</article-title>. <source>Bioinformatics</source> <volume>36</volume> (<issue>19</issue>), <fpage>4918</fpage>&#x2013;<lpage>4927</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa598</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Long</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kwoh</surname>
<given-names>C. K.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2020b</year>). <article-title>Ensembling graph attention networks for human microbe-drug association prediction</article-title>. <source>Bioinformatics</source> <volume>36</volume> (<issue>Suppl. ment_2</issue>), <fpage>i779</fpage>&#x2013;<lpage>i786</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa891</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>L&#xf3;pez</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Fern&#xe1;ndez</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Garc&#xed;a</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Palade</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Herrera</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>An insight into classification with imbalanced data: empirical results and current trends on using data intrinsic characteristics</article-title>. <source>Inf. Sci.</source> <volume>250</volume>, <fpage>113</fpage>&#x2013;<lpage>141</lpage>. <pub-id pub-id-type="doi">10.1016/j.ins.2013.07.007</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lou</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Teng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Predicting miRNA-disease associations via learning multimodal networks and fusing mixed neighborhood information</article-title>. <source>Briefings Bioinforma.</source> <volume>23</volume> (<issue>5</issue>), <fpage>bbac159</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbac159</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Luo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Long</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>NTSHMDA: prediction of human microbe-disease association based on random walk by integrating network topological similarity</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinforma.</source> <volume>17</volume> (<issue>4</issue>), <fpage>1341</fpage>&#x2013;<lpage>1351</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2018.2883041</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Generalized matrix factorization based on weighted hypergraph learning for microbe-drug association prediction</article-title>. <source>Comput. Biol. Med.</source> <volume>145</volume>, <fpage>105503</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2022.105503</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Macpherson</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Harris</surname>
<given-names>N. L.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Interactions between commensal intestinal bacteria and the immune system</article-title>. <source>Nat. Rev. Immunol.</source> <volume>4</volume> (<issue>6</issue>), <fpage>478</fpage>&#x2013;<lpage>485</lpage>. <pub-id pub-id-type="doi">10.1038/nri1373</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Mao</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mohri</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhong</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Cross-entropy loss functions: theoretical analysis and applications</article-title>,&#x201d; in <source>International conference on Machine learning</source>. <publisher-name>pmlr</publisher-name>, <fpage>23803</fpage>&#x2013;<lpage>23828</lpage>.</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McCoubrey</surname>
<given-names>L. E.</given-names>
</name>
<name>
<surname>Favaron</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Awad</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Orlu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Gaisford</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Basit</surname>
<given-names>A. W.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Colonic drug delivery: formulating the next generation of colon-targeted therapeutics</article-title>. <source>J. Control. Release</source> <volume>353</volume>, <fpage>1107</fpage>&#x2013;<lpage>1126</lpage>. <pub-id pub-id-type="doi">10.1016/j.jconrel.2022.12.029</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McCurdy</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lawrence</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Quintas</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Woosley</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Flamm</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Tseng</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>
<italic>In vitro</italic> activity of delafloxacin and microbiological response against fluoroquinolone-susceptible and nonsusceptible <italic>Staphylococcus aureus</italic> isolates from two phase 3 studies of acute bacterial skin and skin structure infections</article-title>. <source>Antimicrob. agents Chemother.</source> <volume>61</volume> (<issue>9</issue>). <pub-id pub-id-type="doi">10.1128/aac.00772-17</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Orogade</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Akuse</surname>
<given-names>R. M.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Changing patterns in sensitivity of causative organisms of septicaemia in children: the need for quinolones</article-title>. <source>Afr. J. Med. Med. Sci.</source> <volume>33</volume> (<issue>1</issue>), <fpage>69</fpage>&#x2013;<lpage>72</lpage>.</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Panebianco</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Barchetti</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Simone</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Del Monte</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ciardi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Grompone</surname>
<given-names>M. D.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Negative multiparametric magnetic resonance imaging for prostate cancer: what&#x27;s next?</article-title> <source>Eur. Eurology</source> <volume>74</volume> (<issue>1</issue>), <fpage>48</fpage>&#x2013;<lpage>54</lpage>. <pub-id pub-id-type="doi">10.1016/j.eururo.2018.03.007</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Patel</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Plant essential oils and allied volatile fractions as multifunctional additives in meat and fish-based food products: a review</article-title>. <source>Food Addit. and Contam. Part A</source> <volume>32</volume> (<issue>7</issue>), <fpage>1049</fpage>&#x2013;<lpage>1064</lpage>. <pub-id pub-id-type="doi">10.1080/19440049.2015.1040081</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rajput</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Thakur</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sharma</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>aBiofilm: a resource of anti-biofilm agents and their potential implications in targeting antibiotic drug resistance</article-title>. <source>Nucleic acids Res.</source> <volume>46</volume> (<issue>D1</issue>), <fpage>D894-D900</fpage>&#x2013;<lpage>D900</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkx1157</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rehman</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Patrick</surname>
<given-names>W. M.</given-names>
</name>
<name>
<surname>Lamont</surname>
<given-names>I. L.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Mechanisms of ciprofloxacin resistance in <italic>Pseudomonas aeruginosa</italic>: new approaches to an old problem</article-title>. <source>J. Med. Microbiol.</source> <volume>68</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1099/jmm.0.000873</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Saghrouchni</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Barnossi</surname>
<given-names>A. E.</given-names>
</name>
<name>
<surname>Mssillou</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Lavkor</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Ay</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Kara</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Potential of carvacrol as plant growth-promotor and green fungicide against fusarium wilt disease of perennial ryegrass</article-title>. <source>Front. Plant Sci.</source> <volume>14</volume>, <fpage>973207</fpage>. <pub-id pub-id-type="doi">10.3389/fpls.2023.973207</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schwabe</surname>
<given-names>R. F.</given-names>
</name>
<name>
<surname>Jobin</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>The microbiome and cancer</article-title>. <source>Nat. Rev. Cancer</source> <volume>13</volume> (<issue>11</issue>), <fpage>800</fpage>&#x2013;<lpage>812</lpage>. <pub-id pub-id-type="doi">10.1038/nrc3610</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sommer</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>B&#xe4;ckhed</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>The gut microbiota-masters of host development and physiology</article-title>. <source>Nat. Rev. Microbiol.</source> <volume>11</volume> (<issue>4</issue>), <fpage>227</fpage>&#x2013;<lpage>238</lpage>. <pub-id pub-id-type="doi">10.1038/nrmicro2974</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>Y. Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>D. H.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>S. B.</given-names>
</name>
<name>
<surname>Ming</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J. Q.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>MDAD: a special resource for microbe-drug associations</article-title>. <source>Front. Cell. Infect. Microbiol.</source> <volume>8</volume>, <fpage>424</fpage>. <pub-id pub-id-type="doi">10.3389/fcimb.2018.00424</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kuang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>GSAMDA: a computational model for predicting potential microbe-drug associations based on graph attention network and sparse autoencoder</article-title>. <source>BMC Bioinforma.</source> <volume>23</volume> (<issue>1</issue>), <fpage>492</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-022-05053-7</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Trinh</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Gavin</surname>
<given-names>H. E.</given-names>
</name>
<name>
<surname>Satchell</surname>
<given-names>K. J. F.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Efficacy of ceftriaxone, cefepime, doxycycline, ciprofloxacin, and combination therapy for Vibrio vulnificus foodborne septicemia</article-title>. <source>Antimicrob. agents Chemother.</source> <volume>61</volume> (<issue>12</issue>). <pub-id pub-id-type="doi">10.1128/aac.01106-17</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ventura</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>O&#x27;flaherty</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Claesson</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Turroni</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Klaenhammer</surname>
<given-names>T. R.</given-names>
</name>
<name>
<surname>van Sinderen</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>Genome-scale analyses of health-promoting bacteria: probiogenomics</article-title>. <source>Nat. Rev. Microbiol.</source> <volume>7</volume> (<issue>1</issue>), <fpage>61</fpage>&#x2013;<lpage>71</lpage>. <pub-id pub-id-type="doi">10.1038/nrmicro2047</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Clinical characteristics of 138 hospitalized patients with 2019 novel coronavirus-infected pneumonia in wuhan, China</article-title>. <source>jama</source> <volume>323</volume> (<issue>11</issue>), <fpage>1061</fpage>&#x2013;<lpage>1069</lpage>. <pub-id pub-id-type="doi">10.1001/jama.2020.1585</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Ji</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Ye</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <source>Heterogeneous graph attention network the world wide web conference</source>, <fpage>2022</fpage>&#x2013;<lpage>2032</lpage>.</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hong</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Interactions between fish isolates Pseudomonas fluorescens and <italic>Staphylococcus aureus</italic> in dual-species biofilms and sensitivity to carvacrol</article-title>. <source>Food Microbiol.</source> <volume>91</volume>, <fpage>103506</fpage>. <pub-id pub-id-type="doi">10.1016/j.fm.2020.103506</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>iPiDi-PUL: identifying Piwi-interacting RNA-disease associations based on positive unlabeled learning</article-title>. <source>Briefings Bioinforma.</source> <volume>22</volume> (<issue>3</issue>), <fpage>bbaa058</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbaa058</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ley</surname>
<given-names>R. E.</given-names>
</name>
<name>
<surname>Volchkov</surname>
<given-names>P. Y.</given-names>
</name>
<name>
<surname>Stranges</surname>
<given-names>P. B.</given-names>
</name>
<name>
<surname>Avanesyan</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Stonebraker</surname>
<given-names>A. C.</given-names>
</name>
<etal/>
</person-group> (<year>2008</year>). <article-title>Innate immunity and intestinal microbiota in the development of Type 1 diabetes</article-title>. <source>Nature</source> <volume>455</volume> (<issue>7216</issue>), <fpage>1109</fpage>&#x2013;<lpage>1113</lpage>. <pub-id pub-id-type="doi">10.1038/nature07336</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="book">
<collab>WHO Global</collab> (<year>2019</year>). <source>Tuberculosis report</source>. <publisher-name>World Health Organization</publisher-name>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="http://www.who.int/tb/publications/global_report/en/">http://www.who.int/tb/publications/global_report/en/</ext-link>.</comment>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X. L.</given-names>
</name>
<name>
<surname>Mei</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Kwoh</surname>
<given-names>C. K.</given-names>
</name>
<name>
<surname>Ng</surname>
<given-names>S. K.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Positive-unlabeled learning for disease gene identification</article-title>. <source>Bioinformatics</source> <volume>28</volume> (<issue>20</issue>), <fpage>2640</fpage>&#x2013;<lpage>2647</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bts504</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>You</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Sui</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Graph contrastive learning with augmentations</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>33</volume>, <fpage>5812</fpage>&#x2013;<lpage>5823</lpage>.</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Predicting drug-disease associations through layer attention graph convolutional network</article-title>. <source>Briefings Bioinforma.</source> <volume>22</volume> (<issue>4</issue>), <fpage>bbaa243</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbaa243</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yun</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jeong</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Graph transformer networks</article-title>. <source>Adv. neural Inf. Process. Syst.</source>, <fpage>32</fpage>.</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zeng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Predicting disease-associated circular RNAs using deep forests combined with positive-unlabeled learning methods</article-title>. <source>Briefings Bioinforma.</source> <volume>21</volume> (<issue>4</issue>), <fpage>1425</fpage>&#x2013;<lpage>1436</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbz080</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Duan</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Prediction of microbe-drug associations based on KATZ measure</article-title>,&#x201d; in <source>2019 IEEE international conference on bioinformatics and biomedicine (BIBM)</source>. <publisher-name>IEEE</publisher-name>, <fpage>183</fpage>&#x2013;<lpage>187</lpage>.</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zimmermann</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Curtis</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Factors that influence the immune response to vaccination</article-title>. <source>Clin. Microbiol. Rev.</source> <volume>32</volume> (<issue>2</issue>). <pub-id pub-id-type="doi">10.1128/cmr.00084-18</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>