<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Pharmacol.</journal-id>
<journal-title>Frontiers in Pharmacology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Pharmacol.</abbrev-journal-title>
<issn pub-type="epub">1663-9812</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1398231</article-id>
<article-id pub-id-type="doi">10.3389/fphar.2024.1398231</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Pharmacology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>MPASL: multi-perspective learning knowledge graph attention network for synthetic lethality prediction in human cancer</article-title>
<alt-title alt-title-type="left-running-head">Zhang et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fphar.2024.1398231">10.3389/fphar.2024.1398231</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Ge</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1154273/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Yitong</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2671785/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yan</surname>
<given-names>Chaokun</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/787100/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Jianlin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1154238/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liang</surname>
<given-names>Wenjuan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2591785/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Luo</surname>
<given-names>Junwei</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/832518/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Luo</surname>
<given-names>Huimin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1034463/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Computer and Information Engineering</institution>, <institution>Henan University</institution>, <addr-line>Kaifeng</addr-line>, <addr-line>Henan</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Henan Key Laboratory of Big Data Analysis and Processing</institution>, <institution>Henan University</institution>, <addr-line>Kaifeng</addr-line>, <addr-line>Henan</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>College of Computer Science and Technology</institution>, <institution>Henan Polytechnic University</institution>, <addr-line>Jiaozuo</addr-line>, <addr-line>Henan</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/685112/overview">Sajjad Gharaghani</ext-link>, University of Tehran, Iran</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/838718/overview">Chunhou Zheng</ext-link>, Anhui University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/531759/overview">Quan Zou</ext-link>, University of Electronic Science and Technology of China, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Huimin Luo, <email>luohuimin@henu.edu.cn</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>21</day>
<month>05</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1398231</elocation-id>
<history>
<date date-type="received">
<day>09</day>
<month>03</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>26</day>
<month>04</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Zhang, Chen, Yan, Wang, Liang, Luo and Luo.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Zhang, Chen, Yan, Wang, Liang, Luo and Luo</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Synthetic lethality (SL) is widely used to discover the anti-cancer drug targets. However, the identification of SL interactions through wet experiments is costly and inefficient. Hence, the development of efficient and high-accuracy computational methods for SL interactions prediction is of great significance. In this study, we propose MPASL, a multi-perspective learning knowledge graph attention network to enhance synthetic lethality prediction. MPASL utilizes knowledge graph hierarchy propagation to explore multi-source neighbor nodes related to genes. The knowledge graph ripple propagation expands gene representations through existing gene SL preference sets. MPASL can learn the gene representations from both gene-entity perspective and entity-entity perspective. Specifically, based on the aggregation method, we learn to obtain gene-oriented entity embeddings. Then, the gene representations are refined by comparing the various layer-wise neighborhood features of entities using the discrepancy contrastive technique. Finally, the learned gene representation is applied in SL prediction. Experimental results demonstrated that MPASL outperforms several state-of-the-art methods. Additionally, case studies have validated the effectiveness of MPASL in identifying SL interactions between genes.</p>
</abstract>
<kwd-group>
<kwd>synthetic lethality prediction</kwd>
<kwd>knowledge graph</kwd>
<kwd>multi-perspective learning</kwd>
<kwd>deep learning</kwd>
<kwd>attention mechanism</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Experimental Pharmacology and Drug Discovery</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Cancer is a genetic disease caused by the accumulation of multiple mutations resulting from the interaction of internal and external factors (<xref ref-type="bibr" rid="B2">Barab&#xe1;si et al., 2011</xref>). Traditional cancer treatments such as chemotherapy often have serious side effects and harm healthy cells (<xref ref-type="bibr" rid="B16">Hanahan and Weinberg, 2011</xref>). Synthetic lethality (SL) is a genetic interaction that kills cancer cells selectively without damaging healthy cells (<xref ref-type="bibr" rid="B5">Boone et al., 2007</xref>; <xref ref-type="bibr" rid="B18">Hartwell et al., 1997</xref>; Iglehart and Silver, 2009). SL offers a tremendous depth of research opportunities for anti-cancer drug development and targeted cancer therapy, with researchers making great efforts to identify SL pairs. Discovering SL gene pairs relies heavily on high-throughput wet-lab screening techniques including RNAi screening (<xref ref-type="bibr" rid="B3">Bartz et al., 2006</xref>; <xref ref-type="bibr" rid="B27">Luo et al., 2009</xref>; <xref ref-type="bibr" rid="B14">Gregory et al., 2010</xref>; <xref ref-type="bibr" rid="B4">Blank et al., 2013</xref>; <xref ref-type="bibr" rid="B9">Chang et al., 2016</xref>) and CRISPR screening (<xref ref-type="bibr" rid="B15">Han et al., 2017</xref>; <xref ref-type="bibr" rid="B38">Shen et al., 2017</xref>). However, lab experiment-based screening methods are time-consuming and expensive and increase the risk of off-target effects (<xref ref-type="bibr" rid="B23">Liu et al., 2019</xref>). Thus, there is an urgent need for efficient and economical methods to overcome the deficiencies of high-throughput screening techniques (<xref ref-type="bibr" rid="B19">Huang et al., 2019</xref>).</p>
<p>To overcome these limitations, a several computational methods have been developed for SL prediction. These methods fall into two categories: (i) knowledge-based methods and (ii) supervised machine-learning methods (<xref ref-type="bibr" rid="B52">Zhu et al., 2023</xref>). Knowledge-based methods rely on prior knowledge or assumptions (i.e., gene mutations (<xref ref-type="bibr" rid="B26">Lu et al., 2020</xref>) or CNVs (<xref ref-type="bibr" rid="B25">Lu et al., 2018</xref>)) to detect SL pairs. For example, Zhang et al. (<xref ref-type="bibr" rid="B50">Zhang et al., 2015</xref>) proposed a combination of data-driven models with signaling pathway knowledge to discover SL interaction pairs by simulating the effects of gene knockout on cell death. Srihari et al. (<xref ref-type="bibr" rid="B39">Srihari et al., 2015</xref>) used copy-number and gene expression data to identify SL interactions. However, knowledge-based methods do not comprehensively utilize underlying patterns of known SL interactions. Machine learning methods such as decision trees (<xref ref-type="bibr" rid="B46">Wong et al., 2004</xref>), support vector machines (<xref ref-type="bibr" rid="B32">Paladugu et al., 2008</xref>; <xref ref-type="bibr" rid="B35">Qi et al., 2008</xref>), random forests (<xref ref-type="bibr" rid="B11">Das et al., 2019</xref>), and ensemble classifiers (<xref ref-type="bibr" rid="B33">Pandey et al., 2010</xref>; <xref ref-type="bibr" rid="B47">Wu et al., 2014</xref>) expedite the identification of SL pairs are challenging to apply to large-scale data due to the complex matrix operations.</p>
<p>Tremendous developments in deep learning-based methods have shown them to be effective in many biomedical tasks, including drug-target prediction (<xref ref-type="bibr" rid="B29">Mohamed et al., 2020</xref>), drug-disease prediction (<xref ref-type="bibr" rid="B49">Yu et al., 2021</xref>) and drug synergy prediction (<xref ref-type="bibr" rid="B51">Zhang et al., 2023</xref>) along with successful applications in SL prediction (<xref ref-type="bibr" rid="B19">Huang et al., 2019</xref>; <xref ref-type="bibr" rid="B23">Liu et al., 2019</xref>; <xref ref-type="bibr" rid="B8">Cai et al., 2020</xref>; <xref ref-type="bibr" rid="B22">Liany et al., 2020</xref>; <xref ref-type="bibr" rid="B17">Hao et al., 2021</xref>; <xref ref-type="bibr" rid="B24">Long et al., 2021</xref>). For example, Long et al. (<xref ref-type="bibr" rid="B24">Long et al., 2021</xref>) proposed a graph contextualized attention network to predict SL interactions. This model deploys a dual-attention mechanism to capture the importance of neighbors and feature graphs for node representation learning. Cai et al. (<xref ref-type="bibr" rid="B8">Cai et al., 2020</xref>) modeled SL interactions as a graph and adopted a dual-drop GNN to address the sparsity of SL networks. However, most of these methods are limited in the expressive capacity of homogeneous graphs.</p>
<p>Knowledge graphs (KGs) are multi-relational heterogeneous graphs where the nodes and edges correspond to different types of entities and relations, respectively (<xref ref-type="bibr" rid="B44">Wang et al., 2017</xref>; <xref ref-type="bibr" rid="B42">2019b</xref>). They overcome the limitations of homogeneous graphs by using rich semantic information between graph entities to discover potential relations. These have begun to equip bioinformaticians with powerful weapons for combining heterogeneous data plainly for SL pairs prediction. Wang et al. (<xref ref-type="bibr" rid="B45">Wang et al., 2021</xref>) presented a KGNN-based model, KG4SL, to predict SL interactions. It uses independent knowledge embeddings to capture the underlying biological mechanisms of interconnected SL pairs. Zhu et al. (<xref ref-type="bibr" rid="B52">Zhu et al., 2023</xref>) utilized relations in knowledge graphs to represent SL-related factors and learned latent representations of genes through message aggregation. It is evident that employing KG entities such as gene, pathway and their neighbors yields a more accurate embedding representation, but previously KG-based methods ignore the preferences of existing SL interactions and layer-wise differences of entities.</p>
<p>To solve these problems, we develop a novel end-to-end SL prediction model, MPASL, based on multi-perspective learning knowledge graph attention network. Our model consists of four main modules. First, we find gene neighbors via KG hierarchy propagation. Second, KG ripple propagation exploits existing SL interactions preferences to obtain gene representations with finer granularity. Third, MPASL enhances gene representations through a mixed perspective of gene-entity and entity-entity interactions. Specifically, in gene-entity interaction, the knowledge graph relation attention mechanism is designed to score and aggregate gene-oriented entity embeddings to characterize the importance of relationships and informativeness for each entity. Then, the entity enhancement layer obtains the gene-oriented entity embeddings by aggregating the embedding representations of entities and genes. Subsequently, in entity-entity interaction, the discrepancy contrastive layer refine entity embeddings by comparing the various layer-wise neighborhood features of entities, and the attention aggregator obtains the final gene embedding representations by assigning different weight coefficients to the entities. Finally, the objective function using the embedded representation of genes is defined to obtain the predictive scores for unobserved SL pairs.</p>
<p>The contributions of this work are described as follows.<list list-type="simple">
<list-item>
<p>&#x2022; We propose a novel end-to-end KG-based framework named MPASL, which synthetically and effectively uses ripple propagation and a mixed perspective of gene-entity and entity-entity interactions to learn gene embeddings in the KG.</p>
</list-item>
<list-item>
<p>&#x2022; To capture the preferences of existing SL interactions and discover potential hierarchical interests of genes, we introduce ripple propagation, which helps to rationally extend the potential interactions of genes and enrich the representation of genes.</p>
</list-item>
<list-item>
<p>&#x2022; Considering the layer-wise differences between entities, a mixed perspective module obtains a more informative representation of genes from entity-entity perspective by comparing the layer-wise entity embeddings gained from gene-entity perspective learning.</p>
</list-item>
<list-item>
<p>&#x2022; Comprehensive <italic>in silico</italic> experiments on SynLethDB dataset demonstrate that our MPASL model consistently outperforms other state-of-the-art methods.</p>
</list-item>
</list>
</p>
<p>The remainder of this paper is organized as follows. The proposed method and the dataset we used are presented in <xref ref-type="sec" rid="s2">Section 2</xref>. <xref ref-type="sec" rid="s3">Section 3</xref> presented the results and discussion, and <xref ref-type="sec" rid="s4">Section 4</xref> concluded the paper and discussed the further work.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<title>2 Materials and methods</title>
<p>In this section, we introduce the MPASL model. First, we discuss the SL prediction problem. Second, we introduce the dataset used by our model. Then, we provide a detailed explanation of the MPASL model framework and its components. Finally, we discuss the predictions of SL made by the MPASL model.</p>
<sec id="s2-1">
<title>2.1 Problem formulation</title>
<p>We model the SL interactions using an SL graph represented by <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mi mathvariant="italic">SL</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>(</mml:mo>
<mml:mi>V</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>E</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, where <italic>V</italic> represents a set of genes, &#x7c;<italic>V</italic>&#x7c; is the number of genes involved in SL pairs, and <italic>E</italic> denotes a set of interactions between SL pairs. We use a matrix <italic>A</italic> &#x2208; {0,1}<sup>&#x7c;<italic>V</italic>&#x7c;&#xd7;&#x7c;<italic>V</italic>&#x7c;</sup> to represent the adjacency matrix of the SL graph. In this adjacency matrix, if there is an SL interaction between Gene <italic>m</italic> and Gene <italic>n</italic>, then <italic>A</italic>
<sub>
<italic>mn</italic>
</sub> &#x003D; 1 and 0 otherwise.</p>
<p>In addition to the synthetic lethality between a pair of genes, we consider the auxiliary information of the genes and other related entities in the form of a knowledge graph. The knowledge graph, SynLethKG, is modeled as a heterogeneous graph, with nodes representing diverse entities and edges capturing the relationships between these entities. SynLethKG is represented by <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mi mathvariant="italic">KG</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>(</mml:mo>
<mml:mi>N</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>E</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf4">
<mml:math id="m4">
<mml:mi mathvariant="script">N</mml:mi>
</mml:math>
</inline-formula> corresponds to a set of nodes of an entity, <inline-formula id="inf5">
<mml:math id="m5">
<mml:mi mathvariant="script">E</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">N\times R\times N</mml:mi>
</mml:math>
</inline-formula> represents the set of interactions from the set of relations <italic>R</italic> in the KG between two entities in <italic>N</italic>. Each edge is modeled as a triple <inline-formula id="inf6">
<mml:math id="m6">
<mml:mi mathvariant="script">T</mml:mi>
</mml:math>
</inline-formula> &#x003D; <inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">E</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> of entities and relations. For an entity-relation-entity triplet, <italic>h</italic>, <italic>r</italic>, and <italic>t</italic> denote the head entity, relationship, and tail entity of the triple, respectively, with head entities <inline-formula id="inf8">
<mml:math id="m8">
<mml:mi>h</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
</mml:math>
</inline-formula>, tail entities <inline-formula id="inf9">
<mml:math id="m9">
<mml:mi>t</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
</mml:math>
</inline-formula>, and relation entities <inline-formula id="inf10">
<mml:math id="m10">
<mml:mi>r</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">E</mml:mi>
</mml:math>
</inline-formula>. For example, (Lung adenocarcinoma, Associates, <italic>SMAD7</italic>) indicates that <italic>SMAD7</italic> is associated with lung adenocarcinoma (<xref ref-type="bibr" rid="B48">Yeung et al., 2016</xref>; <xref ref-type="bibr" rid="B10">Dai et al., 2020</xref>). In the graph, nodes represent entities and edges represent relationships from the head entity node to the tail entity node.</p>
<p>Given the SL graph <inline-formula id="inf11">
<mml:math id="m11">
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mi mathvariant="italic">SL</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> and the KG of synthetic lethality <inline-formula id="inf12">
<mml:math id="m12">
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mi mathvariant="italic">KG</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>, the task is to predict whether there exists a synthetic lethal relationship between genes <italic>m</italic> and <italic>n</italic>. This is done by learning a mapping function <inline-formula id="inf13">
<mml:math id="m13">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi mathvariant="italic">mn</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>(</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>;</mml:mo>
<mml:mo>&#x0398;</mml:mo>
<mml:mo>;</mml:mo>
<mml:mi>G</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, which automatically generates gene embeddings from SynLethKG <inline-formula id="inf15">
<mml:math id="m15">
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mi mathvariant="italic">KG</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> and estimates the probability of SL interaction between gene <italic>m</italic> and <italic>n</italic> in the SL graph <inline-formula id="inf16">
<mml:math id="m16">
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mi mathvariant="italic">SL</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> to identify potential SL pairs, where &#x398; represents the weight parameter of the model function <inline-formula id="inf17">
<mml:math id="m17">
<mml:mi mathvariant="script">F</mml:mi>
</mml:math>
</inline-formula>.</p>
</sec>
<sec id="s2-2">
<title>2.2 Dataset description</title>
<p>SynLethDB (<xref ref-type="bibr" rid="B43">Wang et al., 2022</xref>) is a comprehensive and up-to-date database that containing information on SL interactions. It collects SL gene pairs from various sources, including biochemical analysis, public databases (<xref ref-type="bibr" rid="B37">Schmidt et al., 2013</xref>; <xref ref-type="bibr" rid="B31">Oughtred et al., 2019</xref>),computational predictions (<xref ref-type="bibr" rid="B36">Ryan et al., 2014</xref>), and text mining. It covers SL gene pairs in humans and four model organisms (mice, fruit flies, worms and yeast) and a gene-related knowledge graph called SynLethKG, which comprises 11 types of entities and 24 types of relationships. SynLethKG collects a variety of relationships for genes involved in synthetic lethal gene pairs, including gene-compound associations, gene-cancer associations, and other features about genes, drugs, and cancers, such as (Anatomy, expresses, gene), (Disease, presents, symptom) and (Gene, regulates, gene). In addition, 7 out of the 11 types of entities are directly related to genes, namely, anatomy, biological process, cellular component, compound, disease, molecular function, and pathway. According to (<xref ref-type="bibr" rid="B45">Wang et al., 2021</xref>), we used the same synthetic lethality data and knowledge graph as it. The SL gene pairs in SynLethDB have been widely used for training and testing machine learning models for SL prediction. Since the number of negative samples provided by SynLethDB for SL interactions is much less than the number of positive samples, we generated negative samples using the method used in KG4SL (<xref ref-type="bibr" rid="B45">Wang et al., 2021</xref>), where an equal number of unknown gene pairs were randomly selected as non-SL gene pairs to balance out the difference in distribution between positive and negative samples. In our study, we specifically focused on human SL interactions. The final SL dataset we used included 72,804 gene pairs involving 10,004 genes. Additionally, the KG used in our study had 54,012 nodes and 2,231,921 edges. <xref ref-type="table" rid="T1">Tables 1</xref> and <xref ref-type="table" rid="T2">2</xref> summarize the statistics for SL and SynLethKG. <xref ref-type="table" rid="T3">Tables 3</xref> and <xref ref-type="table" rid="T4">4</xref> show detailed information about the entities and relationships of SynLethKG.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Statistical information on SL datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="left">No.of genes</th>
<th align="left">No.of interactions</th>
<th align="left">Positive pairs</th>
<th align="left">Negative pairs</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">SL data</td>
<td align="left">10,004</td>
<td align="left">72,804</td>
<td align="left">36,402</td>
<td align="left">36,402</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>SynlethKG&#x2019;s statistics.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Datasets</th>
<th align="left">Entity types</th>
<th align="left">Relationship types</th>
<th align="left">No.of nodes</th>
<th align="left">No.of edges</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">SynlethKG</td>
<td align="left">11</td>
<td align="left">24</td>
<td align="left">54,012</td>
<td align="left">2,231,921</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Details of entities in SynLethKG.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Type</th>
<th align="left">No.of entities</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Anatomy</td>
<td align="left">400</td>
</tr>
<tr>
<td align="left">Biological process</td>
<td align="left">12,703</td>
</tr>
<tr>
<td align="left">Cellular component</td>
<td align="left">1,670</td>
</tr>
<tr>
<td align="left">Compound</td>
<td align="left">2,065</td>
</tr>
<tr>
<td align="left">Disease</td>
<td align="left">136</td>
</tr>
<tr>
<td align="left">Gene</td>
<td align="left">25,260</td>
</tr>
<tr>
<td align="left">Molecular function</td>
<td align="left">3,203</td>
</tr>
<tr>
<td align="left">Pathway</td>
<td align="left">2,069</td>
</tr>
<tr>
<td align="left">Pharmacologic class</td>
<td align="left">377</td>
</tr>
<tr>
<td align="left">Side effect</td>
<td align="left">5,702</td>
</tr>
<tr>
<td align="left">Symptom</td>
<td align="left">427</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Details of relationships in SynLethKG.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Type</th>
<th align="left">No.of relationships</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">(Anatomy, downregulates, gene)</td>
<td align="left">31</td>
</tr>
<tr>
<td align="left">(Anatomy, express, gene)</td>
<td align="left">6,17,175</td>
</tr>
<tr>
<td align="left">(Anatomy, upregulates, gene)</td>
<td align="left">26</td>
</tr>
<tr>
<td align="left">(Compound, binds, gene)</td>
<td align="left">16,323</td>
</tr>
<tr>
<td align="left">(Compound, causes, side effect)</td>
<td align="left">1,39,428</td>
</tr>
<tr>
<td align="left">(Compound, downregulates, gene)</td>
<td align="left">21,526</td>
</tr>
<tr>
<td align="left">(Compound, palliates, disease)</td>
<td align="left">384</td>
</tr>
<tr>
<td align="left">(Compound, resembles, compound)</td>
<td align="left">6,266</td>
</tr>
<tr>
<td align="left">(Compound, treats, disease)</td>
<td align="left">752</td>
</tr>
<tr>
<td align="left">(Compound, upregulates, gene)</td>
<td align="left">19,200</td>
</tr>
<tr>
<td align="left">(Disease, associates, gene)</td>
<td align="left">24,328</td>
</tr>
<tr>
<td align="left">(Disease, downregulates, gene)</td>
<td align="left">7,616</td>
</tr>
<tr>
<td align="left">(Disease, localizes, anatomy)</td>
<td align="left">3,373</td>
</tr>
<tr>
<td align="left">(Disease, presents, symptom)</td>
<td align="left">3,401</td>
</tr>
<tr>
<td align="left">(Disease, resembles, disease)</td>
<td align="left">404</td>
</tr>
<tr>
<td align="left">(Disease, upregulates, gene)</td>
<td align="left">7,730</td>
</tr>
<tr>
<td align="left">(Gene, covaries, gene)</td>
<td align="left">62,966</td>
</tr>
<tr>
<td align="left">(Gene, interacts, gene)</td>
<td align="left">1,47,638</td>
</tr>
<tr>
<td align="left">(Gene, participates, biological process)</td>
<td align="left">6,19,712</td>
</tr>
<tr>
<td align="left">(Gene, participates, cellular component)</td>
<td align="left">97,652</td>
</tr>
<tr>
<td align="left">(Gene, participates, molecular function)</td>
<td align="left">1,10,042</td>
</tr>
<tr>
<td align="left">(Gene, participates, pathway)</td>
<td align="left">57,441</td>
</tr>
<tr>
<td align="left">(Gene, regulates, gene)</td>
<td align="left">2,67,302</td>
</tr>
<tr>
<td align="left">(Pharmacologic class, includes, compound)</td>
<td align="left">1,205</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-3">
<title>2.3 Framework design</title>
<p>The overall pipeline of MPASL is shown in <xref ref-type="fig" rid="F1">Figure 1</xref>. The model consists of four modules including knowledge graph hierarchy propagation, knowledge graph ripple propagation, a mixed perspective of gene-entity and entity-entity interactions module and prediction module. In order to present the article more clearly, a mixed perspective of gene-entity and entity-entity interactions module is divided into two parts: gene-entity interaction and entity-entity interaction.<list list-type="simple">
<list-item>
<p>(1) Knowledge graph hierarchy propagation. This layer maps entities and relationships in the KG to vectors. We then recursively explore the set of multi-source neighbor nodes that are directly or indirectly related to genes in the KG.</p>
</list-item>
<list-item>
<p>(2) Knowledge graph ripple propagation. In this module, we introduce finer-grained entity embedding propagation using the set of existing SL interaction preferences for genes. This recursively extends the representation of genes with supplementary edge information, allowing for the automatic discovery of potential paths from genes with SL interactions to candidate genes. This approach connects the existing SL interaction set of genes with the prediction records, bringing interpretability to SL prediction.</p>
</list-item>
<list-item>
<p>(3) Gene-entity interaction. We split gene-entity interaction into KG relation attention mechanism and an entity enhancement layer. These layers score, aggregate, and update the embeddings of specific genes and entities with their neighborhood information, explicitly capturing the higher-order structural information and similarities in the knowledge graph and contributing to a stable learning process.</p>
</list-item>
<list-item>
<p>(4) Entity-entity interaction. We use a discrepancy contrastive layer to hierarchically compare the connected information of entities across different layers. We also employ an attention aggregator to obtain different weight coefficients for neighborhoods in the mixed perspective of entity. Iteratively propagating and updating entity representations with multiple layers of information increases the diversity of predicted embeddings.</p>
</list-item>
<list-item>
<p>(5) Prediction module. This module illustrates the learning and prediction of SL, using a series of aggregated and updated gene representations to compute prediction scores.</p>
</list-item>
</list>
</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Architecture of MPASL. <bold>(A)</bold> KG. The KG assists SL prediction and consists of 11 kinds of entities and 24 kinds of relationships. <bold>(B)</bold> Knowledge graph hierarchy propagation identifies sets of neighboring nodes for gene entities. Symbol <italic>e</italic> represents the initial entities associated with gene entities, which are sets of tail entities directly related to genes. Multi-hop sets correspond to triples associated with genes. The knowledge-based higher-order interaction information for genes is stored in these multi-hop sets. <bold>(C)</bold> Knowledge graph ripple propagation uses the KG to model gene embeddings at a finer level. <bold>(D)</bold> Gene-entity interaction. D(1) A KG relation attention mechanism applies attention scoring to surrounding relations in a gene-specific manner. D(2) The entity enhancement layer aggregates entity embeddings specific to genes to allocate different amounts of information to refine their embeddings. <bold>(E)</bold> Entity-entity interaction. E(1) Discrepancy contrastive layer integrates different high-order connectivity information for genes in terms of depth and width. E(2) The attention aggregator module employs an attention aggregator to assign different weight coefficients for different genes and generate updated gene embeddings. <bold>(F)</bold> The prediction module outputs the predicted probabilities of synthetic lethality between genes.</p>
</caption>
<graphic xlink:href="fphar-15-1398231-g001.tif"/>
</fig>
<sec id="s2-3-1">
<title>2.3.1 Knowledge graph hierarchy propagation</title>
<p>MPASL&#x2019;s knowledge graph hierarchy propagation obtains a set of multi-hop neighboring nodes for a set of genes. This layer encodes the crucial hierarchical information into the gene representations, enriching those representations constructed from entities in the KG and including the existing set of SL interactions.</p>
<p>The rich semantic connections between entities in the KG help identify potential complex relationships between entities. These complex relationships provide an additional perspective for exploring SL genes, aiding in the discovery of potential connections between genes and improving the accuracy of SL prediction. Obtaining relevant gene information from the KG requires information of associated entities having highly correlated relationships. Essentially, the entities having SL relationship with a gene provide at least some information about gene attributes. By transforming and comparing genes with entities, turning the related entity set obtained from existing SL interactions into an initial seed set for propagation in the KG, we capture information on gene-gene interactions. With the initial seed set, we can propagate KG associations from near to far along the KG, obtaining an extended entity set and a triple set of <italic>p</italic>-hops, effectively enriching the potential vector representation of genes. In summary, modeling gene representations by using relevant entities in the KG enhances gene information.</p>
<p>To do all of this, we first define the extended entity set of genes. For the input gene <italic>o</italic>, the set of entities with SL interactions with that input gene is treated as seeds in the KG. Then it extends along the KG to form a set of p-hopped extended entity sets <inline-formula id="inf18">
<mml:math id="m18">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> of the gene <italic>o</italic>, effectively expressing the interaction information of the potential semantics of the entities. The adjacent entity sets of gene <italic>o</italic> can be recursively represented as:<disp-formula id="e1">
<mml:math id="m19">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x003D;</mml:mo>
<mml:mfenced close="}" open="{">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mtext>&#x2009;&#x2009;</mml:mtext>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
<mml:mtext>&#x2009;&#x2009;</mml:mtext>
<mml:mi>h</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mi>p</mml:mi>
<mml:mo>&#x003D;</mml:mo>
<mml:mn>1,2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
<p>where <italic>p</italic> represents the distance from the initial set of entities. <inline-formula id="inf19">
<mml:math id="m20">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x003D;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x003D;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is the initial set of genes having SL relationships with gene <italic>o</italic> and serves as the seed set of gene <italic>o</italic> in the KG. This design emphasizes the original information of genes and reduces biases caused by multiple propagation layers, making it more effective in expanding potential vector representations of entities.</p>
<p>For a central gene in a KG subgraph, the set of entities <inline-formula id="inf20">
<mml:math id="m21">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x003D;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x003D;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> that already have synthetic lethality relationships with this gene is regarded as the starting point in the KG. The set of <italic>p</italic>-hopping triplet propagation constructed with this starting point is explored along the KG relationship:<disp-formula id="e2">
<mml:math id="m22">
<mml:msubsup>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x003D;</mml:mo>
<mml:mfenced close="}" open="{">
<mml:mrow>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mtext>&#x2009;&#x2009;</mml:mtext>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
<mml:mtext>&#x2009;&#x2009;</mml:mtext>
<mml:mi>h</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mi>p</mml:mi>
<mml:mo>&#x003D;</mml:mo>
<mml:mn>1,2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
<label>(2)</label>
</disp-formula>It is meaningful to construct the model using knowledge graphs as edge information, as adjacent entities can be seen as intuitive extensions of gene features. The knowledge graph extends the neighboring nodes in each layer, propagating layer by layer, from near to far, effectively capturing high-order interactive information based on the KG through hierarchy propagation. Symbolically, <italic>&#x25b;</italic> consists solely of tail entities, <italic>S</italic> is a set of knowledge triplets, <italic>p</italic> represents (one or more) hops, and <italic>l</italic>
<sub>
<italic>p</italic>
</sub> is the number of hops. To reduce the computational burden of MPASL, we use a fixed-size set of neighbors (<xref ref-type="bibr" rid="B42">Wang et al., 2019b</xref>) for each entity instead of the complete neighbor set.</p>
</sec>
<sec id="s2-3-2">
<title>2.3.2 Knowledge graph ripple propagation</title>
<p>We extend gene representations by supplementing auxiliary information with a KG ripple propagation to model interactions between genes in a finer grained manner. This technique relies on traversing all relevant entities and associations along ripple propagation in the KG. This process recursively captures the topological neighborhood structure of the central entity in multi-hop ripple sets. This helps to expand potential preference genes, increase the diversity of predicted embeddings, and discover potential SL relationships. When a given tail entity in the KG has different head entities and relationships, it carries different meanings and potential vector representations. The gene representation of the KG ripple propagation is constructed from the gene SL response <italic>O</italic>
<sub>
<italic>m</italic>
</sub> generated by the triplet propagation set <italic>S</italic>
<sub>
<italic>m</italic>
</sub> to explore the potential gene relationships.</p>
<p>To perform this operation, we first define the gene potential SL response <inline-formula id="inf21">
<mml:math id="m23">
<mml:msubsup>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> for the 0-hop based on entity <inline-formula id="inf22">
<mml:math id="m24">
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula>, where <italic>h</italic>
<sub>
<italic>i</italic>
</sub> represents the head entity that has existing SL relationship with gene <italic>m</italic> and <inline-formula id="inf23">
<mml:math id="m25">
<mml:msubsup>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> is the one-hop triplet propagation set of gene <italic>m</italic> obtained from KG hierarchy propagation. Each gene <italic>n</italic> is assigned a different weight towards the SL preference response of gene <italic>m</italic>:<disp-formula id="e3">
<mml:math id="m26">
<mml:msubsup>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x003D;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
<label>(3)</label>
</disp-formula>
<disp-formula id="e4">
<mml:math id="m27">
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x003D;</mml:mo>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">f</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mtext>&#x2009;&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced close="]" open="[">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(4)</label>
</disp-formula>where <italic>W</italic>
<sub>
<italic>a</italic>
</sub> is a trainable parameter. The vector <inline-formula id="inf24">
<mml:math id="m28">
<mml:msubsup>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> represents the 0th-order response of known SL interactions of gene <italic>m</italic> with respect to entity embedding <italic>n</italic>. In part C of <xref ref-type="fig" rid="F1">Figure 1</xref>, we use the orange rectangle to represent the 0th hop SL response, and the p-hop (<italic>p</italic> &#x2265; 1) SL response is represented by the blue rectangle.</p>
<p>Second, apart from the 0th jump, gene embeddings <italic>m</italic> are achieved by adding SL non-zero hop responses. The ripple set <inline-formula id="inf25">
<mml:math id="m29">
<mml:msubsup>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> is a set of triples that are <italic>p</italic> hops away from the seed set <inline-formula id="inf26">
<mml:math id="m30">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula>. These ripple sets are used to interact with the SL 0th-order response to obtain the <italic>p</italic> hop response of gene <italic>m</italic> to SL. Given the gene embedding <italic>n</italic> and the one-hop triplet propagation set <inline-formula id="inf27">
<mml:math id="m31">
<mml:msubsup>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> of gene <italic>m</italic>, for each triplet (<italic>h</italic>
<sub>
<italic>i</italic>
</sub>, <italic>r</italic>
<sub>
<italic>i</italic>
</sub>, <italic>t</italic>
<sub>
<italic>i</italic>
</sub>) in <inline-formula id="inf28">
<mml:math id="m32">
<mml:msubsup>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula>, the associated probability is assigned by comparing gene <italic>n</italic> with the head entity <italic>h</italic>
<sub>
<italic>i</italic>
</sub> and relation <italic>r</italic>
<sub>
<italic>i</italic>
</sub> in the triplet propagation set <inline-formula id="inf29">
<mml:math id="m33">
<mml:msubsup>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula>. Finally, after obtaining the correlation probability <italic>k</italic>
<sub>
<italic>i</italic>
</sub>, and the SL response <inline-formula id="inf30">
<mml:math id="m34">
<mml:msubsup>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> of gene <italic>m</italic> is calculated as a sum of the weighted tails corresponding to the correlation probability <italic>k</italic>
<sub>
<italic>i</italic>
</sub>. Finally, the vector <inline-formula id="inf31">
<mml:math id="m35">
<mml:msubsup>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> is returned:<disp-formula id="e5">
<mml:math id="m36">
<mml:msubsup>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x003D;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2208;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mi>p</mml:mi>
<mml:mo>&#x003D;</mml:mo>
<mml:mn>1,2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
<label>(5)</label>
</disp-formula>
<disp-formula id="e6">
<mml:math id="m37">
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x003D;</mml:mo>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">f</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mtext>&#x2009;&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">x</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x003D;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>exp</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2208;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mi>exp</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(6)</label>
</disp-formula>where <italic>p</italic> &#x003e; 0, <inline-formula id="inf32">
<mml:math id="m38">
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> are the head entity <italic>h</italic>
<sub>
<italic>i</italic>
</sub> and relation <italic>r</italic>
<sub>
<italic>i</italic>
</sub>, <inline-formula id="inf33">
<mml:math id="m39">
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> is the tail entity, and <inline-formula id="inf34">
<mml:math id="m40">
<mml:mi>n</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> is the embedding of gene <italic>n</italic>. In the embedding relationship <italic>r</italic>
<sub>
<italic>i</italic>
</sub> space, genes and entities may have different similarities under different relationships, and the associated probability <italic>k</italic>
<sub>
<italic>i</italic>
</sub> can be regarded as measuring the degree of similarity between genes <italic>n</italic> and entities <italic>h</italic>
<sub>
<italic>i</italic>
</sub> in the space of relation <italic>r</italic>
<sub>
<italic>i</italic>
</sub>.</p>
<p>We repeat the process of KG ripple propagation to obtain the first-order response <inline-formula id="inf35">
<mml:math id="m41">
<mml:msubsup>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> of genes <italic>m</italic> and the second-order response <inline-formula id="inf36">
<mml:math id="m42">
<mml:msubsup>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> of genes <italic>m</italic>, and this process can be iteratively performed on the triplet propagation set <inline-formula id="inf37">
<mml:math id="m43">
<mml:msubsup>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> of gene <italic>m</italic> in <italic>i</italic> &#x003D; 1, &#x2026; , <italic>p</italic>. After integrating all gene preference responses <inline-formula id="inf38">
<mml:math id="m44">
<mml:msubsup>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula>, we generate the final embedding of gene <italic>m</italic> by integrating all <italic>p</italic>-order responses:<disp-formula id="e7">
<mml:math id="m45">
<mml:msub>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x003D;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mfenced close="]" open="[">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(7)</label>
</disp-formula>
<disp-formula id="e8">
<mml:math id="m46">
<mml:mi>m</mml:mi>
<mml:mo>&#x003D;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x002B;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
<label>(8)</label>
</disp-formula>
</p>
</sec>
<sec id="s2-3-3">
<title>2.3.3 Gene-entity interaction</title>
<p>To capture the high-order similarities between gene-related entities in the KG, we propose a gene-entity interaction module. It consists of two parts: a KG relation attention mechanism and an entity enhancement layer. Each entity in the KG has different neighboring entities and relationships, leading to different meanings and potential vector representations. Furthermore, there exist complex associations among neighboring entities. We construct a weighted subgraph specific to each SL-related gene from the KG, allowing us to focus on the relevant entities. To capture entity embeddings, we apply a KG relation attention mechanism that takes into account the relationships between an entity and its individual neighbors, allowing us to describe the importance of each relationship to a specific entity and provide a more detailed understanding of its context. Additionally, we equip the gene-specific entity embeddings with enhancement operations to stabilize the latent representation of the entity in the embedding space.</p>
<sec id="s2-3-3-1">
<title>2.3.3.1 KG relation attention mechanism</title>
<p>When MPASL collects information from the vicinity of gene <italic>n</italic> in the KG, it scores each relation surrounding gene <italic>n</italic> in a manner specific to gene <italic>m</italic>. Thus, the gene <italic>m</italic>-oriented manner can be viewed as an early layer that increases the interaction of gene <italic>m</italic> with its weighted subgraph center entity <italic>n</italic>, and then the gene-oriented KG relation attention mechanism aggregates neighbor information in a gene <italic>m</italic>-specific manner. For any central entity <italic>n</italic> in the weighted subgraph oriented to gene <italic>m</italic>, different relationships have different indication weights for an entity, and the key step is to identify relevant nodes and determine the weight of edges to avoid assigning the same weight to different neighbors in the process of information aggregation. The weight of each edge is defined by a relation scoring function specific to gene <italic>m</italic>, and the proposed KG relation attention exploiting the information of gene <italic>m</italic>, gene <italic>n</italic>, and the relation to determine which neighboring entity connected to gene <italic>n</italic> is more informative. Therefore, each neighboring entity is weighted by attention <italic>&#x3c0;</italic>, where <italic>m</italic> represents different known genes and <italic>r</italic>
<sub>
<italic>n</italic>,<italic>e</italic>
</sub> represents the relationship <italic>r</italic> from the entity <italic>n</italic> to the neighboring entity <italic>e</italic>. We aggregate and weight each neighboring node of the entity to generate the final representation n(N(n)) of any central entity in the gene <italic>m</italic>-specific weighted subgraph:<disp-formula id="e9">
<mml:math id="m47">
<mml:mi>n</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x003D;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mi>e</mml:mi>
</mml:math>
<label>(9)</label>
</disp-formula>Assuming <italic>n</italic> is the central node, <italic>N</italic>(<italic>n</italic>) is a set of entities directly connected to <italic>n</italic>, and the size of <italic>N</italic>(<italic>n</italic>) can vary greatly among all entities. To maintain efficiency and consistency in each batch calculation mode, we uniformly extract a fixed number of <italic>k</italic> neighbors for each entity to represent its local structure (<xref ref-type="bibr" rid="B42">Wang et al., 2019b</xref>), and repeat this process <italic>p</italic> times.</p>
<p>In a subgraph specific to gene <italic>m</italic>, for the SL pair (<italic>m</italic>, <italic>n</italic>), the weight of the edge <italic>r</italic>
<sub>
<italic>m</italic>,<italic>n</italic>
</sub> is calculated as <inline-formula id="inf39">
<mml:math id="m48">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula>, where <italic>e</italic> is one of the entities specific to the gene <italic>m</italic> subgraph, and <italic>e</italic> &#x2208; <italic>N</italic>(<italic>n</italic>). In addition, <italic>m</italic> and <italic>r</italic>
<sub>
<italic>n</italic>,<italic>e</italic>
</sub> are feature embeddings of the gene <italic>m</italic> and relation <italic>r</italic>
<sub>
<italic>n</italic>,<italic>e</italic>
</sub>, and the attention score <inline-formula id="inf40">
<mml:math id="m49">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> denotes the attention weight of the relation <italic>r</italic>
<sub>
<italic>n</italic>,<italic>e</italic>
</sub> with respect to the gene <italic>m</italic>. The higher the attention weight, the more important the neighboring entity is, and the more informative the neighboring entity connected to gene <italic>m</italic> becomes. The incorporation of an attention mechanism, enables learning different weights for different neighbors (<xref ref-type="bibr" rid="B40">Veli&#x10d;kovi&#x107; et al., 2017</xref>). To compute the attention scores of the neighbors in the weighted subgraph <italic>&#x3c0;</italic>, we implement the function <inline-formula id="inf41">
<mml:math id="m50">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> by means of a neural network similar to the attention mechanism. To generate the final function <inline-formula id="inf42">
<mml:math id="m51">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> specific to any central entity of the weighted subgraph of the gene <italic>m</italic> we use the following formulas:<disp-formula id="e10">
<mml:math id="m52">
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x003D;</mml:mo>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mi mathvariant="normal">U</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x002B;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(10)</label>
</disp-formula>
<disp-formula id="e11">
<mml:math id="m53">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x003D;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mi mathvariant="normal">U</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x002B;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x002B;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(11)</label>
</disp-formula>where ReLU is the nonlinear activation function, &#x2016; represents the concatenation operation, and <italic>W</italic> and <italic>b</italic> are the trainable weights and biases. Specifically, <italic>W</italic>
<sub>1</sub> and <italic>b</italic>
<sub>1</sub> in Eq. <xref ref-type="disp-formula" rid="e10">10</xref> represent the weight and bias for the first layer of the neural network, while <italic>W</italic>
<sub>2</sub>, <italic>b</italic>
<sub>2</sub>, <italic>W</italic>
<sub>3</sub> and <italic>b</italic>
<sub>3</sub> denote the weights and biases for the second and output layers in Eq. <xref ref-type="disp-formula" rid="e11">11</xref>, respectively. The nonlinear activation function <italic>&#x3c3;</italic> is set as Sigmoid.</p>
<p>To make the attention coefficients among different entities comparable (with the sum of the attention coefficient of all adjacent nodes being 1), we use the softmax function to normalize the coefficients of all entities <italic>e</italic> related to the gene <italic>n</italic> (<xref ref-type="bibr" rid="B40">Veli&#x10d;kovi&#x107; et al., 2017</xref>). The final attention score highlights the neighboring nodes that should receive more attention to capture the entity embedding. The softmax function can be expressed as:<disp-formula id="e12">
<mml:math id="m54">
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x003D;</mml:mo>
<mml:mi>&#x3c0;</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x003D;</mml:mo>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">f</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">x</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x003D;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>exp</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mi>exp</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(12)</label>
</disp-formula>
</p>
</sec>
<sec id="s2-3-3-2">
<title>2.3.3.2 Entity enhancement layer</title>
<p>To further enhance the interaction between genes and entities, we propose a gene-specific entity enhancement layer. Previous approaches have neglected the effect of multiple entity embeddings on gene richness and overlooked the comprehensive expression of entities and genes. For different genes, KG entities have different amounts of information to describe their properties. For example, <italic>BNIP3</italic> is a well-known tumor suppressor, while <italic>FTO</italic>, as an N6-methyladenosine RNA demethylase, is upregulated in human breast cancer. It has been observed that <italic>FTO</italic> suppresses cell apoptosis by downregulating <italic>BNIP3</italic> (<xref ref-type="bibr" rid="B30">Niu et al., 2019</xref>). Under hypoxic conditions, the mRNA levels of <italic>BNIP3</italic> increase in CHO cell lines, and this effect is mediated by Hif-1<italic>&#x3b1;</italic> (<xref ref-type="bibr" rid="B7">Bruick, 2000</xref>). Therefore, the entity enhancement layer aggregates each entity with genes through an aggregation operation to enhance and enrich the entity embeddings. The enhancement function can be either linear or nonlinear:<disp-formula id="e13">
<mml:math id="m55">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x003D;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>g</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x002B;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
<label>(13)</label>
</disp-formula>
<disp-formula id="e14">
<mml:math id="m56">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x003D;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>g</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x002B;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(14)</label>
</disp-formula>where <italic>W</italic>
<sub>
<italic>e</italic>
</sub> and <italic>b</italic>
<sub>
<italic>e</italic>
</sub> are the trainable weight matrix and bias, and <italic>agg</italic> is a nonlinear activation function.</p>
<p>In this study, we implemented four types of aggregation methods <inline-formula id="inf43">
<mml:math id="m57">
<mml:mi>a</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>g</mml:mi>
<mml:mo>:</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2192;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> are follows:<list list-type="simple">
<list-item>
<p>&#x2022; Sum Aggregator (<xref ref-type="bibr" rid="B42">Wang et al., 2019b</xref>) refers to a process of summing the representation vectors of two entities, and applying a nonlinear transformation to the resulting vector:</p>
</list-item>
</list>
<disp-formula id="e15">
<mml:math id="m58">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x003D;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mo>&#x002B;</mml:mo>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x002B;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(15)</label>
</disp-formula>
<list list-type="simple">
<list-item>
<p>&#x2022; Concat Aggregator (<xref ref-type="bibr" rid="B42">Wang et al., 2019b</xref>) combines the representation vectors of two entities before applying a nonlinear transformation:</p>
</list-item>
</list>
<disp-formula id="e16">
<mml:math id="m59">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x003D;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x002B;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(16)</label>
</disp-formula>
<list list-type="simple">
<list-item>
<p>&#x2022; Pooling Aggregator (<xref ref-type="bibr" rid="B13">Glorot et al., 2011</xref>) calculates the maximum value from multiple vectors within the same dimension and subsequently applies a nonlinear transformation:</p>
</list-item>
</list>
<disp-formula id="e17">
<mml:math id="m60">
<mml:mi>a</mml:mi>
<mml:mi>g</mml:mi>
<mml:msubsup>
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">pool</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x003D;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x002B;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(17)</label>
</disp-formula>
<list list-type="simple">
<list-item>
<p>&#x2022; Top-k Aggregator (<xref ref-type="bibr" rid="B21">Kumar et al., 2009</xref>) efficiently aggregates information from multiple sorted lists of vectors to compute the top <italic>k</italic> objects:</p>
</list-item>
</list>
<disp-formula id="e18">
<mml:math id="m61">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x003D;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>p</mml:mi>
<mml:mtext>_</mml:mtext>
<mml:mi>K</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x002B;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(18)</label>
</disp-formula>
</p>
<p>The function <italic>Top</italic>_<italic>K</italic> (<italic>data</italic>, <italic>k</italic>) extracts the top <italic>k</italic> data values in order. As shown in Eq. <xref ref-type="disp-formula" rid="e18">18</xref>, the top <italic>k</italic> values are taken after sorting the vector <italic>&#x3c3;</italic>(<italic>W</italic> (<italic>e</italic>, <italic>m</italic>) &#x002B; <italic>b</italic>) in descending order.</p>
</sec>
</sec>
<sec id="s2-3-4">
<title>2.3.4 Entity-entity interaction</title>
<p>The entity-entity interaction consists of a discrepancy contrastive layer and an attention aggregator. The former focuses on capturing higher-order connectivity between entities, hierarchically comparing layered information to further improve entity embeddings. The latter performs weighted aggregation of embedding to avoid noise caused by excessive node embedding information, which could otherwise affect prediction results.</p>
<sec id="s2-3-4-1">
<title>2.3.4.1 Discrepancy contrastive layer</title>
<p>Our focus is on incorporating the latent information of neighbors at different distances into the information comparison at each layer, capturing higher-order message passing between entities, and enhancing the representation of entity embeddings through the overall differentiation of hierarchical entities. We introduce the hierarchical modeling capabilities of the model in terms of depth and width.</p>
<p>For depth, we integrate the gene <inline-formula id="inf44">
<mml:math id="m62">
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> and neighborhood information <inline-formula id="inf45">
<mml:math id="m63">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> collected from different depths. By comparing neighbors of different orders in high-order message passing, each node receives potential vector representations from neighboring nodes or further <italic>d</italic>-order neighbors. Then we aggregate them into <inline-formula id="inf46">
<mml:math id="m64">
<mml:mi>a</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>:</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2192;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> to generate the mixed next-order depth embedding <inline-formula id="inf47">
<mml:math id="m65">
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>&#x002B;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula>. Here, we use the Top-<italic>k</italic> Aggregator to aggregate gene representations and their neighborhood information into a single vector.</p>
<p>For width, the feature differences between entities located at different width distances means this wide-layer feature difference plays a critical role. In terms of the training space of the model, entity features at different width levels should be compared (<xref ref-type="bibr" rid="B1">Abu-El-Haija et al., 2019</xref>) so that the model can choose potential information by comparing neighbors at various distances. Therefore, we perform a contrastive mixed operation of neighborhood latent features within different width distances. As the relevance of each layer in the network varies, it is possible for entities to be connected to neighboring nodes with different attributes or labels. We use the width-layer matrix <italic>M</italic>
<sub>
<italic>w</italic>
</sub> to integrate the deep neighborhood information <inline-formula id="inf48">
<mml:math id="m66">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> with different properties in a layer-wise and progressively deeper manner, updating the high-order embedding representation of wide layer entities <inline-formula id="inf49">
<mml:math id="m67">
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
<mml:mo>&#x002B;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula>:<disp-formula id="e19">
<mml:math id="m68">
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
<mml:mo>&#x002B;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x003D;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mfenced close="]" open="[">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(19)</label>
</disp-formula>
<disp-formula id="e20">
<mml:math id="m69">
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>&#x002B;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x003D;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>g</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(20)</label>
</disp-formula>
</p>
</sec>
<sec id="s2-3-4-2">
<title>2.3.4.2 Attention aggregator module</title>
<p>After <italic>l</italic> rounds of discrepancy contrastive layers, we obtain multiple embedding representations of gene <italic>n</italic>. This module uses the gene <italic>n</italic> representation set <inline-formula id="inf50">
<mml:math id="m70">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x003D;</mml:mo>
<mml:mfenced close="}" open="{">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mi>i</mml:mi>
<mml:mo>&#x003D;</mml:mo>
<mml:mn>0,1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>l</mml:mi>
</mml:math>
</inline-formula> to update the embedding of gene <italic>n</italic> uniformly. We use an attention aggregator module that assigns different importance levels to each embedding, avoiding giving each embedding the same weight when aggregating information. For the potential features of the gene <italic>n</italic>, the attention aggregator first learns the attention scores for each embedding. Then, the scores are normalized to derive weight coefficients for the embeddings. Finally, the attention aggregator performs a weighted aggregation on all embedding representations to update the embedding of the gene <italic>n</italic>:<disp-formula id="e21">
<mml:math id="m71">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x003D;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>6</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>tanh</mml:mi>
<mml:mtext>&#x2009;&#x2009;</mml:mtext>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>6</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>g</mml:mi>
<mml:msubsup>
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">pool</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(21)</label>
</disp-formula>
<disp-formula id="e22">
<mml:math id="m72">
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x003D;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>exp</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:mi>exp</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(22)</label>
</disp-formula>
<disp-formula id="e23">
<mml:math id="m73">
<mml:mi>n</mml:mi>
<mml:mo>&#x003D;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>&#x002B;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(23)</label>
</disp-formula>
<disp-formula id="e24">
<mml:math id="m74">
<mml:mi>a</mml:mi>
<mml:mi>g</mml:mi>
<mml:msubsup>
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">pool</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x003D;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x002B;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(24)</label>
</disp-formula>
</p>
<p>where tanh is the nonlinear activation function assigned to the prediction model. The parameters <inline-formula id="inf51">
<mml:math id="m75">
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>6</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> and <italic>W</italic>
<sub>6</sub>, <inline-formula id="inf52">
<mml:math id="m76">
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> are weight vector and weight matrices, respectively; <inline-formula id="inf53">
<mml:math id="m77">
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> is the bias term; and <italic>&#x3c3;</italic> is the <italic>Sigmoid</italic> activation function.</p>
</sec>
</sec>
<sec id="s2-3-5">
<title>2.3.5 Synthetic lethality prediction</title>
<p>After obtaining the final potential embeddings of gene <italic>m</italic> and gene <italic>n</italic>, MPASL combines the two latent features through the prediction function <italic>f</italic>
<sub>
<italic>SL</italic>
</sub> to obtain the final predicted probability that gene <italic>m</italic> and gene <italic>n</italic> are SL relationships, where <italic>f</italic>
<sub>
<italic>SL</italic>
</sub> is the inner product function. <italic>&#x3c3;</italic> is the <italic>Sigmoid</italic> function, which compresses the output to the range between 0 and 1, indicating the probability of the SLs:<disp-formula id="e25">
<mml:math id="m78">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi mathvariant="italic">mn</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mi>&#x03C3;</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
</mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>SL</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
</mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(25)</label>
</disp-formula>
</p>
</sec>
</sec>
<sec id="s2-4">
<title>2.4 Objective function</title>
<p>We now consider the real-valued label function <inline-formula id="inf54">
<mml:math id="m79">
<mml:msub>
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>:</mml:mo>
<mml:mi mathvariant="script">E</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:math>
</inline-formula> on the KG, which is constrained to take a specific value <italic>l</italic>
<sub>
<italic>m</italic>
</sub>(<italic>n</italic>) &#x003D; <italic>y</italic>
<sub>
<italic>mn</italic>
</sub> at node <inline-formula id="inf55">
<mml:math id="m80">
<mml:mi>n</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mo>&#x2286;</mml:mo>
<mml:mi mathvariant="script">E</mml:mi>
</mml:math>
</inline-formula>. If gene <italic>m</italic> is found to be relevant to <italic>n</italic>, then <italic>l</italic>
<sub>
<italic>m</italic>
</sub>(<italic>n</italic>) &#x003D; 1, otherwise <italic>l</italic>
<sub>
<italic>m</italic>
</sub>(<italic>n</italic>) &#x003D; 0. We use label smoothness to act on the supervised signals of regularized edge weights (<xref ref-type="bibr" rid="B41">Wang et al., 2019a</xref>):<disp-formula id="e26">
<mml:math id="m81">
<mml:mi>R</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x003D;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mi>R</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x003D;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mi mathvariant="script">J</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(26)</label>
</disp-formula>where <italic>A</italic>
<sub>
<italic>m</italic>
</sub> aggregates the representation vectors of neighboring entities. The ideal edge weight matrix <italic>A</italic> should reproduce the true relevance labels of each entity while satisfying the smoothness of relevancy labels. We combine the knowledge-aware graph neural network with least squares regularization and use negative sampling during the training process to optimize MPASL. The complete loss function is obtained as:<disp-formula id="e27">
<mml:math id="m82">
<mml:mi mathvariant="script">L</mml:mi>
<mml:mo>&#x003D;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">G</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x003D;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mi mathvariant="script">J</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x003D;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="double-struck">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x223c;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
<mml:mspace width="1em"/>
<mml:mi mathvariant="script">J</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x002B;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:mi mathvariant="script">F</mml:mi>
<mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x002B;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:mi>R</mml:mi>
<mml:mfenced close=")" open="(">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(27)</label>
</disp-formula>
</p>
<p>where the first term <inline-formula id="inf56">
<mml:math id="m83">
<mml:mi mathvariant="script">J</mml:mi>
</mml:math>
</inline-formula> is the cross-entropy loss, <italic>N</italic>
<sup>
<italic>m</italic>
</sup> is the number of negative samples for gene <italic>m</italic>; with <italic>N</italic>
<sup>
<italic>m</italic>
</sup> &#x003D; &#x7c;{<italic>n</italic>: <italic>y</italic>
<sub>
<italic>mn</italic>
</sub> &#x003D; 1}&#x7c;, and <italic>P</italic> is a negative sampling distribution and follows a uniform distribution. The second term is L2 regularization. The third part <italic>R</italic> (&#x22c5;) corresponds to the label smoothness component, which can be viewed as adding the constraint of edge weight <italic>A</italic>. Therefore, <italic>R</italic> (&#x22c5;) serves as a regularization on <italic>A</italic> to assist in learning the edge weights. <italic>&#x3bb;</italic> and <italic>&#x3b3;</italic> are balance hyperparameters.</p>
</sec>
</sec>
<sec id="s3">
<title>3 Experiments and results</title>
<p>We compare the performance of the MPASL with several baseline models to comprehensively evaluate its performance. Additionally, we conducted parameter sensitivity analysis and ablation studies to further investigate the model&#x2019;s performance. The MPASL model was implemented using Python 3.6 and TensorFlow 1.15.0. In the SynLethDB dataset, we split the gene pairs into training, validation, and test sets in a ratio of 7:1:2. We use the area under the ROC curve (AUC) and the area under the precision-recall curve (AUPR) as evaluation metrics to assess the predictive performance. Finally, we present a case study to demonstrate the mechanisms of potential SL interactions between two genes.</p>
<sec id="s3-1">
<title>3.1 Parameter settings</title>
<p>We evaluated our model parameters using 5-fold cross-validation and used a grid search to choose the optimal hyperparameter settings. We tested the following MPASL parameters: batch sizes <inline-formula id="inf57">
<mml:math id="m84">
<mml:mo>&#x2208;</mml:mo>
<mml:mfenced close="}" open="{">
<mml:mrow>
<mml:mn>32,64,128,256,512,1024,2048</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula>, learning rates <inline-formula id="inf58">
<mml:math id="m85">
<mml:mo>&#x2208;</mml:mo>
<mml:mfenced close="}" open="{">
<mml:mrow>
<mml:mn>6</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula>, the entity embedding dimensions <inline-formula id="inf59">
<mml:math id="m86">
<mml:mo>&#x2208;</mml:mo>
<mml:mfenced close="}" open="{">
<mml:mrow>
<mml:mn>8,16,32,64,128,256,512</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula>, the numbers of layers for KG ripple propagation, and the depth and width of discrepancy contrastive layers <inline-formula id="inf60">
<mml:math id="m87">
<mml:mo>&#x2208;</mml:mo>
<mml:mfenced close="}" open="{">
<mml:mrow>
<mml:mn>1,2,3,4,5</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula>, and the ripple preference set sizes <inline-formula id="inf61">
<mml:math id="m88">
<mml:mo>&#x2208;</mml:mo>
<mml:mfenced close="}" open="{">
<mml:mrow>
<mml:mn>4,8,16,32,64</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula>. After these tests, we set the number of KG neighbor samples to 8, initialized the number of embedding dimensions to 128, set the early stopping level to 5 and set the regularization weight 1 &#xd7; 10<sup>&#x2212;8</sup>. <xref ref-type="table" rid="T5">Table 5</xref> provides hyperparameter settings in detail.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>The hyperparameter setting.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Parameter</th>
<th align="center">Setting</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Batch size</td>
<td align="center">512</td>
</tr>
<tr>
<td align="center">Learning rate</td>
<td align="center">6 &#xd7; 10<sup>&#x2212;5</sup>
</td>
</tr>
<tr>
<td align="center">dim</td>
<td align="center">128</td>
</tr>
<tr>
<td align="center">p_hop</td>
<td align="center">2</td>
</tr>
<tr>
<td align="center">depth</td>
<td align="center">2</td>
</tr>
<tr>
<td align="center">width</td>
<td align="center">3</td>
</tr>
<tr>
<td align="center">L2_weight</td>
<td align="center">1 &#xd7; 10<sup>&#x2212;8</sup>
</td>
</tr>
<tr>
<td align="center">LS_weight</td>
<td align="center">1 &#xd7; 10<sup>&#x2212;8</sup>
</td>
</tr>
<tr>
<td align="center">optimizer</td>
<td align="center">Adam</td>
</tr>
<tr>
<td align="center">n_samples</td>
<td align="center">8</td>
</tr>
<tr>
<td align="center">ripple_set_size</td>
<td align="center">8</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-2">
<title>3.2 Comparison with previous studies</title>
<p>To validate the performance of MPASL, we compared our model with several recently proposed baseline methods for SL prediction. These benchmark methods include SL<sup>2</sup>MF, GRSMF, DDGCN, GCATSL, KG4SL, and SLGNN. It is worth noting that the first four methods do not utilize KGs to generate gene embeddings. We used the default settings specified in their original implementations in our tests. Below are brief descriptions of these comparison methods.<list list-type="simple">
<list-item>
<p>(1) SL<sup>2</sup>MF (<xref ref-type="bibr" rid="B23">Liu et al., 2019</xref>) uses logical matrix factorization and further integrates gene similarity based on gene ontology (GO) annotations to predict human SL interactions.</p>
</list-item>
<list-item>
<p>(2) GRSMF (<xref ref-type="bibr" rid="B19">Huang et al., 2019</xref>) is a graph regularized self-representation matrix decomposition model predicting SL interactions from regularized graphs of data from different sources.</p>
</list-item>
<list-item>
<p>(3) DDGCN (<xref ref-type="bibr" rid="B8">Cai et al., 2020</xref>) predicts sparse SL interactions using dual-dropout graph convolutional networks (GCNs).</p>
</list-item>
<list-item>
<p>(4) GCATSL (<xref ref-type="bibr" rid="B24">Long et al., 2021</xref>) performs SL prediction using a graph contextual attention network.</p>
</list-item>
<list-item>
<p>(5) KG4SL (<xref ref-type="bibr" rid="B45">Wang et al., 2021</xref>) represents the first novel SL interaction prediction model based on knowledge graphs and graph neural networks, effectively leveraging rich semantic information encoded in KGs.</p>
</list-item>
<list-item>
<p>(6) SLGNN (<xref ref-type="bibr" rid="B52">Zhu et al., 2023</xref>) is a factor-aware knowledge graph neural network for learning gene embeddings and predicting SL interactions.</p>
</list-item>
</list>
</p>
<p>It is also important to address the potential bias that may arise when there are more positive than negative training pairs. In such cases, many prediction algorithms achieve high performance on the test set by simply manipulating the features of each pair. We observed this situation in SL prediction methods as well. Reliable estimation of prediction error is challenging, especially when the model is uncertain and requires independent test subjects. These test subjects must not participate in model construction or model selection. A more effective approach is to utilize stratified nested cross-validation (<xref ref-type="bibr" rid="B34">Preuer et al., 2018</xref>), where the test set is selected to exclude synthetic lethality gene pairs, denoted as the &#x201c;Leave out synthetic lethality&#x201d; setting. We used a 5-fold nested cross-validation setup in which hyperparameters were selected in the inner loop based on validation error, and then the best performance model for the inner loop was evaluated on the outer test fold to obtain performance estimates that were not affected by hyperparameter selection.</p>
<p>Our model was experimented under two evaluation settings: random cross-validation and stratified cross-validation. The prediction results, denoted as &#x201c;Random CV&#x201d; and &#x201c;Leave out synthetic lethality&#x201d;, are shown in <xref ref-type="table" rid="T6">Table 6</xref>. From the AUC and AUPR scores, our MPASL outperformed the other methods. Specifically, on the SynlethDB dataset, the MPASL model achieved an AUC value of 0.9656 and an AUPR value of 0.9798. In the Leave out synthetic lethality setting, the MPASL model achieved an AUC of 0.8766 and an AUPR of 0.8941. These values surpassed those of other methods. For the Leave out synthetic lethality, compared to the state-of-the-art model SLGNN, MPASL improved performance by 2.73% in AUC and 9.31% in AUPR. These results indicate that our proposed MPASL model had a stronger generalization ability and effectively enhanced the predictive performance of synthetic lethality.</p>
<table-wrap id="T6" position="float">
<label>TABLE 6</label>
<caption>
<p>Performance comparison of MPASL and baselines.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left" rowspan="2">Model</th>
<th align="center" colspan="2">Random CV</th>
<th align="center" colspan="2">Leave out synthetic lethality</th>
</tr>
<tr>
<th align="center">AUC-ROC</th>
<th align="center">AUC-PR</th>
<th align="center">AUC-ROC</th>
<th align="center">AUC-PR</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">SL<sup>2</sup>MF</td>
<td align="center">0.7812 &#xb1; 0.0034</td>
<td align="center">0.8614 &#xb1; 0.0021</td>
<td align="center">0.4604 &#xb1; 0.0045</td>
<td align="center">0.5002 &#xb1; 0.0061</td>
</tr>
<tr>
<td align="left">GRSMF</td>
<td align="center">0.9184 &#xb1; 0.0039</td>
<td align="center">0.9361 &#xb1; 0.0024</td>
<td align="center">0.6951 &#xb1; 0.0037</td>
<td align="center">0.7011 &#xb1; 0.0052</td>
</tr>
<tr>
<td align="left">DDGCN</td>
<td align="center">0.8491 &#xb1; 0.0106</td>
<td align="center">0.8998 &#xb1; 0.0056</td>
<td align="center">0.6402 &#xb1; 0.0335</td>
<td align="center">0.6352 &#xb1; 0.0334</td>
</tr>
<tr>
<td align="left">GCATSL</td>
<td align="center">0.9122 &#xb1; 0.0108</td>
<td align="center">0.9175 &#xb1; 0.0078</td>
<td align="center">0.7056 &#xb1; 0.0292</td>
<td align="center">0.7085 &#xb1; 0.0288</td>
</tr>
<tr>
<td align="left">KG4SL</td>
<td align="center">0.9446 &#xb1; 0.0009</td>
<td align="center">0.9544 &#xb1; 0.0012</td>
<td align="center">0.7272 &#xb1; 0.0005</td>
<td align="center">0.7623 &#xb1; 0.0003</td>
</tr>
<tr>
<td align="left">SLGNN</td>
<td align="center">0.9620 &#xb1; 0.0023</td>
<td align="center">0.9703 &#xb1; 0.0019</td>
<td align="center">0.8493 &#xb1; 0.0046</td>
<td align="center">0.8010 &#xb1; 0.0057</td>
</tr>
<tr>
<td align="left">MPASL</td>
<td align="center">
<bold>0.9656</bold> &#xb1; 0.0049</td>
<td align="center">
<bold>0.9798</bold> &#xb1; 0.0032</td>
<td align="center">
<bold>0.8766</bold> &#xb1; 0.0107</td>
<td align="center">
<bold>0.8941</bold> &#xb1; 0.0042</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The superior predictive performance of MPASL is attributable to several key factors. First, MPASL enriches gene representations by leveraging existing SL interaction data and incorporating all relevant entities present in the KG. It effectively integrates KG hierarchy propagation and KG ripple propagation into gene embeddings enhancing the gene embeddings. MPASL also incorporates embeddings of relevant entities, weights the neighboring entities and emphasizes the most important entities, thus enriching the representation. In the process of gathering KG information, MPASL considers and blends hierarchical information and performs hierarchical contrast and aggregation, enabling the modeling of nonlinear features and higher-order interactions. This facilitates the integration of various higher-order correlation information associated with genes and neighboring entities, thereby capturing and representing the intricate interactions among gene embeddings more effectively.</p>
</sec>
<sec id="s3-3">
<title>3.3 Parameter sensitivity analysis</title>
<p>To gain a deeper understanding of MPASL, we researched the effect of different components on the model&#x2019;s performance. First, we examined the effect of depth and width in the discrepancy contrastive layer. Then, we explored the influence of different entity embedding dimensions. Next, we studied the effect of preference set sampling size for KG ripple propagation and the effect of attention aggregator mechanism. All of the following studies were conducted based on the &#x201c;Leave out synthetic lethality&#x201d; setting.</p>
<sec id="s3-3-1">
<title>3.3.1 Effect of the depth and width of discrepancy contrastive layer</title>
<p>We evaluated the impact of the discrepancy contrastive layer in MPASL by varying its depth and width. As shown in <xref ref-type="fig" rid="F2">Figures 2A, B</xref>, we conducted experiments within the range of {1, 2, 3, 4, 5}.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>
<bold>(A)</bold> Impact of depths of discrepancy contrastive layer <bold>(B)</bold> Impact of widths of discrepancy contrastive layer.</p>
</caption>
<graphic xlink:href="fphar-15-1398231-g002.tif"/>
</fig>
<p>The results indicate that the performance is optimal when the depth and width are 2 and 3, respectively. Specifically, sometimes relying solely on first-order neighboring entities is insufficient to fully explore the correlations and dependencies between entities. When the depth or width is increased to 4 or 5 layers, more noise is introduced into the model. Therefore, it is necessary to balance the dependence of positive signals on distance and the noise of negative signals to find an appropriate balance in terms of depth and width allows for the exploration of potential embeddings of nodes as comprehensively as possible.</p>
</sec>
<sec id="s3-3-2">
<title>3.3.2 Effect of the number of embedding dimensions</title>
<p>We explored the effects of the number of embedding dimensions on the performance of MPASL. As shown in <xref ref-type="fig" rid="F3">Figure 3</xref>, we observed that the AUC and AUPR were maximized with 128 embedding dimensions. At larger numbers, the AUC and AUPR values gradually declined. In this result, it is indicates that within a particular range, increasing the embedding dimension effectively encodes more information from the KG, leading to improved performance in terms of AUC and AUPR. However, exceeding the optimal embedding dimension results in overfitting, leading to a decrease in predictive performance. Therefore, we observed an initial upward trend followed by a decline in the AUC and AUPR scores as the embedding dimension continued to increase.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Embedding dimension of AUC and AUPR.</p>
</caption>
<graphic xlink:href="fphar-15-1398231-g003.tif"/>
</fig>
<p>Based on these findings, the key is to strike a balance when selecting the embedding dimensions for MPASL. Setting the embedding dimensions to 128 appears to be the optimal choice for capturing essential information from the KG while preserving generalization ability. This finding emphasizes the importance of appropriately adjusting the embedding dimension to achieve optimal performance.</p>
</sec>
<sec id="s3-3-3">
<title>3.3.3 Effect of the KG ripple preference set size</title>
<p>We investigated the impact of different sample sizes for the preference set used in the KG ripple propagation of MPASL. We varied the sample sizes within the range of {4, 8, 16, 32, 64} and analyzed their effects on the model performance. The results of the analysis are shown in <xref ref-type="fig" rid="F4">Figure 4</xref>.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Effect of different ripple preference set size.</p>
</caption>
<graphic xlink:href="fphar-15-1398231-g004.tif"/>
</fig>
<p>The results indicate that MPASL performance is optimal when the sample size of the preference set is set to 8. This means that a smaller sample size of the preference set still allows MPASL to capture sufficient information and effectively enhance gene embeddings with a limited number of known SL interactions for genes. As the preference set is further expanded, entities with lower relevance to the genes start to be included, leading to inaccurate gene embeddings and a decrease in performance. Therefore, selecting an appropriate sample size for the preference set, striking a balance between capturing sufficient relevant information as well as avoiding the inclusion of irrelevant entities, maximizes performance.</p>
</sec>
<sec id="s3-3-4">
<title>3.3.4 Effect of aggregators</title>
<p>We evaluated the effect of different attention aggregators in MPASL: Concat, Sum, Pool, and Top-k. These are labeled as MPASL-con, MPASL-sum, MPASL-pool, and MPASL-top, respectively, in our results. As shown in <xref ref-type="table" rid="T7">Table 7</xref>, the model achieved best predictive performance when using the Top-k aggregator.</p>
<table-wrap id="T7" position="float">
<label>TABLE 7</label>
<caption>
<p>Effect of different attention aggregators.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Aggregators</th>
<th align="left">AUC-ROC</th>
<th align="left">AUC-PR</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">MPASL-con</td>
<td align="left">0.8610</td>
<td align="left">0.8767</td>
</tr>
<tr>
<td align="left">MPASL-sum</td>
<td align="left">0.8627</td>
<td align="left">0.8832</td>
</tr>
<tr>
<td align="left">MPASL-pool</td>
<td align="left">0.8688</td>
<td align="left">0.8857</td>
</tr>
<tr>
<td align="left">MPASL-top</td>
<td align="left">
<bold>0.8766</bold>
</td>
<td align="left">
<bold>0.8941</bold>
</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s3-4">
<title>3.4 Ablation study</title>
<p>We verified the influence of the important components on the performance of MPASL through an ablation study and designed the following four of its variants. The following study was conducted based on five-fold random cross-validation and stratified nested cross-validation settings, expressed as &#x201c;Random CV&#x201d; and &#x201c;Leave out synthetic lethality&#x201d; respectively.<list list-type="simple">
<list-item>
<p>(1) MPASL<sub>w/o RP</sub>: MPASL without the knowledge graph ripple propagation.</p>
</list-item>
<list-item>
<p>(2) MPASL<sub>w/o EL(e)</sub>: MPASL without the entity enhancement layer to update entity embedding representation.</p>
</list-item>
<list-item>
<p>(3) MPASL<sub>w/o EL(att)</sub>: MPASL without the attention aggregator to allocate weight information for entity embedding representation of different layers.</p>
</list-item>
<list-item>
<p>(4) MPASL<sub>w/o R(E)</sub>: MPASL after deleting the entity and its associated types of relationships.</p>
</list-item>
</list>
</p>
<p>We compared MPASL with several of its variants, and the results are given in <xref ref-type="table" rid="T8">Table 8</xref>. The performance achieved by the model in different cases can be summarized as follows:<list list-type="simple">
<list-item>
<p>&#x2022; A key component of MPASL is the knowledge graph ripple propagation. We introduce the ripple propagation in order to capture the preferences of existing SL interactions to enrich the representation of genes. MPASL<sub>w/o RP</sub>, with the proposed KG ripple propagation removed, achieved significantly lower scores. This is because it only considered entity embedding, ignoring the set of known SL interactions of genes and the preferences of genes when aggregating entities and relationships in KG. This highlights the importance of the KG ripple propagation in our SL prediction.</p>
</list-item>
<list-item>
<p>&#x2022; To enhance gene-specific entity information, we used the entity enhancement layer to enrich entity representations. MPASL without the entity enhancement layer, MPASL<sub>w/o EL(e)</sub> was significantly outperformed by MPASL. <xref ref-type="table" rid="T8">Table 8</xref> confirms that the entity enhancement layer improves gene-specific entity information and contributes to enhanced performance.</p>
</list-item>
<list-item>
<p>&#x2022; Removing the attention aggregator, MPASL<sub>w/o EL(att)</sub>, also worsened performance compared to MPASL. compared to MPASL. <xref ref-type="table" rid="T8">Table 8</xref> shows the importance of the attention aggregator in capturing relatively important entities and relationships from the KG, aiding in determining the weights of neighboring messages.</p>
</list-item>
<list-item>
<p>&#x2022; The performance of MPASL<sub>w/o R(E)</sub> with the removal of a particular entity and associated relationship also decreases compared to MPASL. The experimental results indicated that entities and relations in SynLethKG are helpful for SL prediction.</p>
</list-item>
</list>
</p>
<table-wrap id="T8" position="float">
<label>TABLE 8</label>
<caption>
<p>Performance comparison between different variants.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left" rowspan="2">Methods</th>
<th align="center" colspan="2">Random CV</th>
<th align="center" colspan="2">Leave out synthetic lethality</th>
</tr>
<tr>
<th align="center">AUC-ROC</th>
<th align="center">AUC-PR</th>
<th align="center">AUC-ROC</th>
<th align="center">AUC-PR</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">MPASL<sub>w/o&#xa0;RP</sub>
</td>
<td align="center">0.9395</td>
<td align="center">0.9423</td>
<td align="center">0.5614</td>
<td align="center">0.5797</td>
</tr>
<tr>
<td align="left">MPASL<sub>w/o&#xa0;EL(e)</sub>
</td>
<td align="center">0.9476</td>
<td align="center">0.9556</td>
<td align="center">0.8172</td>
<td align="center">0.8274</td>
</tr>
<tr>
<td align="left">MPASL<sub>w/o&#xa0;EL(att)</sub>
</td>
<td align="center">0.9543</td>
<td align="center">0.9674</td>
<td align="center">0.8271</td>
<td align="center">0.8346</td>
</tr>
<tr>
<td align="left">MPASL<sub>w/o&#xa0;R(E)</sub>
</td>
<td align="center">0.9616</td>
<td align="center">0.9743</td>
<td align="center">0.8601</td>
<td align="center">0.8739</td>
</tr>
<tr>
<td align="left">MPASL</td>
<td align="center">
<bold>0.9656</bold>
</td>
<td align="center">
<bold>0.9798</bold>
</td>
<td align="center">
<bold>0.8766</bold>
</td>
<td align="center">
<bold>0.8941</bold>
</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-5">
<title>3.5 Case study</title>
<p>To further examine the performance of MPASL, a case study was conducted using the SynlethDB dataset. The training samples included all observed known SL interactions, we used the training model to predict the SL status of unknown gene pairs. Unknown gene pairs were classified based on their prediction scores, and literature evidence was sought in the biomedical literature to support the predictions. We specifically focused on SL pairs involving the cancer gene <italic>KRAS</italic>. <italic>KRAS</italic> is one of the most widely screened genes for SL interactions and it ranks among the most frequently mutated genes in humans, particularly in cases of cancer (<xref ref-type="bibr" rid="B12">Downward, 2015</xref>). It is also a highly prioritized therapeutic target due to its involvement in inducing cell stasis, apoptosis, and DNA repair. In particular, we studied the top 20&#xa0;SL pairs associated with <italic>KRAS</italic> as shown in <xref ref-type="table" rid="T9">Table 9</xref>. Among these SL gene pairs, we selected the <italic>KRAS</italic>-<italic>RAD50</italic> gene pair from the test data for further analysis. The protein encoded by the <italic>RAD50</italic> gene plays a crucial role in repairing DNA double-strand breaks. It interacts with <italic>MRE11</italic> and <italic>NBS1</italic> to form a complex. This complex binds to DNA and displays multiple enzymatic activities that are essential for functions such as non-homologous end joining, DNA double-strand break repair, activation of cell cycle checkpoints, maintenance of telomeres, and facilitation of meiotic recombination. This highlights the crucial role of these genes in cell growth and vitality, making it reasonable to predict their SL relationship for cancer therapeutics. In the case study, the predicted result for the <italic>KRAS</italic>-<italic>RAD50</italic> gene pair aligned with the known labels, demonstrating the accurate predictive ability of MPASL for SL pairs and emphasizing the potential therapeutic significance of the predicted <italic>KRAS</italic>-<italic>RAD50</italic> SL pair in cancer treatment.</p>
<table-wrap id="T9" position="float">
<label>TABLE 9</label>
<caption>
<p>Top Synthetic lethality gene pairs containing KRAS predicted by MPASL.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Gene 1</th>
<th align="left">Gene 2</th>
<th align="left">PubMed ID</th>
<th align="left">Source</th>
<th align="left">Cell line</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<italic>KRAS</italic>
</td>
<td align="left">
<italic>SCARF1</italic>
</td>
<td align="left">19490893</td>
<td align="left">GenomeRNAi</td>
<td align="left">DLD-1</td>
</tr>
<tr>
<td align="left">
<italic>KRAS</italic>
</td>
<td align="left">
<italic>VDAC1</italic>
</td>
<td align="left">17568748</td>
<td align="left">Synlethality</td>
<td align="left">Human lung cancer</td>
</tr>
<tr>
<td align="left">
<italic>KRAS</italic>
</td>
<td align="left">
<italic>IPMK</italic>
</td>
<td align="left">27655641</td>
<td align="left">RNAi Screen</td>
<td align="left">NA</td>
</tr>
<tr>
<td align="left">
<italic>KRAS</italic>
</td>
<td align="left">
<italic>ZNF200</italic>
</td>
<td align="left">24104479</td>
<td align="left">Text Mining</td>
<td align="left">COAD</td>
</tr>
<tr>
<td align="left">
<italic>KRAS</italic>
</td>
<td align="left">
<italic>GRK3</italic>
</td>
<td align="left">24104479</td>
<td align="left">Text Mining</td>
<td align="left">COAD</td>
</tr>
<tr>
<td align="left">
<italic>KRAS</italic>
</td>
<td align="left">
<italic>NUDT9</italic>
</td>
<td align="left">28700943</td>
<td align="left">High Throughput</td>
<td align="left">NA</td>
</tr>
<tr>
<td align="left">
<italic>KRAS</italic>
</td>
<td align="left">
<italic>SNRPD3</italic>
</td>
<td align="left">24104479</td>
<td align="left">Text Mining</td>
<td align="left">COAD</td>
</tr>
<tr>
<td align="left">
<italic>KRAS</italic>
</td>
<td align="left">
<italic>RAD50</italic>
</td>
<td align="left">24104479</td>
<td align="left">Text Mining</td>
<td align="left">COAD</td>
</tr>
<tr>
<td align="left">
<italic>KRAS</italic>
</td>
<td align="left">
<italic>PARP1</italic>
</td>
<td align="left">20976469</td>
<td align="left">Text Mining</td>
<td align="left">cancer_D009369</td>
</tr>
<tr>
<td align="left">
<italic>KRAS</italic>
</td>
<td align="left">
<italic>PCK1</italic>
</td>
<td align="left">27655641</td>
<td align="left">RNAi Screen</td>
<td align="left">NA</td>
</tr>
<tr>
<td align="left">
<italic>KRAS</italic>
</td>
<td align="left">
<italic>SNRPD3</italic>
</td>
<td align="left">24104479</td>
<td align="left">Text Mining</td>
<td align="left">COAD</td>
</tr>
<tr>
<td align="left">
<italic>KRAS</italic>
</td>
<td align="left">
<italic>RPL10</italic>
</td>
<td align="left">28700943</td>
<td align="left">High Throughput</td>
<td align="left">NA</td>
</tr>
<tr>
<td align="left">
<italic>KRAS</italic>
</td>
<td align="left">
<italic>VGLL2</italic>
</td>
<td align="left">19490893</td>
<td align="left">GenomeRNAi</td>
<td align="left">DLD-1</td>
</tr>
<tr>
<td align="left">
<italic>KRAS</italic>
</td>
<td align="left">
<italic>TOB1</italic>
</td>
<td align="left">24104479</td>
<td align="left">Text Mining</td>
<td align="left">COAD</td>
</tr>
<tr>
<td align="left">
<italic>KRAS</italic>
</td>
<td align="left">
<italic>PCYT2</italic>
</td>
<td align="left">24104479</td>
<td align="left">Text Mining</td>
<td align="left">COAD</td>
</tr>
<tr>
<td align="left">
<italic>KRAS</italic>
</td>
<td align="left">
<italic>RPL7A</italic>
</td>
<td align="left">24104479</td>
<td align="left">Text Mining</td>
<td align="left">COAD</td>
</tr>
<tr>
<td align="left">
<italic>KRAS</italic>
</td>
<td align="left">
<italic>DGKA</italic>
</td>
<td align="left">27655641</td>
<td align="left">RNAi Screen</td>
<td align="left">NA</td>
</tr>
<tr>
<td align="left">
<italic>KRAS</italic>
</td>
<td align="left">
<italic>TRIB3</italic>
</td>
<td align="left">27655641</td>
<td align="left">RNAi Screen</td>
<td align="left">NA</td>
</tr>
<tr>
<td align="left">
<italic>KRAS</italic>
</td>
<td align="left">
<italic>STARD10</italic>
</td>
<td align="left">19490893</td>
<td align="left">GenomeRNAi</td>
<td align="left">DLD-1</td>
</tr>
<tr>
<td align="left">
<italic>KRAS</italic>
</td>
<td align="left">
<italic>MSL2</italic>
</td>
<td align="left">19490893</td>
<td align="left">GenomeRNAi</td>
<td align="left">DLD-1</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>
<xref ref-type="table" rid="T9">Table 9</xref> consists of five columns. The first two columns represent the predicted genes that have an SL relationship, and the third column provides the PubMed ID of publications supporting the prediction. The fourth column presents the specific evidence or rationale behind each predicted SL interaction. Finally, the last column indicates the specific cell lines where the SL interaction has been observed.</p>
<p>
<xref ref-type="fig" rid="F5">Figure 5</xref> illustrates the enrichment analysis results of the <italic>KRAS</italic> SL gene pathway. It highlights several important biological functions, including rRNA processing, protein phosphorylation, cellular response to DNA damage stimulus and histone modification. These enriched biological functions are closely associated with the expression of the <italic>KRAS</italic> gene and its impact on cellular proliferation or death. rRNA serves as the main component of ribosomes that synthesize proteins in cells. The proteins and enzymes encoded by genes are involved in the synthesis and processing of rRNA, regulating and promoting the maturation of rRNA. Gene expression levels and regulation can also affect the rate and efficiency of rRNA synthesis and processing. Protein phosphorylation is a vital regulatory mechanism involved in modulating diverse cellular signaling pathways. Consequently, protein kinases and phosphatases have emerged as significant targets for the development of therapeutic drugs. The cellular response to DNA damage stimulus necessitates the coordinated action of multiple DNA repair pathways. Exploiting the specific dependency of tumor cells on certain DNA repair pathways forms the basis for developing synthetic lethality-based anti-cancer research approaches. Histones contribute to maintaining DNA structure, safeguarding genetic information, and regulating gene expression, and the imbalance of histone modifications is highly correlated with tumor initiation and progression. <italic>KRAS</italic> plays a critical role in these processes (<xref ref-type="bibr" rid="B28">Man&#x10d;ek-Keber et al., 2012</xref>; <xref ref-type="bibr" rid="B6">Brubaker et al., 2019</xref>). This enrichment analysis of the <italic>KRAS</italic> synthetic lethality gene pathway validates the predictive capability of MPASL and offers greater insight into the underlying mechanisms behind synthetic lethality.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>SL gene pathway enrichment analysis of KRAS.</p>
</caption>
<graphic xlink:href="fphar-15-1398231-g005.tif"/>
</fig>
</sec>
</sec>
<sec id="s4" sec-type="conclusion">
<title>4 Conclusion</title>
<p>In recent years, synthetic lethality has been successfully used in targeted therapy of tumors and plays an important role in targeted cancer therapy. In this study, we propose a novel SL interaction prediction model called MPASL. Based on known gene information, MPASL uses features from existing SL interaction preferences to update the gene embeddings. It also incorporates gene-entity interaction and entity-entity interaction to enrich entity embedding representation from the KG. It considers inter-layer entity comparisons and gene-related labels to better explore gene representations, stabilize the learning process on the KG, and enhance the predictive ability of the model. The experimental results show MPASL outperforms existing methods.</p>
<p>Pre-training strategies may help improve model performance and interpretability. Therefore, our future work will explore pre-training techniques that automatically learn features to help solve problems such as prior knowledge to extract high-quality gene embedding representations.</p>
</sec>
</body>
<back>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/Supplementary material, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s6">
<title>Author contributions</title>
<p>GZ: Conceptualization, Funding acquisition, Methodology, Project administration, Software, Supervision, Writing&#x2013;original draft, Formal Analysis, Visualization, Writing&#x2013;review and editing. YC: Conceptualization, Formal Analysis, Methodology, Visualization, Writing&#x2013;original draft, Writing&#x2013;review and editing, Data curation. CY: Conceptualization, Funding acquisition, Methodology, Supervision, Writing&#x2013;review and editing. JW: Formal Analysis, Methodology, Supervision, Writing&#x2013;review and editing. WL: Writing&#x2013;review and editing. JL: Writing&#x2013;review and editing. HL: Conceptualization, Formal Analysis, Funding acquisition, Methodology, Supervision, Writing&#x2013;review and editing.</p>
</sec>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This work was supported by the National Natural Science Foundation of China (Nos. 62006070, 61802113); and the Science and Technology Development Plan Project of Henan Province (No. 222102210238). China Postdoctoral Science Foundation (No. 2020M672212).</p>
</sec>
<ack>
<p>GZ conceived and designed the algorithm and analysis. GZ and YC gathered all the data, designed the study, conduct experiments, and draft manuscripts. GZ, YC, CY, HL, and JW contributed to results analysis and discussions, and gave the final approval of the version to be published. WL and JL supervised the study, revised the manuscript. And we thank LetPub (<ext-link ext-link-type="uri" xlink:href="http://www.letpub.com">www.letpub.com</ext-link>) for its linguistic assistance during the preparation of this manuscript.</p>
</ack>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s9" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Abu-El-Haija</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Perozzi</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Kapoor</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Alipourfard</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Lerman</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Harutyunyan</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). &#x201c;<article-title>Mixhop: higher-order graph convolutional architectures via sparsified neighborhood mixing</article-title>,&#x201d; in <source>International conference on machine learning</source> (<publisher-loc>China</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>21</fpage>&#x2013;<lpage>29</lpage>.</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barab&#xe1;si</surname>
<given-names>A.-L.</given-names>
</name>
<name>
<surname>Gulbahce</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Loscalzo</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Network medicine: a network-based approach to human disease</article-title>. <source>Nat. Rev. Genet.</source> <volume>12</volume>, <fpage>56</fpage>&#x2013;<lpage>68</lpage>. <pub-id pub-id-type="doi">10.1038/nrg2918</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bartz</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Burchard</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Imakura</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Martin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Palmieri</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2006</year>). <article-title>Small interfering rna screens reveal enhanced cisplatin cytotoxicity in tumor cells having both brca network and tp53 disruptions</article-title>. <source>Mol. Cell. Biol.</source> <volume>26</volume>, <fpage>9377</fpage>&#x2013;<lpage>9386</lpage>. <pub-id pub-id-type="doi">10.1128/MCB.01229-06</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Blank</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X. J.</given-names>
</name>
<name>
<surname>Cosmopoulos</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Bouck</surname>
<given-names>D. C.</given-names>
</name>
<name>
<surname>Garcia</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Bernard</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Novel dna damage checkpoints mediating cell death induced by the nedd8-activating enzyme inhibitor mln4924</article-title>. <source>Cancer Res.</source> <volume>73</volume>, <fpage>225</fpage>&#x2013;<lpage>234</lpage>. <pub-id pub-id-type="doi">10.1158/0008-5472.CAN-12-1729</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Boone</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bussey</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Andrews</surname>
<given-names>B. J.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Exploring genetic interactions and networks with yeast</article-title>. <source>Nat. Rev. Genet.</source> <volume>8</volume>, <fpage>437</fpage>&#x2013;<lpage>449</lpage>. <pub-id pub-id-type="doi">10.1038/nrg2085</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brubaker</surname>
<given-names>D. K.</given-names>
</name>
<name>
<surname>Paulo</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Sheth</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Poulin</surname>
<given-names>E. J.</given-names>
</name>
<name>
<surname>Popow</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Joughin</surname>
<given-names>B. A.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Proteogenomic network analysis of context-specific kras signaling in mouse-to-human cross-species translation</article-title>. <source>Cell. Syst.</source> <volume>9</volume>, <fpage>258</fpage>&#x2013;<lpage>270</lpage>. <pub-id pub-id-type="doi">10.1016/j.cels.2019.07.006</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bruick</surname>
<given-names>R. K.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>Expression of the gene encoding the proapoptotic nip3 protein is induced by hypoxia</article-title>. <source>Proc. Natl. Acad. Sci.</source> <volume>97</volume>, <fpage>9082</fpage>&#x2013;<lpage>9087</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.97.16.9082</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cai</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hao</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Dual-dropout graph convolutional network for predicting synthetic lethality in human cancers</article-title>. <source>Bioinformatics</source> <volume>36</volume>, <fpage>4458</fpage>&#x2013;<lpage>4465</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa211</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chang</surname>
<given-names>J.-G.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>C.-C.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y.-Y.</given-names>
</name>
<name>
<surname>Che</surname>
<given-names>T.-F.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Y.-S.</given-names>
</name>
<name>
<surname>Yeh</surname>
<given-names>K.-T.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Uncovering synthetic lethal interactions for therapeutic targets and predictive markers in lung adenocarcinoma</article-title>. <source>Oncotarget</source> <volume>7</volume>, <fpage>73664</fpage>&#x2013;<lpage>73680</lpage>. <pub-id pub-id-type="doi">10.18632/oncotarget.12046</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dai</surname>
<given-names>Z.-T.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Xiang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.-M.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Integrated tcga and geo analysis showed that smad7 is an independent prognostic factor for lung adenocarcinoma</article-title>. <source>Medicine</source> <volume>99</volume>, <fpage>e22861</fpage>. <pub-id pub-id-type="doi">10.1097/MD.0000000000022861</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Das</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Camphausen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Shankavaram</surname>
<given-names>U.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Discoversl: an r package for multi-omic data driven prediction of synthetic lethality in cancers</article-title>. <source>Bioinformatics</source> <volume>35</volume>, <fpage>701</fpage>&#x2013;<lpage>702</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty673</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Downward</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Ras synthetic lethal screens revisited: still seeking the elusive prize?</article-title> <source>Clin. Cancer Res.</source> <volume>21</volume>, <fpage>1802</fpage>&#x2013;<lpage>1809</lpage>. <pub-id pub-id-type="doi">10.1158/1078-0432.CCR-14-2180</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Glorot</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Bordes</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2011</year>). &#x201c;<article-title>Deep sparse rectifier neural networks</article-title>,&#x201d; in <conf-name>Proceedings of the Fourteenth International Conference on Artificial Intelligence and Statistics</conf-name>, <conf-loc>USA</conf-loc>, <conf-date>11-13 April 2011</conf-date> (<publisher-name>JMLR Workshop and Conference Proceedings</publisher-name>), <fpage>315</fpage>&#x2013;<lpage>323</lpage>.</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gregory</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Phang</surname>
<given-names>T. L.</given-names>
</name>
<name>
<surname>Neviani</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Alvarez-Calderon</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Eide</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>O&#x2019;Hare</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>Wnt/ca2&#x002B;/nfat signaling maintains survival of ph&#x002B; leukemia cells upon inhibition of bcr-abl</article-title>. <source>Cancer Cell.</source> <volume>18</volume>, <fpage>74</fpage>&#x2013;<lpage>87</lpage>. <pub-id pub-id-type="doi">10.1016/j.ccr.2010.04.025</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Han</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Jeng</surname>
<given-names>E. E.</given-names>
</name>
<name>
<surname>Hess</surname>
<given-names>G. T.</given-names>
</name>
<name>
<surname>Morgens</surname>
<given-names>D. W.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bassik</surname>
<given-names>M. C.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Synergistic drug combinations for cancer identified in a crispr screen for pairwise genetic interactions</article-title>. <source>Nat. Biotechnol.</source> <volume>35</volume>, <fpage>463</fpage>&#x2013;<lpage>474</lpage>. <pub-id pub-id-type="doi">10.1038/nbt.3834</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hanahan</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Weinberg</surname>
<given-names>R. A.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Hallmarks of cancer: the next generation</article-title>. <source>Cell.</source> <volume>144</volume>, <fpage>646</fpage>&#x2013;<lpage>674</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2011.02.013</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Prediction of synthetic lethal interactions in human cancers using multi-view graph auto-encoder</article-title>. <source>IEEE J. Biomed. Health Inf.</source> <volume>25</volume>, <fpage>4041</fpage>&#x2013;<lpage>4051</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2021.3079302</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hartwell</surname>
<given-names>L. H.</given-names>
</name>
<name>
<surname>Szankasi</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Roberts</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Murray</surname>
<given-names>A. W.</given-names>
</name>
<name>
<surname>Friend</surname>
<given-names>S. H.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>Integrating genetic approaches into the discovery of anticancer drugs</article-title>. <source>Science</source> <volume>278</volume>, <fpage>1064</fpage>&#x2013;<lpage>1068</lpage>. <pub-id pub-id-type="doi">10.1126/science.278.5340.1064</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Ou-Yang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Predicting synthetic lethal interactions in human cancers using graph regularized self-representative matrix factorization</article-title>. <source>BMC Bioinforma.</source> <volume>20</volume>, <fpage>657</fpage>&#x2013;<lpage>658</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-019-3197-3</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Iglehart</surname>
<given-names>J. D.</given-names>
</name>
<name>
<surname>Silver</surname>
<given-names>D. P.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Synthetic lethality--a new direction in cancer-drug development</article-title>. <source>cancer-drug Dev.</source> <volume>361</volume>, <fpage>189</fpage>&#x2013;<lpage>191</lpage>. <pub-id pub-id-type="doi">10.1056/NEJMe0903044</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Kumar</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Punera</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Suel</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Vassilvitskii</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2009</year>). &#x201c;<article-title>Top-k aggregation using intersections of ranked inputs</article-title>,&#x201d; in <conf-name>Proceedings of the Second ACM International Conference on Web Search and Data Mining</conf-name>, <conf-loc>USA</conf-loc>, <conf-date>February 5 - 9, 2018</conf-date> (<publisher-name>IEEE</publisher-name>), <fpage>222</fpage>&#x2013;<lpage>231</lpage>.</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liany</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Jeyasekharan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Rajan</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Predicting synthetic lethal interactions using heterogeneous data sources</article-title>. <source>Bioinformatics</source> <volume>36</volume>, <fpage>2209</fpage>&#x2013;<lpage>2216</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz893</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.-L.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Sl 2 mf: predicting synthetic lethality in human cancers via logistic matrix factorization</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinforma.</source> <volume>17</volume>, <fpage>748</fpage>&#x2013;<lpage>757</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2019.2909908</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Long</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kwoh</surname>
<given-names>C. K.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Graph contextualized attention network for predicting synthetic lethality in human cancers</article-title>. <source>Bioinformatics</source> <volume>37</volume>, <fpage>2432</fpage>&#x2013;<lpage>2440</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btab110</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Qian</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Miao</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>The integrative method based on the module-network for identifying driver genes in cancer subtypes</article-title>. <source>Molecules</source> <volume>23</volume>, <fpage>183</fpage>. <pub-id pub-id-type="doi">10.3390/molecules23020183</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>frdriver: a functional region driver identification for protein sequence</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinforma.</source> <volume>18</volume>, <fpage>1773</fpage>&#x2013;<lpage>1783</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2020.3020096</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Luo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Emanuele</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Creighton</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Schlabach</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Westbrook</surname>
<given-names>T. F.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>A genome-wide rnai screen identifies multiple synthetic lethal interactions with the ras oncogene</article-title>. <source>Cell.</source> <volume>137</volume>, <fpage>835</fpage>&#x2013;<lpage>848</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2009.05.006</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Man&#x10d;ek-Keber</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ben&#x10d;ina</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Japelj</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Panter</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Andr&#xe4;</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Brandenburg</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Marcks as a negative regulator of lipopolysaccharide signaling</article-title>. <source>J. Immunol.</source> <volume>188</volume>, <fpage>3893</fpage>&#x2013;<lpage>3902</lpage>. <pub-id pub-id-type="doi">10.4049/jimmunol.1003605</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mohamed</surname>
<given-names>S. K.</given-names>
</name>
<name>
<surname>Nov&#xe1;&#x10d;ek</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Nounu</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Discovering protein drug targets using knowledge graph embeddings</article-title>. <source>Bioinformatics</source> <volume>36</volume>, <fpage>603</fpage>&#x2013;<lpage>610</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz600</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Niu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Rna n6-methyladenosine demethylase fto promotes breast tumor progression through inhibiting bnip3</article-title>. <source>Mol. Cancer</source> <volume>18</volume>, <fpage>46</fpage>&#x2013;<lpage>16</lpage>. <pub-id pub-id-type="doi">10.1186/s12943-019-1004-4</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Oughtred</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Stark</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Breitkreutz</surname>
<given-names>B.-J.</given-names>
</name>
<name>
<surname>Rust</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Boucher</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>The biogrid interaction database: 2019 update</article-title>. <source>Nucleic Acids Res.</source> <volume>47</volume>, <fpage>D529-D541</fpage>&#x2013;<lpage>D541</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gky1079</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paladugu</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ray</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Raval</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Mining protein networks for synthetic genetic interactions</article-title>. <source>BMC Bioinforma.</source> <volume>9</volume>, <fpage>426</fpage>&#x2013;<lpage>514</lpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-9-426</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pandey</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>A. N.</given-names>
</name>
<name>
<surname>Myers</surname>
<given-names>C. L.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>An integrative multi-network and multi-classifier approach to predict genetic interactions</article-title>. <source>PLoS Comput. Biol.</source> <volume>6</volume>, <fpage>e1000928</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1000928</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Preuer</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Lewis</surname>
<given-names>R. P.</given-names>
</name>
<name>
<surname>Hochreiter</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bender</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bulusu</surname>
<given-names>K. C.</given-names>
</name>
<name>
<surname>Klambauer</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Deepsynergy: predicting anti-cancer drug synergy with deep learning</article-title>. <source>Bioinformatics</source> <volume>34</volume>, <fpage>1538</fpage>&#x2013;<lpage>1546</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btx806</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Suhail</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Y.-y.</given-names>
</name>
<name>
<surname>Boeke</surname>
<given-names>J. D.</given-names>
</name>
<name>
<surname>Bader</surname>
<given-names>J. S.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Finding friends and enemies in an enemies-only network: a graph diffusion kernel for predicting novel genetic interactions and co-complex membership from yeast genetic interactions</article-title>. <source>Genome Res.</source> <volume>18</volume>, <fpage>1991</fpage>&#x2013;<lpage>2004</lpage>. <pub-id pub-id-type="doi">10.1101/gr.077693.108</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ryan</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Lord</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Ashworth</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Daisy: picking synthetic lethals from cancer genomes</article-title>. <source>Cancer Cell.</source> <volume>26</volume>, <fpage>306</fpage>&#x2013;<lpage>308</lpage>. <pub-id pub-id-type="doi">10.1016/j.ccr.2014.08.008</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schmidt</surname>
<given-names>E. E.</given-names>
</name>
<name>
<surname>Pelz</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Buhlmann</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kerr</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Horn</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Boutros</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Genomernai: a database for cell-based and <italic>in vivo</italic> rnai phenotypes, 2013 update</article-title>. <source>Nucleic Acids Res.</source> <volume>41</volume>, <fpage>D1021</fpage>&#x2013;<lpage>D1026</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gks1170</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shen</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Sasik</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Luebeck</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Birmingham</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bojorquez-Gomez</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Combinatorial crispr&#x2013;cas9 screens for <italic>de novo</italic> mapping of genetic interactions</article-title>. <source>Nat. Methods</source> <volume>14</volume>, <fpage>573</fpage>&#x2013;<lpage>576</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.4225</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Srihari</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Singla</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ragan</surname>
<given-names>M. A.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Inferring synthetic lethal interactions from mutual exclusivity of genetic events in cancer</article-title>. <source>Biol. Direct</source> <volume>10</volume>, <fpage>57</fpage>&#x2013;<lpage>18</lpage>. <pub-id pub-id-type="doi">10.1186/s13062-015-0086-1</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Veli&#x10d;kovi&#x107;</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Cucurull</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Casanova</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Romero</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lio</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Graph attention networks</article-title>. <source>arXiv Prepr. arXiv:1710.10903</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1710.10903</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Leskovec</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2019a</year>). &#x201c;<article-title>Knowledge-aware graph neural networks with label smoothness regularization for recommender systems</article-title>,&#x201d; in <conf-name>Proceedings of the 25th ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</conf-name>, <conf-loc>USA</conf-loc>, <conf-date>August 4 - 8, 2019</conf-date> (<publisher-name>IEEE</publisher-name>), <fpage>968</fpage>&#x2013;<lpage>977</lpage>.</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2019b</year>). <article-title>Knowledge graph convolutional networks for recommender systems</article-title>. <source>World Wide Web Conf.</source>, <fpage>3307</fpage>&#x2013;<lpage>3313</lpage>. <pub-id pub-id-type="doi">10.1145/3308558.3313417</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Synlethdb 2.0: a web-based knowledge graph database on synthetic lethality for novel anticancer drug discovery</article-title>. <source>Database</source> <volume>2022</volume>, <fpage>baac030</fpage>. <pub-id pub-id-type="doi">10.1093/database/baac030</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Knowledge graph embedding: a survey of approaches and applications</article-title>. <source>IEEE Trans. Knowl. Data Eng.</source> <volume>29</volume>, <fpage>2724</fpage>&#x2013;<lpage>2743</lpage>. <pub-id pub-id-type="doi">10.1109/TKDE.2017.2754499</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Kg4sl: knowledge graph neural network for synthetic lethality prediction in human cancers</article-title>. <source>Bioinformatics</source> <volume>37</volume>, <fpage>i418</fpage>&#x2013;<lpage>i425</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btab271</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wong</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L. V.</given-names>
</name>
<name>
<surname>Tong</surname>
<given-names>A. H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Goldberg</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>King</surname>
<given-names>O. D.</given-names>
</name>
<etal/>
</person-group> (<year>2004</year>). <article-title>Combining biological networks to predict genetic interactions</article-title>. <source>Proc. Natl. Acad. Sci.</source> <volume>101</volume>, <fpage>15682</fpage>&#x2013;<lpage>15687</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.0406614101</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Kwoh</surname>
<given-names>C.-K.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>
<italic>In silico</italic> prediction of synthetic lethality by meta-analysis of genetic interactions, functions, and pathways in yeast and human cancer</article-title>. <source>Cancer Inf.</source> <volume>13</volume>, <fpage>71</fpage>&#x2013;<lpage>80</lpage>. <comment>CIN&#x2013;S14026</comment>. <pub-id pub-id-type="doi">10.4137/CIN.S14026</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yeung</surname>
<given-names>M.-L.</given-names>
</name>
<name>
<surname>Yao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>J. F.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>K.-H.</given-names>
</name>
<name>
<surname>Cheung</surname>
<given-names>K.-F.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Mers coronavirus induces apoptosis in kidney and lung by upregulating smad7 and fgf2</article-title>. <source>Nat. Microbiol.</source> <volume>1</volume>, <fpage>16004</fpage>&#x2013;<lpage>16008</lpage>. <pub-id pub-id-type="doi">10.1038/nmicrobiol.2016.4</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Predicting drug&#x2013;disease associations through layer attention graph convolutional network</article-title>. <source>Briefings Bioinforma.</source> <volume>22</volume>, <fpage>bbaa243</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbaa243</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.-J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.-L.</given-names>
</name>
<name>
<surname>Kwoh</surname>
<given-names>C. K.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Predicting essential genes and synthetic lethality via influence propagation in signaling pathways of cancer cell fates</article-title>. <source>J. Bioinforma. Comput. Biol.</source> <volume>13</volume>, <fpage>1541002</fpage>. <pub-id pub-id-type="doi">10.1142/S0219720015410024</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Kgansynergy: knowledge graph attention network for drug synergy prediction</article-title>. <source>Briefings Bioinforma.</source> <volume>24</volume>, <fpage>bbad167</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbad167</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Slgnn: synthetic lethality prediction in human cancers based on factor-aware knowledge graph neural network</article-title>. <source>Bioinformatics</source> <volume>39</volume>, <fpage>btad015</fpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btad015</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>