<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1606016</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2025.1606016</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Integrating BERT pre-training with graph common neighbours for predicting ceRNA interactions</article-title>
<alt-title alt-title-type="left-running-head">Xie et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2025.1606016">10.3389/fgene.2025.1606016</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Xie</surname>
<given-names>Zhengxing</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2959610/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Ying</surname>
<given-names>Tianping</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Jing</surname>
<given-names>Ge</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liang</surname>
<given-names>Shiyang</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2950357/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Junhua</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Tang</surname>
<given-names>Lianghua</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Guizhou University of Traditional Chinese Medicine</institution>, <addr-line>Guiyang</addr-line>, <addr-line>Guizhou</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>The Second Affiliated Hospital of Guizhou University of Traditional Chinese Medicine</institution>, <addr-line>Guiyang</addr-line>, <addr-line>Guizhou</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Internal Medicine</institution>, <institution>The No. 944 Hospital of Logistic Support Force of PLA</institution>, <addr-line>Jiuquan</addr-line>, <addr-line>Gansu</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>School of Computing and Information Systems, The University of Melbourne</institution>, <addr-line>Melbourne</addr-line>, <addr-line>VC</addr-line>, <country>Australia</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/732997/overview">Martina Calore</ext-link>, University of Padua, Italy</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/942337/overview">Zhihao Ma</ext-link>, East China Normal University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1668746/overview">Morteza Kouhsar</ext-link>, University of Exeter, United Kingdom</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1677454/overview">Jie Ni</ext-link>, Changzhou University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Lianghua Tang, <email>6939006@qq.com</email>
</corresp>
<fn fn-type="equal" id="fn001">
<label>
<sup>&#x2020;</sup>
</label>
<p>These authors have contributed equally to this work and share first authorship</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>03</day>
<month>09</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1606016</elocation-id>
<history>
<date date-type="received">
<day>04</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>08</day>
<month>07</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Xie, Ying, Jing, Liang, Liu and Tang.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Xie, Ying, Jing, Liang, Liu and Tang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Predicting interactions between microRNAs (miRNAs) and competing endogenous RNAs (ceRNAs), including long non-coding RNAs (lncRNAs) and circular RNAs (circRNAs), is essential for understanding gene regulation. With the development of Graph Neural Networks (GNNs), existing works have demonstrated the ability to capture information from miRNA-ceRNA interactions to predict unseen associations. However, current deep GNNs only leverage node-node pairwise features, neglecting the information inherent in the RNA chains themselves, as different RNAs possess chains of varying lengths.</p>
</sec>
<sec>
<title>Methods</title>
<p>To address this issue, we propose a novel model termed the BERT-based ceRNA Graph Predictor (BCGP), which leverages both RNA sequence information and the heterogeneous relationships among lncRNAs, circRNAs, and miRNAs. Our BCGP method employs a transformer-based model to generate contextualized representations that consider the global context of the entire RNA sequence. Subsequently, we enrich the RNA interaction graph using these contextualized representations. Furthermore, to improve the performance of association prediction, BCGP utilizes the Neural Common Neighbour (NCN) technique to capture more refined node features, leading to more informative and flexible representations.</p>
</sec>
<sec>
<title>Results</title>
<p>Through comprehensive experiments on two real-world datasets of lncRNA-miRNA and circRNA-miRNA associations, we demonstrate that BCGP outperforms competitive baselines across various evaluation metrics and achieves higher accuracy in association predictions. In our case studies on two types of miRNAs, we show BCGP&#x2019;s remarkable performance in predicting both miRNA-lncRNA and miRNA-circRNA associations.</p>
</sec>
<sec>
<title>Discussion</title>
<p>Our findings demonstrate that by integrating RNA sequence information with interaction relationships within the graph, the BCGP model significantly enhances the accuracy of association prediction. This provides a new computational tool for understanding complex gene regulatory networks.</p>
</sec>
</abstract>
<kwd-group>
<kwd>lncRNA</kwd>
<kwd>circRNA</kwd>
<kwd>miRNA</kwd>
<kwd>ceRNA</kwd>
<kwd>pre-train</kwd>
<kwd>graph neural network</kwd>
</kwd-group>
<counts>
<page-count count="10"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>RNA</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>MicroRNAs (miRNAs) are a class of small, non-coding RNA molecules that play a crucial role in the regulation of gene expression (<xref ref-type="bibr" rid="B41">Ye et al., 2019</xref>). MiRNAs are present in plants, animals and some viruses and they can significantly affect a broad range of biological processes (<xref ref-type="bibr" rid="B2">Bartel, 2018</xref>). Specifically, they primarily regulate gene expression through binding to the 3&#x2032; untranslated regions (<italic>UTRs</italic>) of target mRNAs, leading to their degradation or inhibition of translation. The degree of complementarity between the miRNA and the target mRNA is crucial, as it determines the mechanism of repression. MiRNA sponges, also known as competing endogenous RNAs (ceRNAs), embody a sophisticated biological mechanism that serves to regulate the activity of miRNAs within cellular environments (<xref ref-type="bibr" rid="B1">Alkan and Akg&#xfc;l, 2022</xref>). Such a mechanism leverages RNA molecules containing multiple miRNA binding sites to effectively &#x201c;absorb&#x201d; or &#x201c;sponge&#x201d; certain miRNAs. As a result, it diminishes the suppressive impacts of these miRNAs on their intended target mRNAs. There are numerous previous studies that have employed machine learning to forecast the miRNA-disease associations, achieving satisfactory results (<xref ref-type="bibr" rid="B8">Chen et al., 2021</xref>; <xref ref-type="bibr" rid="B18">Ha et al., 2020</xref>).</p>
<p>Long non-coding RNAs (lncRNAs), a unique type of RNA with over 200 nucleotides, do not have protein-coding capacity. Circular RNAs (circRNAs) constitute another category of non-coding RNA, distinguished by their unique covalently closed-loop configuration. Unlike linear RNAs, circRNAs lack both a 5&#x2032; cap and a 3&#x2032; poly-A tail in their sequences. They are synthesized through a mechanism known as back-splicing, wherein a splice donor site downstream is connected to a splice acceptor site upstream, resulting in the circularization of the RNA molecule (<xref ref-type="bibr" rid="B23">Kristensen et al., 2022</xref>). Both lncRNA and circRNA are considered as the major types of ceRNAs, where they can sponge specific miRNAs when they have miRNA binding sites, which can potentially alleviate the inhibitory effects these miRNAs exert on their target mRNAs.</p>
<p>Inspired by methods for predicting miRNA-disease associations (<xref ref-type="bibr" rid="B17">Ha et al., 2019</xref>), several approaches for ceRNA association prediction have emerged in recent years. For example, <xref ref-type="bibr" rid="B46">Zhu et al. (2017)</xref> discovered that lnc-mg promotes myogenesis by sponging microRNA-125b to regulate the abundance of the IGF2 protein. Furthermore, <xref ref-type="bibr" rid="B42">Zhang et al. (2018)</xref> identified that lncRNA MAR1 acts as a miR-487b sponge to promote skeletal muscle differentiation and regeneration. Similarly, <xref ref-type="bibr" rid="B38">Yang et al. (2018)</xref> found that circ-ITCH can sponge miR-17 and miR-224, thereby upregulating the expression of p21 and PTEN, which in turn suppresses the aggressive biological behaviors associated with bladder cancer.</p>
<p>Traditionally, determining the associations between miRNA and ceRNAs requires the use of technologies such as High-Throughput Sequencing (<xref ref-type="bibr" rid="B26">Li et al., 2018</xref>), RNA Immunoprecipitation (<xref ref-type="bibr" rid="B13">Gawronski et al., 2018</xref>) and Dual-Luciferase Reporter (<xref ref-type="bibr" rid="B5">Cao et al., 2016</xref>). However, these methods can cost a large investment of time and resources. With the advancement of deep learning technologies especially the deep Graph Neural Networks (GNNs) (<xref ref-type="bibr" rid="B22">Kipf and Welling, 2016</xref>) and the accumulation of historical experimental data, there has been a surge in efforts to predict lncRNA-miRNA and circRNA-miRNA associations using computational techniques. For example, <xref ref-type="bibr" rid="B34">Wang W. et al. (2022)</xref> proposed a model, named GCNCRF, to predict lncRNA-miRNA associations based on the graph convolutional network (GCN) and conditional random field. Furthermore, a recent work (<xref ref-type="bibr" rid="B36">Wang Z. et al., 2023</xref>) has employed a sequence pre-training-based Graph Neural Network to predict associations between lncRNAs and miRNAs.</p>
<p>Although effective, existing methods still have several limitations. First, most of the graph-based models have not effectively utilized the information contained within RNA sequences. As the foundational elements, an RNA sequence contains nearly all the information of the RNA (<xref ref-type="bibr" rid="B6">Charles Richard and Eichhorn, 2018</xref>). Using an appropriate sequential model to analyze these RNA sequences could uncover enormous characteristics of each RNA. Furthermore, most previous studies used only homogeneous or bipartite graphs to predict the associations between lncRNA-miRNA or circRNA-miRNA. However, according to the mechanism of ceRNAs (<xref ref-type="bibr" rid="B45">Zhong et al., 2018</xref>), circRNAs and lncRNAs, although two different types of RNAs, both function as miRNA sponges within this network. Therefore, a heterogeneous graph can be used to construct the ceRNA network among circRNA, lncRNA, and miRNA to further boost the performance.</p>
<p>To address the aforementioned challenges, we propose a novel framework, the BERT-based ceRNA Graph Predictor (BCGP), which uniquely integrates sequence-level and structural information for comprehensive ceRNA interaction prediction. Unlike existing methods, BCGP innovatively combines Bidirectional Encoder Representations from Transformers (BERT) for sequence pre-training and a heterogeneous graph model for fine-tuning. Specifically, our framework incorporates heterogeneous relations between lncRNAs, miRNAs, and circRNAs to capture both contextual and relational dependencies. In the pre-training stage, BCGP uses BERT with Masked Language Modeling (MLM) as the training objective, a choice justified by extensive comparative analysis, to derive high-quality embeddings for different types of RNA sequences. These embeddings are then seamlessly integrated into a heterogeneous graph, where nodes represent RNAs and edges capture their intricate relationships. In the fine-tuning stage, BCGP leverages the Neural Common Neighbour (NCN) method (<xref ref-type="bibr" rid="B35">Wang X. et al., 2023</xref>), further enhancing the expressive power of the graph embeddings by incorporating relational patterns. Extensive experiments have demonstrated the effectiveness of our novel BCGP framework, consistently outperforming other state-of-the-art models. Our ablation studies confirm the critical contributions of each component within BCGP, while case studies on two specific miRNAs, hsa-miR-143 and hsa-miR-6808-5p, validate the superior practicality and real-world relevance of our approach. By bridging the gap between sequence-level and structural modeling, our work establishes a new paradigm for ceRNA interaction prediction.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<sec id="s2-1">
<title>2.1 Overview</title>
<p>In this section, we will describe our proposed method BCGP that we use to generate pre-trained embeddings for lncRNAs, circRNAs, or miRNAs and fine-tune with general Graph Neural Networks to recover the unseen associations between them. Additionally, we integrate a novel training method, Neural Common Neighbour (NCN) (<xref ref-type="bibr" rid="B35">Wang X. et al., 2023</xref>) to further enhance the performance of the fine-tuning GNNs. In <xref ref-type="sec" rid="s2-2">Section 2.2</xref>, we will present the notations used in this article and briefly describe our research problem formulation. Second, in <xref ref-type="sec" rid="s2-3">Section 2.3</xref>, we demonstrate the pre-training stage of BCGP, where we use the <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-mers method to split all RNA sequences into fragments of equal length, then we use Bidirectional Encoder Representations from Transformers (BERT) (<xref ref-type="bibr" rid="B9">Devlin et al., 2018</xref>), which is a transformer-based contextualized language representation model to generate pre-trained embeddings. Third, in <xref ref-type="sec" rid="s2-4">Section 2.4</xref>, we describe a general fine-tuning method to incorporate a GNN to leverage the pre-trained embeddings obtained from <xref ref-type="sec" rid="s2-3">Section 2.3</xref>. To boost the performance of the fine-tuning GNN, we also present the integration of a novel training method named Neural Common Neighbour. We use <xref ref-type="fig" rid="F1">Figure 1</xref> to illustrate the overview of our BCGP method. The detailed mathematical formulation of our model is provided in <xref ref-type="disp-formula" rid="e1">Equations 1</xref>&#x2013;<xref ref-type="disp-formula" rid="e13">13</xref>.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Overview framework of the proposed BCGP model. In the pre-training stage, BCGP first tokenizes an RNA sequence with <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-mers and utilizes the BERT model to learn contextualized embeddings for each RNA sequence. In the fine-tuning stage, BCGP leverages GNN and Neural Common Neighbour (NCN) to capture the complex relationship and learn informative node representations. <bold>(A)</bold> Pre-training stage. <bold>(B)</bold> Fine-tunning stage.</p>
</caption>
<graphic xlink:href="fgene-16-1606016-g001.tif">
<alt-text content-type="machine-generated">Diagram of a two-stage model for RNA processing. In the pre-training stage, a corpus of RNAs, including lncRNA, circRNA, and miRNA, is tokenized into nucleobases and processed into K-mers. These K-mers are encoded with a pre-training encoder using models like Random, Text2Vec, Doc2Vec, and BERT. In the fine-tuning stage, embeddings for lncRNA, miRNA, and circRNA are initialized in a heterogeneous graph and processed through a GNN encoder and NCN enhancer for link predictions.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2-2">
<title>2.2 Preliminaries</title>
<p>To initialize our task, we first define all entities and their associated information. In particular, we denote sets of lncRNAs, circRNAs and miRNAs as <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>lnc</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>circ</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>mi</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> respectively. Considering the associations between lncRNAs, circRNAs, and miRNAs we construct an undirected graph <inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mi mathvariant="script">V</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="script">E</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold">A</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:mi mathvariant="script">V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the set of all RNAs <inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="script">V</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>lnc</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x222a;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>circ</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x222a;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>mi</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:mi mathvariant="script">E</mml:mi>
<mml:mo>&#x2286;</mml:mo>
<mml:mi mathvariant="script">V</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="script">V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the set of associations between RNAs and adjacency matrix <inline-formula id="inf10">
<mml:math id="m10">
<mml:mrow>
<mml:mi mathvariant="bold">A</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is a symmetric matrix, which is defined as follows:<disp-formula id="e1">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="{" close="">
<mml:mrow>
<mml:mtable class="cases">
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mn>1</mml:mn>
<mml:mspace width="1em"/>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mtext>if&#x2009;</mml:mtext>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">E</mml:mi>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mn>0</mml:mn>
<mml:mspace width="1em"/>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mtext>otherwise</mml:mtext>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <inline-formula id="inf11">
<mml:math id="m12">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi mathvariant="script">V</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is the number of all types of RNAs. The degree of node <inline-formula id="inf12">
<mml:math id="m13">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is <inline-formula id="inf13">
<mml:math id="m14">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold">A</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2254;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The neighbours of node <inline-formula id="inf14">
<mml:math id="m15">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> are the nodes connected to <inline-formula id="inf15">
<mml:math id="m16">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, which is defined as <inline-formula id="inf16">
<mml:math id="m17">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold">A</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2254;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo stretchy="false">&#x2223;</mml:mo>
<mml:mi>v</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">V</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3e;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. For brevity, we use <inline-formula id="inf17">
<mml:math id="m18">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> to denote <inline-formula id="inf18">
<mml:math id="m19">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold">A</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> since <inline-formula id="inf19">
<mml:math id="m20">
<mml:mrow>
<mml:mi mathvariant="bold">A</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is fixed. The common neighbour refers to the nodes connected to <inline-formula id="inf20">
<mml:math id="m21">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf21">
<mml:math id="m22">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>: <inline-formula id="inf22">
<mml:math id="m23">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2229;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Let <inline-formula id="inf23">
<mml:math id="m24">
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> be the collection of RNA sequences, including all lncRNAs, circRNAs and miRNAs. Specifically, <inline-formula id="inf24">
<mml:math id="m25">
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where each <inline-formula id="inf25">
<mml:math id="m26">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents a unique RNA sequence.</p>
<p>In this next section, we will provide a detailed description of our proposed method, including the techniques in the pre-training and fine-tuning stages.</p>
</sec>
<sec id="s2-3">
<title>2.3 Pre-training stage</title>
<p>During the pre-training stage, our goal is to effectively project the information of RNAs from their nucleotide sequences to latent embeddings. Specifically, we aim to simplify computational demands while enriching the semantic content of the embeddings, which captures the intrinsic patterns and relations within RNA sequences. Such embeddings not only preserve the biological significance and genetic information of RNA sequences but also serve as the initial features of nodes for the subsequent fine-tuning stage so that those sequence information can be encoded into the graph representation.</p>
<p>Instead of treating each base (A, C, G, U) as an individual token, given an RNA sequence <inline-formula id="inf26">
<mml:math id="m27">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, we leverage the <inline-formula id="inf27">
<mml:math id="m28">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-mers tokenization to segment <inline-formula id="inf28">
<mml:math id="m29">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> into overlapping and equal-length segment, with <inline-formula id="inf29">
<mml:math id="m30">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> indicating the length of each segment. Let <inline-formula id="inf30">
<mml:math id="m31">
<mml:mrow>
<mml:mi mathvariant="bold">C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> be the collection of RNA <inline-formula id="inf31">
<mml:math id="m32">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-mer sequences, and <inline-formula id="inf32">
<mml:math id="m33">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> can be denoted as follows:<disp-formula id="e2">
<mml:math id="m34">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="{" close="}">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>:</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x2223;</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1,2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <inline-formula id="inf33">
<mml:math id="m35">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> represents the total number of <inline-formula id="inf34">
<mml:math id="m36">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-mers of <inline-formula id="inf35">
<mml:math id="m37">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf36">
<mml:math id="m38">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the length of <inline-formula id="inf37">
<mml:math id="m39">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. For example, the RNA sequence &#x2018;UAACAC&#x2019; can be tokenized to a sequence of four 3-mers: UAA, AAC, ACA, CAC. This method not only enables the implementation of sequence embedding algorithms, but also deepens the understanding of richer contextual information for each nucleic acid sequence.</p>
<p>After tokenizing each RNA sequence into the overlapping segment using <inline-formula id="inf38">
<mml:math id="m40">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-mer tokenization, we utilize BERT to capture both contextual and structural information from the whole corpus of RNA sequences with an attention mechanism. As a widely used transformer-based language representation model, BERT enables the generation of contextualized representations that consider the global context of the entire sequence, allowing the identification of the intricate patterns and relationships within the RNA sequences.</p>
<p>Given a sequence of <inline-formula id="inf39">
<mml:math id="m41">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-mers <inline-formula id="inf40">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> derived from an RNA sequence <inline-formula id="inf41">
<mml:math id="m43">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, we first initialize each <inline-formula id="inf42">
<mml:math id="m44">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-mer into a high-dimensional vector through an embedding process, resulting in the embedding matrix <inline-formula id="inf43">
<mml:math id="m45">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf44">
<mml:math id="m46">
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the dimension of the embedding vectors. To obtain contextual and informative embedding <inline-formula id="inf45">
<mml:math id="m47">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, BCGP performs the multi-head self-attention mechanism on <inline-formula id="inf46">
<mml:math id="m48">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, which is defined as:<disp-formula id="e3">
<mml:math id="m49">
<mml:mrow>
<mml:mtext>MultiHead</mml:mtext>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2295;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>head</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mtext>head</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mtext>head</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>O</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
<disp-formula id="e4">
<mml:math id="m50">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>head</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>softmax</mml:mtext>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
<disp-formula id="e5">
<mml:math id="m51">
<mml:mrow>
<mml:mtext>where</mml:mtext>
<mml:mspace width="0.3333em"/>
<mml:mspace width="0.3333em"/>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mspace width="0.3333em"/>
<mml:mspace width="0.3333em"/>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mspace width="0.3333em"/>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>where <inline-formula id="inf47">
<mml:math id="m52">
<mml:mrow>
<mml:mo>&#x2295;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> denotes the concatenation operation, <inline-formula id="inf48">
<mml:math id="m53">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>O</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf49">
<mml:math id="m54">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf50">
<mml:math id="m55">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf51">
<mml:math id="m56">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> are the learnable parameters for linear projection for the <inline-formula id="inf52">
<mml:math id="m57">
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th head. Here, <inline-formula id="inf53">
<mml:math id="m58">
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the number of attention heads, and <inline-formula id="inf54">
<mml:math id="m59">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the scaling factor used to maintain numerical stability and facilitate stable gradients during training, often set to <inline-formula id="inf55">
<mml:math id="m60">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>/</mml:mo>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. After applying the multi-head self-attention mechanism multiple times, we can derive the contextual embedding <inline-formula id="inf56">
<mml:math id="m61">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> from the output of the last layer, which is denoted as:<disp-formula id="e6">
<mml:math id="m62">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mtext>MultiHead</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>Last</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
</p>
<p>Building upon the aforementioned self-attention mechanism, we adapt the Masked Language Modeling (MLM) to train the BERT model on RNA sequences. For each RNA sequence, we randomly select regions constituting <inline-formula id="inf57">
<mml:math id="m63">
<mml:mrow>
<mml:mn>15</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of the sequence and mask contiguous <inline-formula id="inf58">
<mml:math id="m64">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-mers within these regions. Using the surrounding context, the model is then trained to predict the masked <inline-formula id="inf59">
<mml:math id="m65">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-mers. The training objective to minimize the cross-entropy loss which is defined as follows:<disp-formula id="e7">
<mml:math id="m66">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>MLM</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>where <inline-formula id="inf60">
<mml:math id="m67">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> represents the one-hot encoded ground-truth vector for the masked <inline-formula id="inf61">
<mml:math id="m68">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-mers and <inline-formula id="inf62">
<mml:math id="m69">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the predicted probability distribution over the <inline-formula id="inf63">
<mml:math id="m70">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-mer vocabulary for each of the <inline-formula id="inf64">
<mml:math id="m71">
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> masked positions. Specifically, the predicted probability distribution <inline-formula id="inf65">
<mml:math id="m72">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> for a masked position is defined as follows:<disp-formula id="e8">
<mml:math id="m73">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>Softmax</mml:mtext>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>where <inline-formula id="inf66">
<mml:math id="m74">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the <inline-formula id="inf67">
<mml:math id="m75">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th encoded representation from <inline-formula id="inf68">
<mml:math id="m76">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf69">
<mml:math id="m77">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf70">
<mml:math id="m78">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the parameters of the linear classifier respectively.</p>
</sec>
<sec id="s2-4">
<title>2.4 Fine-tuning stage</title>
<p>After obtaining the initial RNA embeddings for lncRNAs, circRNAs and miRNAs from BERT during the pre-training stage, we construct a heterogeneous graph. This graph integrates known associations between lncRNAs and miRNAs, as well as the high-quality associations between circRNAs and miRNAs, with these relationships represented as edges. The pre-trained RNA sequence embeddings are utilized as node features in this graph. To obtain effective node embeddings, we leverage GNNs to capture the intricate and complex relationships between entities.</p>
<p>The current <italic>de facto</italic> design of GNNs follows the message passing framework (<xref ref-type="bibr" rid="B22">Kipf and Welling, 2016</xref>), which is based on the core idea of recursive neighborhood aggregation. Specifically, for an <inline-formula id="inf71">
<mml:math id="m79">
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-layer GNN, the representation learning function of the <inline-formula id="inf72">
<mml:math id="m80">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th layer is represented as:<disp-formula id="e9">
<mml:math id="m81">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mi mathvariant="normal">E</mml:mi>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">E</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mfenced open="{" close="}">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x222a;</mml:mo>
<mml:mfenced open="{" close="}">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>:</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>
<disp-formula id="e10">
<mml:math id="m82">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi mathvariant="normal">O</mml:mi>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mi mathvariant="normal">B</mml:mi>
<mml:mi mathvariant="normal">I</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mi mathvariant="normal">E</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>where <inline-formula id="inf73">
<mml:math id="m83">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is a message vector computed from the representations of the neighbors <inline-formula id="inf74">
<mml:math id="m84">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> from the previous layer i.e., <inline-formula id="inf75">
<mml:math id="m85">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>-th layer, <inline-formula id="inf76">
<mml:math id="m86">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is a set of nodes adjacent to <inline-formula id="inf77">
<mml:math id="m87">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf78">
<mml:math id="m88">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the representation of node <inline-formula id="inf79">
<mml:math id="m89">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> at the <inline-formula id="inf80">
<mml:math id="m90">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th layer with <inline-formula id="inf81">
<mml:math id="m91">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf82">
<mml:math id="m92">
<mml:mrow>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mi mathvariant="normal">E</mml:mi>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf83">
<mml:math id="m93">
<mml:mrow>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi mathvariant="normal">O</mml:mi>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mi mathvariant="normal">B</mml:mi>
<mml:mi mathvariant="normal">I</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mi mathvariant="normal">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> are the component functions of GNN layers. It is worth noting that our proposed BCGP approach is a general framework that can be incorporated with various GNNs.</p>
<p>To further boost the performance of association prediction, we leverage the Neural Common Neighbour (<xref ref-type="bibr" rid="B35">Wang X. et al., 2023</xref>) to capture more refined node features such as multi-hop structure and attribute information. Specifically, after obtaining the representation <inline-formula id="inf84">
<mml:math id="m94">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> of node <inline-formula id="inf85">
<mml:math id="m95">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, instead of directly using the node representation for link prediction, BCGP focuses on the pairwise relationships between nodes, specifically leveraging the common neighbours of each pair nodes under consideration. For a target link between node <inline-formula id="inf86">
<mml:math id="m96">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and node <inline-formula id="inf87">
<mml:math id="m97">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, BCGP sums up the representation of their common neighbours obtained from the GNNs. This emphasizes the structural context and the shared neighbourhood, which are pivotal for predicting the existence of a link. Formally, the pairwise representation <inline-formula id="inf88">
<mml:math id="m98">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>for a potential link between node <inline-formula id="inf89">
<mml:math id="m99">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and node <inline-formula id="inf90">
<mml:math id="m100">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> can be represented as:<disp-formula id="e11">
<mml:math id="m101">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2229;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mtext>GNN</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold">A</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold">Z</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>where <inline-formula id="inf91">
<mml:math id="m102">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the parameters of the GNN, <inline-formula id="inf92">
<mml:math id="m103">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2229;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the set of common neighbours between node <inline-formula id="inf93">
<mml:math id="m104">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and node <inline-formula id="inf94">
<mml:math id="m105">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. The aggregated pairwise representation <inline-formula id="inf95">
<mml:math id="m106">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is then used to compute the probability <inline-formula id="inf96">
<mml:math id="m107">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> of a link between node <inline-formula id="inf97">
<mml:math id="m108">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and node <inline-formula id="inf98">
<mml:math id="m109">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. This is achieved by passing <inline-formula id="inf99">
<mml:math id="m110">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> through a final prediction layer, such as a fully connected layer with a sigmoid activation function, which is denoted as follows:<disp-formula id="e12">
<mml:math id="m111">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(12)</label>
</disp-formula>where <inline-formula id="inf100">
<mml:math id="m112">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the sigmoid function, and <inline-formula id="inf101">
<mml:math id="m113">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf102">
<mml:math id="m114">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are learnable parameters of the prediction layer.</p>
<p>To train the model on the link prediction (LP) task, we use the binary cross-entropy loss, which is denoted as:<disp-formula id="e13">
<mml:math id="m115">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>LP</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi mathvariant="script">M</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">M</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mi>log</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(13)</label>
</disp-formula>where <inline-formula id="inf103">
<mml:math id="m116">
<mml:mrow>
<mml:mi mathvariant="script">M</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the set of all node pairs, which contains both positive (existing links) and negative (non-existing links) samples, <inline-formula id="inf104">
<mml:math id="m117">
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi mathvariant="script">M</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> denotes the total number of node pairs in the set <inline-formula id="inf105">
<mml:math id="m118">
<mml:mrow>
<mml:mi mathvariant="script">M</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf106">
<mml:math id="m119">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the ground truth label for the link between nodes <inline-formula id="inf107">
<mml:math id="m120">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf108">
<mml:math id="m121">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf109">
<mml:math id="m122">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> if a link exists and <inline-formula id="inf110">
<mml:math id="m123">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> otherwise.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<p>In this section, we present the evaluation performance of our BCGP. We begin by describing the datasets used in our experiments. Following this, we introduce the experimental results, including the evaluation of pre-training, fine-tuning, performance comparisons with baselines, hyperparameter analysis, and case studies. For more details on the evaluation metrics and experimental settings, please refer to <xref ref-type="sec" rid="s1">Sections 1</xref>, <xref ref-type="sec" rid="s2">2</xref> of the <xref ref-type="sec" rid="s11">Supplementary Material</xref>.</p>
<sec id="s3-1">
<title>3.1 Datasets</title>
<p>In this study, we focus on leveraging a comprehensive dataset to model lncRNA-miRNA and circRNA-miRNA associations. We have constructed a total of four datasets: lncRNA-miRNA association 1 (LMA1), circRNA-miRNA associations 1 (CMA1), lncRNA-miRNA association 2 (LMA2), and circRNA-miRNA associations 2 (CMA2). For the lncRNA-miRNA associations in LMA1, based on previous research (<xref ref-type="bibr" rid="B36">Wang Z. et al., 2023</xref>), we utilized the LncACTdb 3.0 (<xref ref-type="bibr" rid="B33">Wang P. et al., 2022</xref>) database. From this database, we extracted 1,057 experimentally verified lncRNA-miRNA associations, containing 284 lncRNAs and 520 miRNAs. Regarding the circRNA-miRNA associations in CMA1, building upon prior studies (<xref ref-type="bibr" rid="B16">Guo et al., 2024</xref>), we used CircBank (<xref ref-type="bibr" rid="B27">Liu et al., 2019</xref>) as the primary data source, obtaining 20,771 high-quality associations involving 3,802 circRNAs and 1,273 miRNAs. The lncRNA-miRNA associations in LMA2 were sourced from lncRNASNP V3.0 (<xref ref-type="bibr" rid="B39">Yang et al., 2023</xref>), comprising a total of 8,502 lncRNA-miRNA associations, including 467 lncRNAs and 254 miRNAs. As for the circRNA-miRNA associations in CMA2, they were derived from the dataset 1 of the KGANCDA (<xref ref-type="bibr" rid="B24">Lan et al., 2022</xref>), with a total of 702 circRNA-miRNA associations, encompassing 471 circRNAs and 439 miRNAs. The sequences of lncRNAs were sourced from LNCipedia (<xref ref-type="bibr" rid="B32">Volders et al., 2019</xref>) and NONCODE (<xref ref-type="bibr" rid="B44">Zhao et al., 2021</xref>), the sequences of circRNAs were obtained from CircBase (<xref ref-type="bibr" rid="B14">Gla&#x17e;ar et al., 2014</xref>), and the sequences of miRNAs were acquired from miRBase (<xref ref-type="bibr" rid="B15">Griffiths-Jones et al., 2007</xref>).</p>
</sec>
<sec id="s3-2">
<title>3.2 Examination of pre-training</title>
<p>To rigorously evaluate the performance of pre-training within our method, we compare several pre-training methods commonly used for initializing node embedding in GNN. We include two baselines, namely, the Random Embedding and the Adjacency Matrix Embedding methods which do not consider any RNA sequence information. The Random Embedding method initializes node embeddings with random values generated from a Gaussian distribution, while the Adjacency Matrix Embedding method leverages the adjacency matrix, representing the relationships between nodes in a lncRNA-miRNA-circRNA association graph, to generate node embeddings. Furthermore, we also compare three sequence-specific pre-training methods, including Text2vec (<xref ref-type="bibr" rid="B28">Mikolov et al., 2013</xref>), Doc2vec (<xref ref-type="bibr" rid="B25">Le and Mikolov, 2014</xref>) and HyenaDNA (<xref ref-type="bibr" rid="B29">Nguyen et al., 2024</xref>). Firstly, Text2vec employs a &#x201c;bag of words&#x201d; model for the <inline-formula id="inf111">
<mml:math id="m124">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-mer representations of RNA sequence, converting each lncRNA, circRNA or miRNA sequence into a numerical representation based on <inline-formula id="inf112">
<mml:math id="m125">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-mer occurrence frequency. Meanwhile, Doc2vec extends the Word2vec algorithm that generates document-level embeddings of the RNA sequences. Additionally, HyenaDNA leverages pre-trained HyenaDNA to generate the embeddings of RNA sequences at the nucleotide level, capturing the long-range dependencies within the RNA sequences.</p>
<p>The experimental results on the LMA1 and CMA1 datasets presented in <xref ref-type="table" rid="T1">Table 1</xref>, demonstrate the effectiveness of BCGP-BERT. For lncRNA-miRNA prediction, BCGP-BERT achieves the highest scores across all evaluation metrics, with an F1 score of 0.439, AUC of 0.904, AP of 0.435, and NDCG of 0.816, outperforming all other pre-training methods. Similarly, for circRNA-miRNA prediction, BCGP-BERT excels in all metrics, with an F1 score of 0.572, AUC of 0.948, AP of 0.712, and NDCG of 0.957. These results suggest that BCGP-BERT enhances association prediction by learning contextualized representations that capture the global context of entire sequences. Furthermore, the findings also highlight the effectiveness of using the BERT model trained on task-related data through Masked Language Modeling, enabling BCGP-BERT to discover the intricate pattern of RNA sequences and demonstrate its robustness in predicting RNA interactions. Additional pre-training results on the LMA2 and CMA2 datasets are provided in <xref ref-type="sec" rid="s11">Supplementary Appendix S3</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Overall performance comparison of different pre-training methods fine-tuned with GCN on lncRNA-miRNA and circRNA-miRNA association prediction tasks using the LMA1 and CMA1 datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Methods</th>
<th colspan="4" align="left">lncRNA-miRNA</th>
<th colspan="4" align="left">circRNA-miRNA</th>
</tr>
<tr>
<th align="left">Pre-train</th>
<th align="left">F1</th>
<th align="left">AUC</th>
<th align="left">AP</th>
<th align="left">NDCG</th>
<th align="left">F1</th>
<th align="left">AUC</th>
<th align="left">AP</th>
<th align="left">NDCG</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">BCGP-Random</td>
<td align="left">0.397</td>
<td align="left">0.891</td>
<td align="left">0.397</td>
<td align="left">0.801</td>
<td align="left">0.500</td>
<td align="left">0.921</td>
<td align="left">0.591</td>
<td align="left">0.933</td>
</tr>
<tr>
<td align="left">BCGP-Text2vec</td>
<td align="left">0.428</td>
<td align="left">0.900</td>
<td align="left">0.425</td>
<td align="left">0.809</td>
<td align="left">0.550</td>
<td align="left">0.928</td>
<td align="left">0.650</td>
<td align="left">0.945</td>
</tr>
<tr>
<td align="left">BCGP-Doc2vec</td>
<td align="left">0.413</td>
<td align="left">0.895</td>
<td align="left">0.381</td>
<td align="left">0.780</td>
<td align="left">0.545</td>
<td align="left">0.932</td>
<td align="left">0.652</td>
<td align="left">0.944</td>
</tr>
<tr>
<td align="left">BCGP-HyenaDNA</td>
<td align="left">0.427</td>
<td align="left">0.894</td>
<td align="left">0.393</td>
<td align="left">0.789</td>
<td align="left">0.542</td>
<td align="left">0.940</td>
<td align="left">0.663</td>
<td align="left">0.948</td>
</tr>
<tr>
<td align="left">BCGP-BERT</td>
<td align="left">
<bold>0.439</bold>
</td>
<td align="left">
<bold>0.904</bold>
</td>
<td align="left">
<bold>0.435</bold>
</td>
<td align="left">
<bold>0.816</bold>
</td>
<td align="left">
<bold>0.572</bold>
</td>
<td align="left">
<bold>0.948</bold>
</td>
<td align="left">
<bold>0.712</bold>
</td>
<td align="left">
<bold>0.957</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The best results of four evaluation metrics (F1, AUC, AP, and NDCG) are highlighted in bold.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3-3">
<title>3.3 Examination of fine-tuning</title>
<p>To evaluate the performance of BCGP integrated with different fine-tuning methods, we conduct association prediction experiments using six different GNNs. Specifically, we compare GAT (<xref ref-type="bibr" rid="B31">Veli&#x10d;kovi&#x107; et al., 2017</xref>), GATv2 (<xref ref-type="bibr" rid="B4">Brody et al., 2021</xref>), FiLM (<xref ref-type="bibr" rid="B3">Brockschmidt, 2020</xref>), GraphSAGE (<xref ref-type="bibr" rid="B19">Hamilton et al., 2017</xref>), SGC (<xref ref-type="bibr" rid="B37">Wu et al., 2019</xref>), and GCN (<xref ref-type="bibr" rid="B22">Kipf and Welling, 2016</xref>) using the BCGP-BERT pre-training framework. All methods share the same embedding dimension. As detailed in <xref ref-type="table" rid="T2">Table 2</xref>, for the lncRNA-miRNA association prediction, GCN outperforms all other fine-tuning methods, achieving an F1 score of 0.439, AUC of 0.904, AP of 0.435, and NDCG of 0.816. For the circRNA-miRNA association prediction, GCN also demonstrates a robust performance with an F1 score of 0.572, AUC of 0.948, AP of 0.712, and NDCG of 0.957, while SGC closely follows with competitive performance. These results demonstrate that GCN is the most effective fine-tuning method on BCGP-BERT for predicting lncRNA-miRNA and circRNA-miRNA interactions. Further fine-tuning results on the LMA2 and CMA2 datasets are provided in <xref ref-type="sec" rid="s11">Supplementary Appendix S3</xref>.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Overall performance comparison of different fine-tuning GNN methods on lncRNA-miRNA and circRNA-miRNA association prediction tasks using the LMA1 and CMA1 datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Methods</th>
<th colspan="4" align="left">lncRNA-miRNA</th>
<th colspan="4" align="left">circRNA-miRNA</th>
</tr>
<tr>
<th align="left">Fine-tune</th>
<th align="left">F1</th>
<th align="left">AUC</th>
<th align="left">AP</th>
<th align="left">NDCG</th>
<th align="left">F1</th>
<th align="left">AUC</th>
<th align="left">AP</th>
<th align="left">NDCG</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">GAT</td>
<td align="left">0.310</td>
<td align="left">0.768</td>
<td align="left">0.211</td>
<td align="left">0.697</td>
<td align="left">0.462</td>
<td align="left">0.919</td>
<td align="left">0.527</td>
<td align="left">0.910</td>
</tr>
<tr>
<td align="left">GATv2</td>
<td align="left">0.329</td>
<td align="left">0.723</td>
<td align="left">0.187</td>
<td align="left">0.685</td>
<td align="left">0.415</td>
<td align="left">0.884</td>
<td align="left">0.430</td>
<td align="left">0.887</td>
</tr>
<tr>
<td align="left">FiLM</td>
<td align="left">0.443</td>
<td align="left">0.894</td>
<td align="left">0.367</td>
<td align="left">0.772</td>
<td align="left">0.473</td>
<td align="left">0.931</td>
<td align="left">0.612</td>
<td align="left">0.939</td>
</tr>
<tr>
<td align="left">GraphSAGE</td>
<td align="left">0.302</td>
<td align="left">0.756</td>
<td align="left">0.202</td>
<td align="left">0.691</td>
<td align="left">0.483</td>
<td align="left">0.935</td>
<td align="left">0.620</td>
<td align="left">0.941</td>
</tr>
<tr>
<td align="left">SGC</td>
<td align="left">0.399</td>
<td align="left">0.895</td>
<td align="left">0.421</td>
<td align="left">0.818</td>
<td align="left">0.571</td>
<td align="left">0.947</td>
<td align="left">0.698</td>
<td align="left">0.954</td>
</tr>
<tr>
<td align="left">GCN</td>
<td align="left">
<bold>0.439</bold>
</td>
<td align="left">
<bold>0.904</bold>
</td>
<td align="left">
<bold>0.435</bold>
</td>
<td align="left">
<bold>0.816</bold>
</td>
<td align="left">
<bold>0.572</bold>
</td>
<td align="left">
<bold>0.948</bold>
</td>
<td align="left">
<bold>0.712</bold>
</td>
<td align="left">
<bold>0.957</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The pre-training method used is BCGP-BERT. The best results of four evaluation metrics (F1, AUC, AP and NDCG) are highlighted in bold.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3-4">
<title>3.4 Method comparison</title>
<p>Next, we compare our BCGP to the state-of-the-art SPGNN (<xref ref-type="bibr" rid="B36">Wang Z. et al., 2023</xref>), which utilizes the <inline-formula id="inf113">
<mml:math id="m126">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-mer technique, Doc2Vec model and fine-tuning with GNN for RNA association prediction. Additionally, we include GCNFormer (<xref ref-type="bibr" rid="B40">Yao et al., 2024</xref>), which leverages graph convolutional networks and transformers for predicting lncRNA-disease associations, as a baseline. The experimental results in <xref ref-type="table" rid="T3">Table 3</xref> illustrate that our BCGP consistently outperforms SPGNN across all metrics for both lncRNA-miRNA and circRNA-miRNA association predictions. For lncRNA-miRNA associations, our BCGP achieves an F1 score of 0.439, AUC of 0.903, AP of 0.435, and NDCG of 0.812, outperforming SPGNN&#x2019;s scores of 0.430, 0.894, 0.419, and 0.828, respectively. Similarly, for circRNA-miRNA, BCGP obtains an F1 score of 0.572, AUC of 0.948, AP of 0.712, and NDCG of 0.957, significantly outperforming SPGNN. Additionally, we evaluate the effect of the NCN technique on the performance of BCGP. Compared to the BCGP without NCN, the integration of NCN shows improvements in both lncRNA-miRNA and circRNA-miRNA association predictions. These results demonstrate that BERT has a stronger capability to capture contextual information than the classic Doc2vec method, which allows our method to effectively capture complex and intricate relationships between RNA sequences, leading to more accurate predictions. Furthermore, integrating the NCN technique enables our method to learn more refined node representations, which ultimately enhances the prediction accuracy in complex RNA interactions.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Overall performance comparison of SPGNN and our&#xa0;BCGP method on lncRNA-miRNA-circRNA association prediction task.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left">Methods</th>
<th colspan="4" align="left">lncRNA-miRNA</th>
<th colspan="4" align="left">circRNA-miRNA</th>
</tr>
<tr>
<th align="left">F1</th>
<th align="left">AUC</th>
<th align="left">AP</th>
<th align="left">NDCG</th>
<th align="left">F1</th>
<th align="left">AUC</th>
<th align="left">AP</th>
<th align="left">NDCG</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">SPGNN&#xa0;(Wang et al., 2023b)</td>
<td align="left">0.430</td>
<td align="left">0.894</td>
<td align="left">0.419</td>
<td align="left">
<bold>0.828</bold>
</td>
<td align="left">0.403</td>
<td align="left">0.840</td>
<td align="left">0.510</td>
<td align="left">0.911</td>
</tr>
<tr>
<td align="left">GCNFormer&#xa0;(Yao et al.,&#xa0;2024)</td>
<td align="left">0.305</td>
<td align="left">0.677</td>
<td align="left">0.226</td>
<td align="left">0.686</td>
<td align="left">0.350</td>
<td align="left">0.730</td>
<td align="left">0.453</td>
<td align="left">0.815</td>
</tr>
<tr>
<td align="left">BCGP (w/o NCN)</td>
<td align="left">
<inline-formula id="inf114">
<mml:math id="m127">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mn>0.432</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2020;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<inline-formula id="inf115">
<mml:math id="m128">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mn>0.901</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2020;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<inline-formula id="inf116">
<mml:math id="m129">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mn>0.432</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2020;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">0.808</td>
<td align="left">
<inline-formula id="inf117">
<mml:math id="m130">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mn>0.492</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2020;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<inline-formula id="inf118">
<mml:math id="m131">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mn>0.919</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2020;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<inline-formula id="inf119">
<mml:math id="m132">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mn>0.577</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2020;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<inline-formula id="inf120">
<mml:math id="m133">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mn>0.929</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2020;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
</tr>
<tr>
<td align="left">BCGP (w/NCN)</td>
<td align="left">
<bold>0.439</bold>
<inline-formula id="inf121">
<mml:math id="m134">
<mml:mrow>
<mml:mo>&#x2020;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<bold>0.903</bold>
<inline-formula id="inf122">
<mml:math id="m135">
<mml:mrow>
<mml:mo>&#x2020;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<bold>0.435</bold>
<inline-formula id="inf123">
<mml:math id="m136">
<mml:mrow>
<mml:mo>&#x2020;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">0.812</td>
<td align="left">
<bold>0.572</bold>
<inline-formula id="inf124">
<mml:math id="m137">
<mml:mrow>
<mml:mo>&#x2020;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<bold>0.948</bold>
<inline-formula id="inf125">
<mml:math id="m138">
<mml:mrow>
<mml:mo>&#x2020;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<bold>0.712</bold>
<inline-formula id="inf126">
<mml:math id="m139">
<mml:mrow>
<mml:mo>&#x2020;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<bold>0.957</bold>
<inline-formula id="inf127">
<mml:math id="m140">
<mml:mrow>
<mml:mo>&#x2020;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The best results of four evaluation metrics (F1, AUC, AP, and NDCG) are highlighted in bold. In each dataset, significant improvements over the base model are marked with &#x2020; (paired t-test, p &#x003c; 0.05$).</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3-5">
<title>3.5 Analysis of hyperparameters</title>
<p>In this section, we investigate the impacts of two important parameters of the BCGP, including the <inline-formula id="inf128">
<mml:math id="m141">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-value in <inline-formula id="inf129">
<mml:math id="m142">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-mers and the RNA embedding vector size in the pre-training stage.</p>
<sec id="s3-5-1">
<title>3.5.1 Examination of k-value</title>
<p>In our BCGP method, we employ the <inline-formula id="inf130">
<mml:math id="m143">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-mers tokenization to segment RNA sequences into equal-length segments, where <inline-formula id="inf131">
<mml:math id="m144">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the segment length. To determine the impact of <inline-formula id="inf132">
<mml:math id="m145">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> value, we investigate the performance of BCGP across different <inline-formula id="inf133">
<mml:math id="m146">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> values, limiting <inline-formula id="inf134">
<mml:math id="m147">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to a maximum of 6 due to computational constraints. As illustrated in <xref ref-type="fig" rid="F2">Figure 2</xref>, the experimental results indicate that <inline-formula id="inf135">
<mml:math id="m148">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> provides the optimal performance for both lncRNA-miRNA and circRNA-miRNA association predictions. For lncRNA-miRNA associations, the <inline-formula id="inf136">
<mml:math id="m149">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> setting consistently achieves the highest F1, AP, and NDCG scores. Similarly, for circRNA-miRNA associations, <inline-formula id="inf137">
<mml:math id="m150">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> obtains the highest scores for F1 and NDCG. These results suggest that the choice of <inline-formula id="inf138">
<mml:math id="m151">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> can help BCGP capture sufficient sequence context and identify the most informative patterns for RNA association predictions.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Overall performance (F1, AUC, AP, and NDCG) of BCGP-BERT-GCN across varying k values in k-mers for lncRNA-miRNA (first row) and circRNA-miRNA (second row) association predictions.</p>
</caption>
<graphic xlink:href="fgene-16-1606016-g002.tif">
<alt-text content-type="machine-generated">Bar graphs display overall performance for different k values in k-mers. The top row represents lncRNA-miRNA with metrics: F1, AUC, AP, NDCG. The bottom row is circRNA-miRNA with similar metrics. Each graph shows performance variations from k equals one to six, highlighting different trends for each metric.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3-5-2">
<title>3.5.2 Examination of embedding size in the pre-training stage</title>
<p>After analyzing the impact of <inline-formula id="inf139">
<mml:math id="m152">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> value, we examine the effect of RNA embedding vector size within the BERT model during the pre-training stage. Due to the memory constraint, we vary the embedding size from 64 to 512 and fixed the <inline-formula id="inf140">
<mml:math id="m153">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-value in <inline-formula id="inf141">
<mml:math id="m154">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-mers at 3. As depicted in <xref ref-type="fig" rid="F3">Figure 3</xref>, the embedding size significantly impacts the performance of both lncRNA-miRNA and circRNA-miRNA association predictions. For lncRNA-miRNA, the optimal performance is achieved with an embedding size of 256. In contrast, for circRNA-miRNA, we can observe a consistent improvement across all metrics with increasing embedding sizes. The findings suggest that while larger embedding sizes tend to enhance the ability of the model to capture complex interactions, the optimal embedding size may vary between datasets.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Overall performance (F1, AUC, AP, and NDCG) of BCGP-BERT-GCN across varying embedding sizes for lncRNA-miRNA (first row) and circRNA-miRNA (second row) association predictions.</p>
</caption>
<graphic xlink:href="fgene-16-1606016-g003.tif">
<alt-text content-type="machine-generated">Bar charts showing overall performances for different RNA embedding vector sizes. The top row represents lncRNA-miRNA with metrics F1, AUC, AP, and NDCG. The bottom row represents circRNA-miRNA with the same metrics. Each chart compares embedding sizes of 64, 128, 256, and 512. Performance generally improves with larger embedding sizes across all metrics.</alt-text>
</graphic>
</fig>
</sec>
</sec>
<sec id="s3-6">
<title>3.6 Case study</title>
<p>Prediction of new circRNA-miRNA and lncRNA-miRNA interactions can reveal new biomarkers, identify therapeutic targets, and enhance understanding of the regulatory mechanisms of biological networks. To validate the practicality of our method, we select two miRNAs, namely hsa-miR-143 and hsa-miR-6808-5p, to verify the prediction results of miRNA-lncRNA and miRNA-circRNA associations generated by BCGP. External literature is employed to validate the predictions for miRNA-lncRNA associations, while associations recorded in CircBank (<xref ref-type="bibr" rid="B27">Liu et al., 2019</xref>) are used for verifying miRNA-circRNA associations. For the target miRNAs, 8 out of the top-10 predicted miRNA-lncRNA associations are confirmed in Pubmed, and 9 out of the top 10 miRNA-circRNA associations are validated in Circbank.</p>
<p>Using the BCGP model, we predict lncRNAs linked to hsa-miR-143. As illustrated in <xref ref-type="table" rid="T4">Table 4</xref>, the top-10 predicted lncRNAs are MALAT1, MEG3, NEAT1, UCA1, DANCR, HOTAIR, TUG1, GAS5, KCNQ1OT1, and MIAT, with 8 of these associations being validated in external literature. For instance, MALAT1 the top-ranked lncRNA, was shown in a study by <xref ref-type="bibr" rid="B7">Chen et al. (2017)</xref> to regulate ZEB1 expression by sponging miR-143-3p and promoting the progression of Hepatocellular Carcinoma. Additionally, <xref ref-type="bibr" rid="B10">Dong et al. (2020)</xref> has demonstrated that MEG3 overexpression inhibited LPS-induced injury in PDLCs by deactivating the AKT/IKK pathway by sponging miR-143-3p.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Top-10 Predicted lncRNAs Linked to hsa-miR-143.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Rank</th>
<th align="left">lncRNA</th>
<th align="left">PMID</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">1</td>
<td align="left">MALAT1</td>
<td align="left">28,543,721</td>
</tr>
<tr>
<td align="left">2</td>
<td align="left">MEG3</td>
<td align="left">32,520,926</td>
</tr>
<tr>
<td align="left">3</td>
<td align="left">NEAT1</td>
<td align="left">33,744,906</td>
</tr>
<tr>
<td align="left">4</td>
<td align="left">UCA1</td>
<td align="left">32,130,788</td>
</tr>
<tr>
<td align="left">5</td>
<td align="left">DANCR</td>
<td align="left">Not found</td>
</tr>
<tr>
<td align="left">6</td>
<td align="left">HOTAIR</td>
<td align="left">29,336,659</td>
</tr>
<tr>
<td align="left">7</td>
<td align="left">TUG1</td>
<td align="left">31,264,280</td>
</tr>
<tr>
<td align="left">8</td>
<td align="left">GAS5</td>
<td align="left">36,769,379</td>
</tr>
<tr>
<td align="left">9</td>
<td align="left">KCNQ1OT1</td>
<td align="left">30,691,798</td>
</tr>
<tr>
<td align="left">10</td>
<td align="left">MIAT</td>
<td align="left">Not found</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Similarly, we predict circRNAs that are potentially associated with hsa-miR-6808-5p validated by CircBank. The results presented in <xref ref-type="table" rid="T5">Table 5</xref> reveal that the top-10 ranked circRNAs as hsa_circ_0082878, hsa_circ_0020316, hsa_circ_0049111, hsa_circ_0057955, hsa_circ_0037997, hsa_circ_0000726, hsa_circ_0049109, hsa_circ_0049112, hsa_circ_0016773, and hsa_circ_0085900. Upon searching CircBank for hsa-miR-6808-5p, we find 9 out of the top 10 predicted results in the CircBank dataset. The results of the case study indicate that BCGP possesses commendable practicality.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Top-10 Predicted lncRNAs Linked to hsa-miR-6808-5p.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Rank</th>
<th align="left">lncRNA</th>
<th align="left">Evidence</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">1</td>
<td align="left">hsa_circ_0082878</td>
<td align="left">Found</td>
</tr>
<tr>
<td align="left">2</td>
<td align="left">hsa_circ_0020316</td>
<td align="left">Found</td>
</tr>
<tr>
<td align="left">3</td>
<td align="left">hsa_circ_0049111</td>
<td align="left">Found</td>
</tr>
<tr>
<td align="left">4</td>
<td align="left">hsa_circ_0057955</td>
<td align="left">Found</td>
</tr>
<tr>
<td align="left">5</td>
<td align="left">hsa_circ_0037997</td>
<td align="left">Found</td>
</tr>
<tr>
<td align="left">6</td>
<td align="left">hsa_circ_0000726</td>
<td align="left">Found</td>
</tr>
<tr>
<td align="left">7</td>
<td align="left">hsa_circ_0049109</td>
<td align="left">Found</td>
</tr>
<tr>
<td align="left">8</td>
<td align="left">hsa_circ_0049112</td>
<td align="left">Found</td>
</tr>
<tr>
<td align="left">9</td>
<td align="left">hsa_circ_0016773</td>
<td align="left">Found</td>
</tr>
<tr>
<td align="left">10</td>
<td align="left">hsa_circ_0085900</td>
<td align="left">Not Found</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec sec-type="conclusion" id="s4">
<title>4 Conclusion</title>
<p>In this article, we propose a novel method named BCGP, to leverage RNA sequence information and heterogeneous relationships to enhance the prediction of lncRNA-miRNA and circRNA-miRNA associations. To comprehensively capture contextual and structural information, BCGP integrates BERT in the pre-training stage to consider the global context of the entire sequence. To further enhance the performance of association prediction, BCGP leverages the Neural Common Neighbour technique in the fine-tuning stage to learn more informative and flexible representations. Extensive experiments on two real-world benchmark datasets demonstrate the effectiveness of our BCGP, showing that it significantly improves prediction accuracy by capturing complex interactions in both lncRNA-miRNA and circRNA-miRNA association prediction tasks compared with competitive baselines.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s11">Supplementary Material</xref>, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec sec-type="author-contributions" id="s6">
<title>Author contributions</title>
<p>ZX: Data curation, Writing &#x2013; review and editing, Methodology, Formal Analysis, Writing &#x2013; original draft, Resources, Visualization, Conceptualization. TY: Software, Investigation, Writing &#x2013; review and editing, Resources, Validation, Writing &#x2013; original draft. GJ: Formal Analysis, Conceptualization, Writing &#x2013; review and editing, Resources, Writing &#x2013; original draft. SL: Writing &#x2013; original draft, Writing &#x2013; review and editing, Methodology, Investigation, Conceptualization. JL: Formal Analysis, Writing &#x2013; original draft, Data curation, Methodology, Software, Writing &#x2013; review and editing. LT: Project administration, Methodology, Supervision, Writing &#x2013; original draft, Writing &#x2013; review and editing, Resources, Funding acquisition.</p>
</sec>
<sec sec-type="funding-information" id="s7">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. The project is supported by the Guizhou Provincial Science and Technology Program (Basic Research - ZK (2024) General 414).</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s11">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2025.1606016/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2025.1606016/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Presentation1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Alkan</surname>
<given-names>A. H.</given-names>
</name>
<name>
<surname>Akg&#xfc;l</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Endogenous mirna sponges</article-title>,&#x201d; in <source>miRNomics: MicroRNA Biology and computational analysis</source>, <fpage>91</fpage>&#x2013;<lpage>104</lpage>.</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bartel</surname>
<given-names>D. P.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Metazoan micrornas</article-title>. <source>Cell</source> <volume>173</volume>, <fpage>20</fpage>&#x2013;<lpage>51</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2018.03.006</pub-id>
<pub-id pub-id-type="pmid">29570994</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Brockschmidt</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Gnn-film: graph neural networks with feature-wise linear modulation</article-title>,&#x201d; in <source>International conference on machine learning</source> (<publisher-name>PMLR</publisher-name>), <fpage>1144</fpage>&#x2013;<lpage>1152</lpage>.</citation>
</ref>
<ref id="B4">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Brody</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Alon</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Yahav</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>How attentive are graph attention networks?</article-title>,&#x201d; in <source>International conference on learning representations</source>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Downregulation of lncrna casc2 by microrna-21 increases the proliferation and migration of renal cell carcinoma cells</article-title>. <source>Mol. Med. Rep.</source> <volume>14</volume>, <fpage>1019</fpage>&#x2013;<lpage>1025</lpage>. <pub-id pub-id-type="doi">10.3892/mmr.2016.5337</pub-id>
<pub-id pub-id-type="pmid">27222255</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Charles Richard</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Eichhorn</surname>
<given-names>P. J. A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Platforms for investigating lncrna functions</article-title>. <source>SLAS Technol. Transl. Life Sci. Innov.</source> <volume>23</volume>, <fpage>493</fpage>&#x2013;<lpage>506</lpage>. <pub-id pub-id-type="doi">10.1177/2472630318780639</pub-id>
<pub-id pub-id-type="pmid">29945466</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Long non-coding rna malat1 regulates zeb1 expression by sponging mir-143-3p and promotes hepatocellular carcinoma progression</article-title>. <source>J. Cell. Biochem.</source> <volume>118</volume>, <fpage>4836</fpage>&#x2013;<lpage>4843</lpage>. <pub-id pub-id-type="doi">10.1002/jcb.26158</pub-id>
<pub-id pub-id-type="pmid">28543721</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>T.-H.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.-C.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>C.-C.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Deep-belief network for predicting potential mirna-disease associations</article-title>. <source>Briefings Bioinforma.</source> <volume>22</volume>, <fpage>bbaa186</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbaa186</pub-id>
<pub-id pub-id-type="pmid">32866969</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Devlin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>M.-W.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Toutanova</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Bert: pre-training of deep bidirectional transformers for language understanding</article-title>. <source>arXiv Prepr</source>. <pub-id pub-id-type="doi">10.18653/v1/N19-1423</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Maternally-expressed gene 3 (meg3)/mir-143-3p regulates injury to periodontal ligament cells by mediating the akt/inhibitory <italic>&#x3ba;</italic>b kinase (ikk) pathway</article-title>. <source>Med. Sci. Monit. Int. Med. J. Exp. Clin. Res.</source> <volume>26</volume> (<issue>e922486&#x2013;1</issue>), <fpage>e922486</fpage>. <pub-id pub-id-type="doi">10.12659/MSM.922486</pub-id>
<pub-id pub-id-type="pmid">32520926</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dutta</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dubey</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>K. K.</given-names>
</name>
<name>
<surname>Anand</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Splicevec: distributed feature representations for splice junction prediction</article-title>. <source>Comput. Biol. Chem.</source> <volume>74</volume>, <fpage>434</fpage>&#x2013;<lpage>441</lpage>. <pub-id pub-id-type="doi">10.1016/j.compbiolchem.2018.03.009</pub-id>
<pub-id pub-id-type="pmid">29580738</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Fey</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lenssen</surname>
<given-names>J. E.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Fast graph representation learning with pytorch geometric</article-title>,&#x201d; in <source>ICLR 2019 (RLGM workshop)</source>.</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gawronski</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Uhl</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Y.-Y.</given-names>
</name>
<name>
<surname>Niknafs</surname>
<given-names>Y. S.</given-names>
</name>
<name>
<surname>Ramnarine</surname>
<given-names>V. R.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Mechrna: prediction of lncrna mechanisms from rna&#x2013;rna and rna&#x2013;protein interactions</article-title>. <source>Bioinformatics</source> <volume>34</volume>, <fpage>3101</fpage>&#x2013;<lpage>3110</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty208</pub-id>
<pub-id pub-id-type="pmid">29617966</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gla&#x17e;ar</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Papavasileiou</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Rajewsky</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>circbase: a database for circular rnas</article-title>. <source>Rna</source> <volume>20</volume>, <fpage>1666</fpage>&#x2013;<lpage>1670</lpage>. <pub-id pub-id-type="doi">10.1261/rna.043687.113</pub-id>
<pub-id pub-id-type="pmid">25234927</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Griffiths-Jones</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Saini</surname>
<given-names>H. K.</given-names>
</name>
<name>
<surname>Van Dongen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Enright</surname>
<given-names>A. J.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>mirbase: tools for microrna genomics</article-title>. <source>Nucleic acids Res.</source> <volume>36</volume>, <fpage>D154</fpage>&#x2013;<lpage>D158</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkm952</pub-id>
<pub-id pub-id-type="pmid">17991681</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>L.-X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>You</surname>
<given-names>Z.-H.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>C.-Q.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>M.-L.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>B.-W.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Likelihood-based feature representation learning combined with neighborhood information for predicting circrna&#x2013;mirna associations</article-title>. <source>Briefings Bioinforma.</source> <volume>25</volume>, <fpage>bbae020</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbae020</pub-id>
<pub-id pub-id-type="pmid">38324624</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ha</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Pmamca: prediction of microrna-disease association utilizing a matrix completion approach</article-title>. <source>BMC Syst. Biol.</source> <volume>13</volume>, <fpage>33</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1186/s12918-019-0700-4</pub-id>
<pub-id pub-id-type="pmid">30894171</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ha</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Imipmf: inferring mirna-disease interactions using probabilistic matrix factorization</article-title>. <source>J. Biomed. Inf.</source> <volume>102</volume>, <fpage>103358</fpage>. <pub-id pub-id-type="doi">10.1016/j.jbi.2019.103358</pub-id>
<pub-id pub-id-type="pmid">31857202</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hamilton</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Ying</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Leskovec</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Inductive representation learning on large graphs</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>30</volume>. <pub-id pub-id-type="doi">10.5555/3294771.3294869</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>J&#xe4;rvelin</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kek&#xe4;l&#xe4;inen</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Cumulated gain-based evaluation of ir techniques</article-title>. <source>ACM Trans. Inf. Syst. (TOIS)</source> <volume>20</volume>, <fpage>422</fpage>&#x2013;<lpage>446</lpage>. <pub-id pub-id-type="doi">10.1145/582415.582418</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kingma</surname>
<given-names>D. P.</given-names>
</name>
<name>
<surname>Ba</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Adam: a method for stochastic optimization</article-title>,&#x201d; in <source>International conference on learning representations</source>.</citation>
</ref>
<ref id="B22">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kipf</surname>
<given-names>T. N.</given-names>
</name>
<name>
<surname>Welling</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Semi-supervised classification with graph convolutional networks</article-title>,&#x201d; in <source>International conference on learning representations</source>.</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kristensen</surname>
<given-names>L. S.</given-names>
</name>
<name>
<surname>Jakobsen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Hager</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Kjems</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>The emerging roles of circrnas in cancer and oncology</article-title>. <source>Nat. Rev. Clin. Oncol.</source> <volume>19</volume>, <fpage>188</fpage>&#x2013;<lpage>206</lpage>. <pub-id pub-id-type="doi">10.1038/s41571-021-00585-y</pub-id>
<pub-id pub-id-type="pmid">34912049</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lan</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Kgancda: predicting circrna-disease associations based on knowledge graph attention network</article-title>. <source>Briefings Bioinforma.</source> <volume>23</volume>, <fpage>bbab494</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbab494</pub-id>
<pub-id pub-id-type="pmid">34864877</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Le</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Mikolov</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>Distributed representations of sentences and documents</article-title>,&#x201d; in <source>International conference on machine learning</source> (<publisher-name>PMLR</publisher-name>), <fpage>1188</fpage>&#x2013;<lpage>1196</lpage>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Transcriptomic analysis of high-throughput sequencing about circrna, lncrna and mrna in bladder cancer</article-title>. <source>Gene</source> <volume>677</volume>, <fpage>189</fpage>&#x2013;<lpage>197</lpage>. <pub-id pub-id-type="doi">10.1016/j.gene.2018.07.041</pub-id>
<pub-id pub-id-type="pmid">30025927</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>B. B.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Circbank: a comprehensive database for circrna with standard nomenclature</article-title>. <source>RNA Biol.</source> <volume>16</volume>, <fpage>899</fpage>&#x2013;<lpage>905</lpage>. <pub-id pub-id-type="doi">10.1080/15476286.2019.1600395</pub-id>
<pub-id pub-id-type="pmid">31023147</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mikolov</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Corrado</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Dean</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Efficient estimation of word representations in vector space</article-title>. <source>arXiv Prepr</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1301.3781</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nguyen</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Poli</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Faizi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Thomas</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wornow</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Birch-Sykes</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Hyenadna: long-range genomic sequence modeling at single nucleotide resolution</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>36</volume>. <pub-id pub-id-type="doi">10.48550/arXiv.2306.15794</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paszke</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gross</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chintala</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chanan</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>DeVito</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Automatic differentiation in pytorch</article-title>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Veli&#x10d;kovi&#x107;</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Cucurull</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Casanova</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Romero</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lio</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Graph attention networks</article-title>. <source>arXiv Prepr</source>. <pub-id pub-id-type="doi">10.17863/CAM.48429</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Volders</surname>
<given-names>P.-J.</given-names>
</name>
<name>
<surname>Anckaert</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Verheggen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Nuytens</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Martens</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Mestdagh</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Lncipedia 5: towards a reference set of human long non-coding rnas</article-title>. <source>Nucleic acids Res.</source> <volume>47</volume>, <fpage>D135</fpage>&#x2013;<lpage>D139</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gky1031</pub-id>
<pub-id pub-id-type="pmid">30371849</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhi</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2022a</year>). <article-title>Lncactdb 3.0: an updated database of experimentally supported cerna interactions and personalized networks contributing to precision medicine</article-title>. <source>Nucleic acids Res.</source> <volume>50</volume>, <fpage>D183</fpage>&#x2013;<lpage>D189</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkab1092</pub-id>
<pub-id pub-id-type="pmid">34850125</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Shuai</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022b</year>). <article-title>Predicting the potential human lncrna&#x2013;mirna interactions based on graph convolution network with conditional random field</article-title>. <source>Briefings Bioinforma.</source> <volume>23</volume>, <fpage>bbac463</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbac463</pub-id>
<pub-id pub-id-type="pmid">36305458</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023a</year>). <article-title>Neural common neighbor with completion for link prediction</article-title>. <source>arXiv Prepr</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2302.00890</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023b</year>). <article-title>Sequence pre-training-based graph neural network for predicting lncrna-mirna associations</article-title>. <source>Briefings Bioinforma.</source> <volume>24</volume>, <fpage>bbad317</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbad317</pub-id>
<pub-id pub-id-type="pmid">37651605</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Souza</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Fifty</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Weinberger</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Simplifying graph convolutional networks</article-title>,&#x201d; in <source>
<italic>International conference on machine learning</italic> (PMLR)</source>, <fpage>6861</fpage>&#x2013;<lpage>6871</lpage>.</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Circular rna circ-itch inhibits bladder cancer progression by sponging mir-17/mir-224 and regulating p21, pten expression</article-title>. <source>Mol. cancer</source> <volume>17</volume>, <fpage>19</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1186/s12943-018-0771-7</pub-id>
<pub-id pub-id-type="pmid">29386015</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Miao</surname>
<given-names>Y.-R.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Lncrnasnp v3: an updated database for functional variants in long non-coding rnas</article-title>. <source>Nucleic Acids Res.</source> <volume>51</volume>, <fpage>D192</fpage>&#x2013;<lpage>D198</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkac981</pub-id>
<pub-id pub-id-type="pmid">36350671</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yao</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zhan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Gcnformer: graph convolutional network and transformer for predicting lncrna-disease associations</article-title>. <source>BMC Bioinforma.</source> <volume>25</volume> (<issue>5</issue>), <fpage>5</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-023-05625-1</pub-id>
<pub-id pub-id-type="pmid">38166659</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ye</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Research advances in the detection of mirna</article-title>. <source>J. Pharm. analysis</source> <volume>9</volume>, <fpage>217</fpage>&#x2013;<lpage>226</lpage>. <pub-id pub-id-type="doi">10.1016/j.jpha.2019.05.004</pub-id>
<pub-id pub-id-type="pmid">31452959</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Z.-K.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Guan</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhuo</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>A newly identified lncrna mar1 acts as a mir-487b sponge to promote skeletal muscle differentiation and regeneration</article-title>. <source>J. cachexia, sarcopenia muscle</source> <volume>9</volume>, <fpage>613</fpage>&#x2013;<lpage>626</lpage>. <pub-id pub-id-type="doi">10.1002/jcsm.12281</pub-id>
<pub-id pub-id-type="pmid">29512357</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Leveraging the attention mechanism to improve the identification of dna n6-methyladenine sites</article-title>. <source>Briefings Bioinforma.</source> <volume>22</volume>, <fpage>bbab351</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbab351</pub-id>
<pub-id pub-id-type="pmid">34459479</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Noncodev6: an updated database dedicated to long non-coding rna annotation in both animals and plants</article-title>. <source>Nucleic acids Res.</source> <volume>49</volume>, <fpage>D165</fpage>&#x2013;<lpage>D171</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkaa1046</pub-id>
<pub-id pub-id-type="pmid">33196801</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Mo</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Xiong</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Circular rnas function as cernas to regulate and control human cancer progression</article-title>. <source>Mol. cancer</source> <volume>17</volume>, <fpage>79</fpage>&#x2013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1186/s12943-018-0827-8</pub-id>
<pub-id pub-id-type="pmid">29626935</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Lnc-mg is a long non-coding rna that promotes myogenesis</article-title>. <source>Nat. Commun.</source> <volume>8</volume>, <fpage>14718</fpage>. <pub-id pub-id-type="doi">10.1038/ncomms14718</pub-id>
<pub-id pub-id-type="pmid">28281528</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>