<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="methods-article" dtd-version="1.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title-group>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1643921</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2025.1643921</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Methods</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>OFGPMA: Optimal frequency graph representation learning for pseudogene and miRNA association prediction</article-title>
<alt-title alt-title-type="left-running-head">Zeng et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2025.1643921">10.3389/fgene.2025.1643921</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Zeng</surname>
<given-names>Yongbin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3094209"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xiong</surname>
<given-names>Lixiang</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Luo</surname>
<given-names>Yungui</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
</contrib-group>
<aff id="aff1">
<label>1</label>
<institution>College Information Science and Engineering, Wuchang Shouyi University</institution>, <city>Wuhan</city>, <country country="CN">China</country>
</aff>
<aff id="aff2">
<label>2</label>
<institution>College of Information Engineering, Wuhan Huaxia Institute of Technology</institution>, <city>Wuhan</city>, <country country="CN">China</country>
</aff>
<author-notes>
<corresp id="c001">
<label>&#x2a;</label>Correspondence: Yungui Luo, <email xlink:href="2020111019@wsyu.edu.cn">2020111019@wsyu.edu.cn</email>
</corresp>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2025-11-26">
<day>26</day>
<month>11</month>
<year>2025</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1643921</elocation-id>
<history>
<date date-type="received">
<day>09</day>
<month>06</month>
<year>2025</year>
</date>
<date date-type="rev-recd">
<day>16</day>
<month>10</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>04</day>
<month>11</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Zeng, Xiong and Luo.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Zeng, Xiong and Luo</copyright-holder>
<license>
<ali:license_ref start_date="2025-11-26">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<p>Pseudogenes are genomic segments that resemble functional genes structurally yet remain biologically inactive. MicroRNAs (miRNAs), a subclass of non-coding RNAs, are critical regulators of various cellular mechanisms. These pseudogenes and miRNAs interact mutually, forming competitive endogenous RNA (ceRNA) networks alongside mRNA to influence physiological processes. Such regulatory networks have been implicated in numerous pathological conditions. Consequently, investigating pseudogene-miRNA associations holds promise for advancing disease diagnostics. Nevertheless, existing approaches to identify these relationships predominantly rely on labor-intensive experimental techniques, demanding substantial time and financial investments. Consequently, developing an effective computational framework that can identify new pseudogene-miRNA associations (PMAs) is crucial. To this end, we propose an optimal frequency graph representation learning framework named OFGPMA, for pseudogene-miRNA association prediction. OFGPMA enhances graph neural network expressiveness by learning both high-frequency energy and low-frequency energy components within the pseudogene-miRNA bipartite graph, utilizing Rayleigh and Chebyshev pooling techniques. This approach captures the graph&#x2019;s global topology via Random Walk with Restart (RWR) and identifies potential local substructure features through enclosing subgraph analysis, thereby achieving a more comprehensive integration of the entire graph information. Comprehensive experiments show that OFGPMA outperforms state-of-the-art methods in terms of performance, while also exhibiting excellent generalization capabilities.</p>
</abstract>
<kwd-group>
<kwd>optimal frequency graph</kwd>
<kwd>global random walk with restart</kwd>
<kwd>local enclosing subgraph</kwd>
<kwd>graph representation learning</kwd>
<kwd>pseudogene and miRNA association prediction</kwd>
</kwd-group>
<funding-group>
<funding-statement>The authors declare that no financial support was received for the research and/or publication of this article.</funding-statement>
</funding-group>
<counts>
<fig-count count="7"/>
<table-count count="5"/>
<equation-count count="7"/>
<ref-count count="51"/>
<page-count count="15"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Computational Genomics</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<label>1</label>
<title>Introduction</title>
<p>Pseudogenes, also known as false genes, are non-functional remnants formed during the evolution of gene families (<xref ref-type="bibr" rid="B2">Carninci et al., 2005</xref>; <xref ref-type="bibr" rid="B33">Shi et al., 2016</xref>). They are similar to normal genes but are DNA sequences that have lost their normal functions and are often found in multi-gene families of eukaryotes (<xref ref-type="bibr" rid="B32">Setoyama et al., 2011</xref>; <xref ref-type="bibr" rid="B24">Ma et al., 2021</xref>). MiRNA is one type of non-coding RNA, with lengths between 19 and 25 nucleotides, and they account for roughly 3% of the genome (<xref ref-type="bibr" rid="B13">Hydbring and Badalian-Very, 2013</xref>; <xref ref-type="bibr" rid="B22">Liu et al., 2016</xref>). Predicting the correlation between the two is of crucial significance for revealing gene regulatory networks, disease mechanisms and the development of precision medicine (<xref ref-type="bibr" rid="B49">Zhang et al., 2012</xref>; <xref ref-type="bibr" rid="B34">Stiegelbauer et al., 2014</xref>). A large number of studies have demonstrated that pseudogenes and miRNAs interact with each other and, together with mRNA, form a ceRNA network. This network plays a role in regulating biological processes and is associated with various diseases. Predicting pseudogene-miRNA associations can provide advisory treatment plans for some difficult and complicated diseases (<xref ref-type="bibr" rid="B30">Salmena et al., 2011</xref>; <xref ref-type="bibr" rid="B28">Rutnam et al., 2014</xref>; <xref ref-type="bibr" rid="B15">Karreth et al., 2015</xref>).</p>
<p>Present miRNA-related databases only offer fundamental information about miRNAs, such as their target genes and genomic locations. Details regarding their connections to diseases, which are crucial for understanding disease mechanisms, are often overlooked. Thankfully, some researchers have begun to recognize the significance of pseudogene&#x2013;miRNA associations (PMAs) and have compiled the currently known associations into databases. For example, starBase v2.0 (<xref ref-type="bibr" rid="B18">Li et al., 2014</xref>) includes 444 pseudogenes and 173 miRNAs, which permits the exploration of their interactions through computational approaches. However, most discoveries of PMA are dependent on biological experiments that are not only time-intensive and resource-demanding but also constrained by the limited number of confirmed PMAs. On the other hand, predicting novel associations between pseudogenes and miRNAs via computational methods facilitates screening of potential PMAs.</p>
<p>Graph signal processing (GSP) adapts signal processing concepts to graphs, encompassing operations such as sampling, convolution, and filtering in the spectral domain. Graph signals are defined as numerical or vector values on graph nodes (<xref ref-type="bibr" rid="B25">Ortega et al., 2018</xref>; <xref ref-type="bibr" rid="B12">Hu et al., 2022</xref>). To analyze these signals, GSP employs spectral decomposition of either the graph Laplacian or adjacency matrix, revealing their spectral characteristics. These characteristics describe how signal energy is distributed among various frequency components inherent to the graph&#x2019;s topology (<xref ref-type="bibr" rid="B6">Dong et al., 2020</xref>). The spectral characteristics essentially describe the degree of fit between the graph signal and the graph topology: low-frequency energy information corresponds to a globally smooth signal distribution (similar values at adjacent nodes), while high-frequency energy information corresponds to local abrupt fluctuations in the signal (differences in values at adjacent nodes) (<xref ref-type="bibr" rid="B31">Sandryhaila and Moura, 2013</xref>; <xref ref-type="bibr" rid="B10">Gavili and Zhang, 2015</xref>; <xref ref-type="bibr" rid="B27">Ramakrishna et al., 2020</xref>). Recently, the concept of graph signal processing has found extensive use in the field of biological networks, mainly focusing on using graph structures to model and conduct in depth analysis of complex biological systems. For example, Peng et al. modeled the drug response of cancer cells as a hypergraph, and simultaneously applied low-frequency component and high-frequency components filters to the hypergraph, effectively extracting both common and differential features among the hypergraph nodes (<xref ref-type="bibr" rid="B26">Peng et al., 2025</xref>).</p>
<p>Current computational approaches leveraging similarity networks in biological applications commonly adopt a key assumption: given a known interaction between pseudogene and miRNA, functionally or structurally similar pseudogenes may also engage with correspondingly similar miRNAs. For example. Zhou et al. integrated pseudogene expression data and miRNA sequence features to construct three similarity networks, namely Jaccard, Cosine, and Pearson, and used Graph Autoencoder (GAE) to aggregate node features and network topological relationships to generate low-dimensional embedding representations (<xref ref-type="bibr" rid="B51">Zhou et al., 2021</xref>). Despite its available predictive performance, PMGAE merely utilizes the structural information of the graph itself and does not treat label information as supervisory signals, resulting in its single train pattern as a non-end-to-end mode. More importantly, the GAE within PMGAE is often limited by the vanilla GCN with two layers, making it challenging for it to aggregate the node features and topological. Moreover, PMGAE adopts pseudogene and miRNA similarity networks, but the similarity assumption maybe does not hold in the association network of pseudogenes and miRNAs. A widespread biological consensus is that minor nucleotide differences can lead to significant variations in the functions of the proteins transcribed and translated from them, which often casts doubt on the availability of the similarity assumption in biological networks.</p>
<p>Recently, owing to its superior performance in graph representation learning, subgraph-based GRL (SGRL) has become a representative method for link prediction (<xref ref-type="bibr" rid="B9">Frasca et al., 2025</xref>; <xref ref-type="bibr" rid="B42">Wu et al., 2025</xref>; <xref ref-type="bibr" rid="B46">Zeng et al., 2025</xref>; <xref ref-type="bibr" rid="B1">Bouritsas et al., 2020</xref>). Unlike prediction models based on the similarity assumption (such as PMGAE), SGRL only extracts closed subgraphs in bipartite graphs and overcame the limitations of similarity assumption (<xref ref-type="bibr" rid="B47">Zhang and Chen, 2019</xref>; <xref ref-type="bibr" rid="B36">Teru et al., 2020</xref>). For instance, Zhang et al. proposed a link prediction model SEAL on the basis of graph neural networks (GNN), which automatically learns heuristic features from local closed subgraphs to address the limitations of traditional predefined heuristic methods (<xref ref-type="bibr" rid="B48">Zhang and Chen, 2025</xref>). Motivated by this method, Xu et al. put forward a subgraph-based model and applied it to enhance the prediction of associations between enhancers and diseases, further improving the accuracy of candidate disease-related enhancers by capturing local closed subgraphs of enhancers and diseases (<xref ref-type="bibr" rid="B44">Xu et al., 2024</xref>). Wang et al. introduced an innovative method called KnowDDI for predicting drug-drug interactions (DDI). It can adaptively extract and optimize subgraphs related to specific drug pairs, thereby enhancing prediction accuracy and interpretability (<xref ref-type="bibr" rid="B38">Wang J. et al., 2023</xref>). Wang et al. proposed a meta-learning-based zero-shot drug-target interaction (DTI) prediction framework for proteins, with its core innovation being the introduction of a weakly supervised subgraph information bottleneck module. This method relies solely on global DTI labels and does not require pocket annotations. It can identify key subgraphs in protein structures as potential binding pockets by dynamically learning the node allocation matrix (<xref ref-type="bibr" rid="B41">Wang Y. et al., 2023</xref>). Swarnkar et al. proposed a method that integrates gene expression data with protein-protein interaction networks (PPI) to identify key disease-related gene modules by recognizing dense subgraphs (<xref ref-type="bibr" rid="B35">Swarnkar et al., 2015</xref>). These methods have all demonstrated the effectiveness of local subgraphs and the non-essentiality of the similarity assumption.</p>
<p>The above-mentioned methods overcome the limitations of similarity-based networks. Their inductive approach uses closed subgraphs to adaptively learn the local neighborhood subgraph information of the target node. However, from the perspective of extracting information from graph structure, the main limitation of their method lies in its insufficient capture of global topological features. Although relevant theories have demonstrated that local subgraphs can approximate high-order heuristics, its core mechanism still relies on the preset h-hop closed subgraph, which is essentially a compromise of a local perspective.</p>
<p>To this end, we introduce a novel optimal frequency graph representation learning for pseudogenes and miRNA interactions prediction (OFGPMA) to address the above problems. Our model consists of two modules: the optimal frequency discovery (OFD) module and the graph representation learning (GRL) module. To enhances the expressive power of graph neural networks, The OFD learn the optimal frequency energy features of graphs through aligning the high-frequency components and low-frequency components information of the graphs. Specifically, OFD explicitly enhances the high-frequency components information in the bipartite graph of pseudogenes and miRNAs through Rayleigh pooling, thereby accurately capturing the key features of the graph nodes. Meanwhile, it implicitly extracts the low-frequency components information of the graph through Chebyshev pooling, generating important representations that reflect the commonalities of each node. Ultimately, by fusing the high-frequency and low-frequency energy information, it simultaneously learns the difference and commonality information of the graph, while resulting in a fused graph with optimal frequency structure. Then, the GRL uses the fused graph for graph representation learning. We use the graph extracted by the random walk with restart (RWR) as the explicit topological structure and the topology subgraph obtained through enclosed subgraph representation learning as the corresponding latent substructure, with the goal of accommodating explicit global topology. In detail, we use the RWR algorithm to globally extract the full graph representation of pseudogenes as explicit topological features, and simultaneously extract the enclosed subgraph features of miRNAs as implicit substructure features. We then fuse the global features of pseudogenes with the local features of miRNAs. Through this method, we can not only overcome the limitations of single local features, but also effectively combine and balance global and local features. In summary, the key contributions of OFGPMA can be outlined as follows:<list list-type="simple">
<list-item>
<p>&#x2022; The OFD focuses on the processing and optimization of graph signals in the frequency domain. By employing an original high-frequency/low-frequency separation, enhancement, and fusion strategy, it generates an optimal frequency graph structure, which significantly enhances the capability of node feature representation.</p>
</list-item>
<list-item>
<p>&#x2022; The GRL focuses on comprehensively utilizing graph topological information by integrating topological features at two distinct scales: global (RWR) and local (enclosing subgraph). This approach overcomes the limitations of a single perspective, thereby achieving more comprehensive network structure modeling.</p>
</list-item>
<list-item>
<p>&#x2022; The superior performance of OFGPMA is validated through comprehensive experiments. The importance of every component within the model is substantiated by ablation tests. Furthermore, case studies reveal OFGPMA&#x2019;s capability to detect previously unknown pseudogene-miRNA interactions.</p>
</list-item>
</list>
</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<label>2</label>
<title>Materials and methods</title>
<sec id="s2-1">
<label>2.1</label>
<title>Data collection</title>
<p>Currently, the only database that records the association between pseudogenes and miRNAs is starBase v2.0 (<xref ref-type="bibr" rid="B18">Li et al., 2014</xref>). We get the association data of pseudogene-miRNA pairs from the starBase database and preprocessed it using the same data processing method as Zhou et al. Ultimately, we obtained the data including 444 pseudogenes, 173 miRNAs and 1,884 pseudogene-miRNA pairs.</p>
</sec>
<sec id="s2-2">
<label>2.2</label>
<title>Overview of OFGPMA</title>
<p>Firstly, we set pseudogene-miRNA association pairs as a bipartite graph <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mi>V</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3be;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is node set and <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mi>&#x3be;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is edge set. Specifically, <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> includes pseudogene node <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and miRNA node <inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>A</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:mi>&#x3be;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> includes pseudogene-miRNA association pairs. Then, we introduce an optimal frequency graph representation learning framework named OFGPMA to infer novel PMAs (<xref ref-type="fig" rid="F1">Figure 1</xref>). Our model mainly comprises of two parts: 1) optimal frequency discovery, which includes Rayleigh pooling and Chebyshev Pooling around the pair (<inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>); 2) graph representation learning, which employs graph-level GNN to learning the embeddings of local enclosing subgraph and global RWR graph.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>(1) The structure of optimal frequency discovery (top). (2) The structure of graph representation learning (bottom). ES denotes enclosing subgraph, AS denotes auxiliary graph and RWR denotes random walk restart.</p>
</caption>
<graphic xlink:href="fgene-16-1643921-g001.tif">
<alt-text content-type="machine-generated">Diagram illustrating a two-part process: &#x22;Optimal Frequency Discovery&#x22; and &#x22;Graph Representation Learning.&#x22; The top section shows high and low-frequency features extraction from a pseudogene-miRNA graph using Rayleigh and Chebyshev pooling layers, followed by contrastive learning to produce a feature fusion graph. The bottom section describes the use of this feature fusion graph for graph representation learning through three methods: enclosing subgraph (ES), auxiliary graph (AS), and random walk restart (RWR), each using loss to refine the graph outputs.</alt-text>
</graphic>
</fig>
<p>The main notations used in this paper are summarized in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Main notations used in this study.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Notation</th>
<th align="left">Description</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">miRNAs</td>
<td align="left">microRNAs</td>
</tr>
<tr>
<td align="left">ceRNA</td>
<td align="left">Competitive endogenous RNA</td>
</tr>
<tr>
<td align="left">PMA</td>
<td align="left">Pseudogene-miRNA association</td>
</tr>
<tr>
<td align="left">GSP</td>
<td align="left">Graph signal processing</td>
</tr>
<tr>
<td align="left">GNN</td>
<td align="left">Graph neural network</td>
</tr>
<tr>
<td align="left">Graph Representation Learning</td>
<td align="left">GRL</td>
</tr>
<tr>
<td align="left">SGRL</td>
<td align="left">Subgraph-based GRL</td>
</tr>
<tr>
<td align="left">OFD</td>
<td align="left">Optimal frequency discovery</td>
</tr>
<tr>
<td align="left">RWR</td>
<td align="left">Random walk with restart</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-3">
<label>2.3</label>
<title>Node representation</title>
<p>MiRNA sequence data is represented as a string composed of four nucleotides. In this paper, we use k-mer to represent miRNA sequences as a 64-dimensional feature vector, where <inline-formula id="inf10">
<mml:math id="m10">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. Similarly, pseudogenes are processed in the same way. The final feature matrix dimension of pseudogenes (P) is 444 <inline-formula id="inf11">
<mml:math id="m11">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 64, and that of miRAN (M) is <inline-formula id="inf12">
<mml:math id="m12">
<mml:mrow>
<mml:mn>173</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>64</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. For specific details, please refer to <xref ref-type="sec" rid="s12">Supplementary Materia Section 1</xref>.</p>
</sec>
<sec id="s2-4">
<label>2.4</label>
<title>Optimal frequency graph discovery</title>
<p>In the OFG module, the Rayleigh pooling is proposed to extract the high-frequency energy information of the pseudogene and miRNA bipartite graph, and use the Chebyshev wavelet transform to learn the low-frequency energy information of the bipartite graph. Subsequently, by integrating the high-frequency and low-frequency energy features while jointly capturing the distinct and shared patterns within the graph, we derive a fused graph that exhibits the most favorable frequency configuration. Through this approach, OFG can significantly enhance the GNN&#x2019;s ability to express graph structure information.</p>
<sec id="s2-4-1">
<label>2.4.1</label>
<title>Rayleigh pooling</title>
<p>The Rayleigh Quotient is an important concept in signal processing, used to characterize the energy distribution of graph signals on the Laplacian matrix (<xref ref-type="bibr" rid="B17">Li, 2004</xref>). Specifically, the Rayleigh Quotient reflects the weighted cumulative energy of graph signals at all frequencies. To explicitly extract the high-frequency component spectral information of the graph, we improved the method in (<xref ref-type="bibr" rid="B7">Dong et al., 2023</xref>) by introducing two parameters <inline-formula id="inf13">
<mml:math id="m13">
<mml:mrow>
<mml:mi>&#x3d1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf14">
<mml:math id="m14">
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, thereby enabling Rayleigh pooling to be more inclined to capture high-frequency energy information and assign higher weights to high-frequency features. In this way, not only can the contribution of high-frequency components be amplified, but also features containing high-frequency energy information can be effectively distinguished. Through this method, the high-frequency features <inline-formula id="inf15">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold-italic">R</mml:mi>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> of the entire graph can be extracted. For specific details, please refer to <xref ref-type="sec" rid="s12">Supplementary Materia Section 2</xref>.</p>
<p>The Rayleigh Pooling helps to identify significant changes within the graph structure and improves the model&#x2019;s capacity for PMA prediction by emphasizing the signal components with the greatest information content.</p>
</sec>
<sec id="s2-4-2">
<label>2.4.2</label>
<title>Chebyshev pooling</title>
<p>Meanwhile, the Chebyshev Wavelet Transform (CWT) is designed to extract the low-frequency features of graphs. The Chebyshev Wavelet Transform is an efficient multi-scale graph signal analysis tool. Its main objective is to capture the multi-band energy characteristics in the graph structure while avoiding the computational bottlenecks existing in traditional spectral methods (<xref ref-type="bibr" rid="B8">Du et al., 2017</xref>). As a spectral domain filtering method based on polynomial approximation, the core idea of the Chebyshev Wavelet Transform lies in designing multiple wavelet filters to cover different frequency ranges. The Chebyshev wavelet transform realizes a learnable low-pass component filter through polynomial approximation. The specific implementation details are provided in <xref ref-type="sec" rid="s12">Supplementary Materia Section 3</xref>. Through this method, we can ultimately obtain the low-frequency features <inline-formula id="inf16">
<mml:math id="m16">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold-italic">C</mml:mi>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi mathvariant="bold-italic">T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> of the entire graph.</p>
</sec>
<sec id="s2-4-3">
<label>2.4.3</label>
<title>Information fusion</title>
<p>After undergoing Rayleigh pooling and Chebyshev pooling, we can obtain the high-frequency energy information <inline-formula id="inf17">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>Q</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and low-frequency energy information <inline-formula id="inf18">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> of the pseudogene and miRNA network. Then, we use information fusion strategy calculated as the embeddings <italic>Embedding:</italic>
<disp-formula id="e1">
<mml:math id="m19">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c0;</mml:mi>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>Q</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <inline-formula id="inf19">
<mml:math id="m20">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the scaling factor and indicates that the model focuses on energy information of different frequencies.</p>
</sec>
</sec>
<sec id="s2-5">
<label>2.5</label>
<title>Graph representation learning</title>
<p>For the graph representation learning of pseudogenes-miRNA pair, there are three steps: 1) miRNAs subgraph extraction. For miRNAs, a closed subgraph representation learning based on local structural features is adopted. 2) RWR graph extraction. For pseudogenes, a random walk restart (RWR) method based on global structural attributes is used. 3) encoder layer. GNN is employed to generate the embeddings of the extracted graph representations, and information fusion is conducted to obtain concise edge embeddings.</p>
<sec id="s2-5-1">
<label>2.5.1</label>
<title>miRNA subgraph extraction</title>
<p>For miRNA, we adopt a closed subgraph representation learning based on local structural features. The extraction of the closed subgraph of miRNA can be divided into two steps: First, construct the main graph <inline-formula id="inf20">
<mml:math id="m21">
<mml:mrow>
<mml:msubsup>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> with the miRNA nodes as the starting points; second, based on the pseudogene auxiliary nodes related to the pseudogenes, extract the auxiliary subgraph <inline-formula id="inf21">
<mml:math id="m22">
<mml:mrow>
<mml:msubsup>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>. Finally, the two subgraphs are merged to jointly form a local closed subgraph for miRNA <inline-formula id="inf22">
<mml:math id="m23">
<mml:mrow>
<mml:msubsup>
<mml:mi>G</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>&#x222a;</mml:mo>
<mml:msubsup>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>. Starting from the miRNAs, we iteratively expand the pseudogene nodes within 1-hop and 3-hop to form the closed subgraphs <inline-formula id="inf23">
<mml:math id="m24">
<mml:mrow>
<mml:msubsup>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> of the miRNAs. For instance, for the path (<italic>m</italic>&#x2192;<italic>p</italic>1&#x2192;<italic>m</italic>1&#x2192;<italic>p</italic>) and (<italic>m</italic>&#x2192;<italic>p</italic>1&#x2192; <italic>m</italic>1&#x2192;<italic>p</italic>2&#x2192;<italic>m</italic>2&#x2192;<italic>p</italic>3&#x2192;<italic>m</italic>), starting from miRNA nodes, extract the adjacent pseudogene nodes to construct a local closed subgraph. It can be observed that in both of these two paths, the odd-numbered jump neighbor nodes of miRNA are all pseudogenes. <xref ref-type="statement" rid="Algorithm_1">Algorithm 1</xref> builds a local subgraph by iteratively expanding the k-hop neighbors of pseudogene node preserving the topological structure closely related to the target while avoiding the interference of the target edge on the prediction. This process provides the subsequent graph neural network with rich semantic local context information.</p>
<p>Finally, <xref ref-type="statement" rid="Algorithm_1">Algorithm 1</xref> integrates motif path information to enhance local topological coverage. After iterating <xref ref-type="statement" rid="Algorithm_1">Algorithm 1</xref> for R times, the subgraph range is gradually expanded to ensure full coverage of the high-order neighbors of the target node.<disp-formula id="e2">
<mml:math id="m25">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x230a;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:msqrt>
<mml:mi>e</mml:mi>
</mml:msqrt>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:mfrac>
<mml:mo>&#x230b;</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <inline-formula id="inf24">
<mml:math id="m26">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf25">
<mml:math id="m27">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represent the count of nodes and the count of edges in a bipartite graph <italic>G</italic>, respectively.</p>
<p>
<statement content-type="algorithm" id="Algorithm_1">
<label>Algorithm 1</label>
<p>Enclosing Subgraph extraction.<list list-type="simple">
<list-item>
<p>1: <bold>Input:</bold> bipartite graph <italic>G</italic>, pseudogene-miRNA pair (m, p), the count of <italic>k</italic>
</p>
</list-item>
<list-item>
<p>2: <bold>Output:</bold> enclosing subgraph <inline-formula id="inf26">
<mml:math id="m28">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">p</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi mathvariant="bold-italic">k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> or auxiliary subgraph <inline-formula id="inf27">
<mml:math id="m29">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="bold-italic">a</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">p</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi mathvariant="bold-italic">k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> about pseudogene-miRNA pair (m, p)</p>
</list-item>
<list-item>
<p>3: <inline-formula id="inf28">
<mml:math id="m30">
<mml:mrow>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">m</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">p</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>4: <bold>for</bold> <inline-formula id="inf29">
<mml:math id="m31">
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> <bold>do</bold>
</p>
</list-item>
<list-item>
<p>5: Find all new miRNA nodes set <inline-formula id="inf30">
<mml:math id="m32">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">M</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold-italic">n</mml:mi>
<mml:mi mathvariant="bold-italic">e</mml:mi>
<mml:mi mathvariant="bold-italic">w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> directly connected to the current pseudogene set P, excluding existing nodes.</p>
</list-item>
<list-item>
<p>6: Find all new pseudogene nodes set <inline-formula id="inf31">
<mml:math id="m33">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold-italic">n</mml:mi>
<mml:mi mathvariant="bold-italic">e</mml:mi>
<mml:mi mathvariant="bold-italic">w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> directly connected to the current miRNA set M, excluding existing nodes.</p>
</list-item>
<list-item>
<p>7: P &#x3d; P <inline-formula id="inf32">
<mml:math id="m34">
<mml:mrow>
<mml:mo>&#x222a;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold-italic">n</mml:mi>
<mml:mi mathvariant="bold-italic">e</mml:mi>
<mml:mi mathvariant="bold-italic">w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>8: M &#x3d; M <inline-formula id="inf33">
<mml:math id="m35">
<mml:mrow>
<mml:mo>&#x222a;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">M</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold-italic">n</mml:mi>
<mml:mi mathvariant="bold-italic">e</mml:mi>
<mml:mi mathvariant="bold-italic">w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>9: Construct subgraph <inline-formula id="inf34">
<mml:math id="m36">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">p</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi mathvariant="bold-italic">k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> or <inline-formula id="inf35">
<mml:math id="m37">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="bold-italic">a</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">p</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi mathvariant="bold-italic">k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> by utilizing node sets <inline-formula id="inf36">
<mml:math id="m38">
<mml:mrow>
<mml:mi mathvariant="bold-italic">P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf37">
<mml:math id="m39">
<mml:mrow>
<mml:mi mathvariant="bold-italic">M</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>10: <bold>end</bold>
</p>
</list-item>
<list-item>
<p>11: Remove edge (m, p) need to be predicted from <inline-formula id="inf38">
<mml:math id="m40">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">p</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi mathvariant="bold-italic">k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> or <inline-formula id="inf39">
<mml:math id="m41">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="bold-italic">a</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">p</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi mathvariant="bold-italic">k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
</list>
</p>
</statement>
</p>
</sec>
<sec id="s2-5-2">
<label>2.5.2</label>
<title>RWR graph extraction</title>
<p>To analyze pseudogenes, we employ a random walk with restart (RWR) approach that utilizes global network topology. The algorithm initiates traversal from pseudogene nodes, systematically identifying miRNA nodes located at odd-hop distances. This process progressively extends to encompass all miRNA nodes within the complete network, enabling comprehensive characterization of pseudogene relationships across the entire graph:<disp-formula id="e3">
<mml:math id="m42">
<mml:mrow>
<mml:mi>&#x3c1;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mi>A</mml:mi>
<mml:msup>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mi>&#x3c1;</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>where <inline-formula id="inf40">
<mml:math id="m43">
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is restart probability, <inline-formula id="inf41">
<mml:math id="m44">
<mml:mrow>
<mml:mi>&#x3c1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is adaptive parameters with <inline-formula id="inf42">
<mml:math id="m45">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c1;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denoting the probability at node <italic>i.</italic> For miRNA nodes, since RWR samples the pseudogene-associated miRNA nodes, h-hops is an odd number, ensuring that each sampled node is a miRNA. The restart probability represents that the probability of choosing a neighbor for the next hop is c, and the probability of returning to the starting point is (1-c). <inline-formula id="inf43">
<mml:math id="m46">
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes starting vector and if <italic>i</italic> is starting node, <inline-formula id="inf44">
<mml:math id="m47">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is set 1 else set 0. Thus, the starting vector <italic>e</italic> allows us to preserve the node&#x2019;s local topological structure and <inline-formula id="inf45">
<mml:math id="m48">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:msup>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> allows us to further visit their neighborhoods. After RWR graph extraction, we can obtain a global graph <inline-formula id="inf46">
<mml:math id="m49">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> from pseudogene sampling.</p>
</sec>
</sec>
<sec id="s2-6">
<label>2.6</label>
<title>Encoder layer</title>
<p>To cover the neighborhood information of both the local encolsing subgraph and the RWR global graph, we merge the two graphs <inline-formula id="inf47">
<mml:math id="m50">
<mml:mrow>
<mml:msubsup>
<mml:mi>G</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf48">
<mml:math id="m51">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Next, we use two layers of GCN to learn topological features for <inline-formula id="inf49">
<mml:math id="m52">
<mml:mrow>
<mml:msubsup>
<mml:mi>G</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf50">
<mml:math id="m53">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Finally, we can get embeddings <inline-formula id="inf51">
<mml:math id="m54">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold-italic">n</mml:mi>
<mml:mi mathvariant="bold-italic">e</mml:mi>
<mml:mi mathvariant="bold-italic">w</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. For specific details, please refer to <xref ref-type="sec" rid="s12">Supplementary Materia Section 4</xref>.</p>
</sec>
<sec id="s2-7">
<label>2.7</label>
<title>Model optimization</title>
<p>The contrastive learning loss function is used to calculate the gap between <inline-formula id="inf52">
<mml:math id="m55">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>Q</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf53">
<mml:math id="m56">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>:<disp-formula id="e4">
<mml:math id="m57">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mi mathvariant="italic">log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi mathvariant="normal">&#x393;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>Q</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mrow>
<mml:mi>log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="normal">&#x393;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>Q</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>where <inline-formula id="inf54">
<mml:math id="m58">
<mml:mrow>
<mml:mi mathvariant="normal">&#x393;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> () is the contrastive discriminator constructed by a simple bilinear function that estimates similarities between <inline-formula id="inf55">
<mml:math id="m59">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>Q</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf56">
<mml:math id="m60">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. We use Kullback-Leibler (KL) divergence to calculate loss between <inline-formula id="inf57">
<mml:math id="m61">
<mml:mrow>
<mml:msup>
<mml:mi>Z</mml:mi>
<mml:mi>p</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf58">
<mml:math id="m62">
<mml:mrow>
<mml:msup>
<mml:mi>Z</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>:<disp-formula id="e5">
<mml:math id="m63">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>K</mml:mi>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>Z</mml:mi>
<mml:mi>m</mml:mi>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>Z</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2211;</mml:mo>
<mml:msub>
<mml:mi>log</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mfrac>
<mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msup>
<mml:mi>Z</mml:mi>
<mml:mi>m</mml:mi>
</mml:msup>
</mml:mrow>
<mml:msup>
<mml:mi>Z</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
<p>The binary cross-entropy loss is employed to optimize OFGPMA:<disp-formula id="e6">
<mml:math id="m64">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>Y</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mover accent="true">
<mml:mi>Y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>Y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>where <italic>N</italic> is the number of all pseudogene-miRNA pairs in the batch. <inline-formula id="inf59">
<mml:math id="m65">
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf60">
<mml:math id="m66">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>Y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> are the ground truth and prediction score, respectively. Coupled with the <inline-formula id="inf61">
<mml:math id="m67">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf62">
<mml:math id="m68">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, OFGPMA can be trained by minimizing the final loss which can be calculated as:<disp-formula id="e7">
<mml:math id="m69">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>where &#x3b1; and &#x3b2; are learnable parameters. The pseudo-code of OFGPMA as follows:</p>
<p>
<statement content-type="algorithm" id="Algorithm_2">
<label>Algorithm 2</label>
<p>OFGPMA train description.<list list-type="simple">
<list-item>
<p>1: <bold>Input:</bold> training set pseudogene-miRNA pairs, k-hops;</p>
</list-item>
<list-item>
<p>2: <bold>Output:</bold> the convergent training model OFGPMA;</p>
</list-item>
<list-item>
<p>3: Randomly initialize model parameters;</p>
</list-item>
<list-item>
<p>4: Construct a bipartite graph G;</p>
</list-item>
<list-item>
<p>5: <bold>Repeat</bold>
</p>
</list-item>
<list-item>
<p>6: Generate a fused graph Gf by <xref ref-type="disp-formula" rid="e1">Equation 1</xref> and supplementary materials Equations 1&#x2013;9 from G;</p>
</list-item>
<list-item>
<p>7: Samples miRNA enclosing subgraph and pseudogene RWR graph from <inline-formula id="inf64">
<mml:math id="m71">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mi>f</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>;</p>
</list-item>
<list-item>
<p>8: Upgrade miRNA and pseudogene representations with two-layer GCN;</p>
</list-item>
<list-item>
<p>9: Update model parameters by minimizing the loss in <xref ref-type="disp-formula" rid="e7">Equation 7</xref>;</p>
</list-item>
<list-item>
<p>10: Training process terminates when the model converges or all epochs are completed;</p>
</list-item>
<list-item>
<p>11: <bold>Return</bold> the train OFGPMA;</p>
</list-item>
</list>
</p>
</statement>
</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<label>3</label>
<title>Results</title>
<sec id="s3-1">
<label>3.1</label>
<title>Evaluation criteria</title>
<p>In OFGPMA. we employ frequently five evaluation metrics to evaluate its performance, including AUC, AUPR, PREC, REC and F1-score. AUC denotes the area under the Receiver Operating Characteristic (ROC) curve, AUPR indicates the area under the Precision-Recall (PR) curve, PREC refers to precision, and REC stands for recall., respectively. For the specific calculation formula, please refer to <xref ref-type="sec" rid="s12">Supplementary Materia Section 5</xref>.</p>
</sec>
<sec id="s3-2">
<label>3.2</label>
<title>Performance of OFGPMA</title>
<p>To assess the performance of OFGPMA, we conducted five-fold cross-validation (5-CV). Specifically, experimentally validated pseudogene-miRNA interactions were used as positive samples. An equal number of negative instances were randomly selected from unconfirmed pseudogene-miRNA pairs. The final dataset for the 5-CV experiments was formed by combining these positive and negative samples.</p>
<p>The 5-CV methodology entailed the random division of the data into five distinct subsets. During each iteration, a single subset was designated as the test set, with the other four subsets combined to form the training set. Importantly, random partitioning ensured that both training and test data within each fold maintained an equal balance of positive and negative samples. To account for variability and minimize bias in the 5-CV findings, performance metrics were averaged over all folds, and their standard deviation was calculated. It should be noted that although AUC summarized overall model efficacy, AUPR furnished a more nuanced perspective (<xref ref-type="bibr" rid="B21">Ling et al., 2025</xref>). Consequently, AUC and AUPR were utilized as the principal performance indicators.</p>
<p>As presented in <xref ref-type="table" rid="T2">Table 2</xref>, OFGPMA achieved an AUC score of 0.8718 and an AUPR score of 0.9105 across the five folds. Performance variations were observed: the third fold yielded a lower AUC value compared to other folds, while the first fold exhibited a higher AUPR value. These fluctuations were attributable to model performance variability induced by different random seeds. Throughout the cross-validation, Precision and Recall metrics demonstrated minor oscillations around their respective means, with an overall limited range of variation. Collectively, these robust results confirmed the potential utility of OFGPMA for predicting potential PMAs.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Performance of OFGPMA.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Fold</th>
<th align="left">AUC</th>
<th align="left">AUPR</th>
<th align="left">Precision</th>
<th align="left">Recall</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">1</td>
<td align="left">0.8711</td>
<td align="left">0.9118</td>
<td align="left">0.9230</td>
<td align="left">0.8993</td>
</tr>
<tr>
<td align="left">2</td>
<td align="left">0.8635</td>
<td align="left">0.8996</td>
<td align="left">0.9193</td>
<td align="left">0.9103</td>
</tr>
<tr>
<td align="left">3</td>
<td align="left">0.8613</td>
<td align="left">0.9110</td>
<td align="left">0.9217</td>
<td align="left">0.9005</td>
</tr>
<tr>
<td align="left">4</td>
<td align="left">0.8767</td>
<td align="left">0.8998</td>
<td align="left">0.9189</td>
<td align="left">0.8996</td>
</tr>
<tr>
<td align="left">5</td>
<td align="left">0.8678</td>
<td align="left">0.9007</td>
<td align="left">0.9218</td>
<td align="left">0.9076</td>
</tr>
<tr>
<td align="left">Mean</td>
<td align="left">0.8718</td>
<td align="left">0.9105</td>
<td align="left">0.9211</td>
<td align="left">0.9015</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-3">
<label>3.3</label>
<title>Comparison experiment</title>
<p>The efficiency of OFGPMA was assessed through two comparative approaches: 1) direct comparison with specialized PMA predictors such as PMAGAE; 2) Secondly, comparison with diverse computational models including random walk, deep learning, and matrix factorization frameworks, alongside models designed for other biomedical entity associations. Each model was evaluated via 5-fold cross-validation using our dataset, with final scores representing the mean values computed over 100 experimental iterations.</p>
<sec id="s3-3-1">
<label>3.3.1</label>
<title>Comparison with PMAGAE</title>
<p>In the first comparison method, we compared OFGPMA with PMAGAE. PMAGAE is the first proposed computational model for predicting the association between pseudogenes and miRNAs. It is based on the similarity network of pseudogenes and miRNAs and is specifically designed for identifying PMAs. PMAGAE leverages the similarities between pseudogenes and miRNAs and calculates the association strength by integrating the similarity features and connections of nodes using GAE (<xref ref-type="bibr" rid="B51">Zhou et al., 2021</xref>). To ensure equitable comparison, we re-implemented PMAGAE under identical random seed conditions. Comparative results (<xref ref-type="fig" rid="F2">Figure 2</xref>) reveal PMAGAE&#x2019;s AUC (0.8623) and AUPR (0.8996), aligning with prior literature yet demonstrating inferior performance relative to OFGPMA.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Model performance of PMAGAE and OFGPMA.</p>
</caption>
<graphic xlink:href="fgene-16-1643921-g002.tif">
<alt-text content-type="machine-generated">Bar graph comparing PMGAE and OFGPMA across AUC and AUPR metrics. Both models show higher scores for AUPR than AUC. PMGAE scores approximately 0.85 for AUC and 0.90 for AUPR, while OFGPMA scores around 0.87 for AUC and 0.95 for AUPR.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3-3-2">
<label>3.3.2</label>
<title>Comparison with other baselines</title>
<p>In our comparative study, we conducted performance evaluations between OFGPMA and nine existing graph neural network approaches that represent current methodological standards. The compared techniques are detailed in the following listing.<list list-type="simple">
<list-item>
<p>&#x2022; Node2Vec (<xref ref-type="bibr" rid="B11">Grover and Leskovec, 2016</xref>): Node2Vec formulates node embedding as an optimization challenge, employing a neighborhood sampling strategy that harmonizes local and global network exploration via a tunable random walk process. The algorithm&#x2019;s flexibility stems from its adjustable bias parameters governing walk behavior.</p>
</list-item>
<list-item>
<p>&#x2022; GCN (<xref ref-type="bibr" rid="B16">Kipf and Welling, 2016</xref>): GCN, as a semi-supervised framework, generates node embeddings through direct processing of graph adjacency matrices. The model operates on the pseudogene-miRNA bipartite network in its raw topological form, deliberately excluding supplementary biological feature to maintain architectural purity.</p>
</list-item>
<list-item>
<p>&#x2022; GAT (<xref ref-type="bibr" rid="B37">Veli&#x10d;kovi&#x107; et al., 2017</xref>): GAT enhances graph processing through attention mechanisms, where node relationships are dynamically weighted. The pseudogene-miRNA bipartite graph serves as direct input to the attention-based predictor for uncovering previously unknown biological relationships.</p>
</list-item>
<list-item>
<p>&#x2022; GIN (<xref ref-type="bibr" rid="B43">Xu et al., 2018</xref>): Renowned for its discriminative power in graph-based prediction, GIN processes the fundamental pseudogene-miRNA network structure to hypothesize new functional associations between these molecular entities. The architecture demonstrates particular efficacy in biological network inference tasks.</p>
</list-item>
<list-item>
<p>&#x2022; NMFMC (<xref ref-type="bibr" rid="B50">Zheng et al., 2022</xref>): NMFMC employs non-negative matrix decomposition to reconstruct incomplete association matrices, enabling the discovery of previously uncharacterized pseudogene-miRNA interactions. The derived predictions serve as valuable comparative data for subsequent validation studies.</p>
</list-item>
<list-item>
<p>&#x2022; ERMDA (<xref ref-type="bibr" rid="B5">Dai et al., 2022</xref>): Through an ensemble learning framework, ERMDA constructs multiple balanced training datasets while learning hierarchical feature representations. Originally designed for miRNA-disease prediction, the algorithm demonstrates transfer learning capability when applied to pseudogene-miRNA network analysis.</p>
</list-item>
<list-item>
<p>&#x2022; NIMGSA (<xref ref-type="bibr" rid="B14">Jin et al., 2022</xref>): Combining graph autoencoder architecture with attention mechanisms, NIMGSA performs neural matrix imputation for biological relationship prediction. The framework demonstrates particular effectiveness when processing sparse pseudogene-miRNA interaction data.</p>
</list-item>
<list-item>
<p>&#x2022; CGHCN (<xref ref-type="bibr" rid="B20">Liang et al., 2024</xref>): CGHCN integrates conventional graph convolution with hypergraph neural operations, capturing both pairwise and higher-order relationships within biological networks. The model excels at identifying complex interaction patterns in omics data.</p>
</list-item>
<list-item>
<p>&#x2022; MSHGANMDA (<xref ref-type="bibr" rid="B40">Wang S. et al., 2023</xref>): Utilizing meta-subgraph representations within an attention-based graph neural framework, MSHGANMDA provides enhanced prediction of molecular interactions. Its architectural flexibility allows direct application to pseudogene-miRNA association mining tasks.</p>
</list-item>
</list>
</p>
<p>Using 5-fold cross-validation and AUC/AUPR scores as primary metrics, we evaluated the proposed OFGPMA model against nine existing approaches. <xref ref-type="fig" rid="F3">Figure 3</xref> illustrates that OFGPMA achieved superior performance in both AUC and AUPR compared to all other models. On the starBase dataset, OFGPMA notably achieved an AUC value of 0.8718. GCN followed as the second-best performer, though a 2.03% performance gap separates it from OFGPMA, confirming our model&#x2019;s significant contribution to improving graph neural network expressiveness. The third-ranked model, NMFMA, while reinforces that local structural information (captured by enclosing subgraphs) is valuable for PMA prediction, OFGPMA&#x2019;s integration of global RWR graph context with local information yields demonstrably stronger results. Since the closed subgraph only captures the local subgraph information of the pseudogene and miRNA bipartite graph, the OFGPMA method, by integrating the global RWR graph information with the local closed subgraph information, can more comprehensively represent the information of the entire graph, which is of great significance in information integration. CHGCN performed the worst among the other nine models, indicating its lower applicability in the PMA prediction task. Collectively, OFGPMA achieves top performance across all evaluated metrics on the starBase dataset, confirming its strong competitive edge. This enhancement is credited to the elaborate Rayleigh pooling, Chebyshev pooling, and global RWR strategy, which can more comprehensively represent the information of the entire graph and capture efficient global topological semantics, respectively.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>AUC and AUPR value of OFGPMA as well as the other nine baseline models.</p>
</caption>
<graphic xlink:href="fgene-16-1643921-g003.tif">
<alt-text content-type="machine-generated">Bar chart comparing AUC and AUPR scores for different methods. Methods listed on the x-axis are GCN, NMFMC, GAT, GIN, ERNDA, NIMGSA, MSHGABMDA, CGHCN, OFGPMA. The y-axis represents scores ranging from 0.70 to 1.00. Blue bars (AUC) and pink bars (AUPR) show varying scores, with AUPR generally higher across methods.</alt-text>
</graphic>
</fig>
</sec>
</sec>
<sec id="s3-4">
<label>3.4</label>
<title>Robustness analysis</title>
<p>An optimal predictive model is expected to exhibit strong robustness and generalization capabilities. To assess the generalization potential of OFGPMA and confirm its broader applicability, this work applied it to several distinct association prediction tasks. Specifically, multiple datasets encompassing miRNA-disease, gene-disease, piRNA-disease, and microbe-disease associations were compiled. The specific data processing procedures are detailed in the <xref ref-type="sec" rid="s12">Supplementary Materia Section 6</xref>. The specific data quantities are shown in <xref ref-type="table" rid="T3">Table 3</xref>.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Datasets on miRNA-disease, gene-disease, piRNA-disease, and microbe-disease associations.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Pair</th>
<th align="left">Type</th>
<th align="left">Number</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="3" align="left">miRNA- disease (<xref ref-type="bibr" rid="B19">Li et al., 2021</xref>)</td>
<td align="left">miRNA</td>
<td align="left">156</td>
</tr>
<tr>
<td align="left">disease</td>
<td align="left">187</td>
</tr>
<tr>
<td align="left">interaction</td>
<td align="left">1,983</td>
</tr>
<tr>
<td rowspan="3" align="left">Gene-disease (<xref ref-type="bibr" rid="B23">Luo et al., 2019</xref>)</td>
<td align="left">gene</td>
<td align="left">2,909</td>
</tr>
<tr>
<td align="left">disease</td>
<td align="left">1,154</td>
</tr>
<tr>
<td align="left">interaction</td>
<td align="left">4,432</td>
</tr>
<tr>
<td rowspan="3" align="left">piRNA-disease (<xref ref-type="bibr" rid="B4">Chen et al., 2024</xref>)</td>
<td align="left">piRNA</td>
<td align="left">4,976</td>
</tr>
<tr>
<td align="left">disease</td>
<td align="left">28</td>
</tr>
<tr>
<td align="left">interaction</td>
<td align="left">7,939</td>
</tr>
<tr>
<td rowspan="3" align="left">Microbe-disease (<xref ref-type="bibr" rid="B39">Wang et al., 2023b</xref>)</td>
<td align="left">microbe</td>
<td align="left">1,177</td>
</tr>
<tr>
<td align="left">disease</td>
<td align="left">134</td>
</tr>
<tr>
<td align="left">interaction</td>
<td align="left">4,499</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Utilizing identical random seeds and evaluation indicator as the primary experiments, the model&#x2019;s generalization performance was systematically evaluated across these datasets (results presented in <xref ref-type="fig" rid="F4">Figure 4</xref>). The obtained AUC values were 0.9307, 0.9136, 0.9489, and 0.9064 for miRNA-disease, gene-disease, piRNA-disease, and microbe-disease predictions, respectively. Corresponding AUPR scores reached 0.9125, 0.9089, 0.9521, and 0.9381. These consistently higher performance metrics across diverse biological association tasks demonstrate OFGPMA&#x2019;s stability and significant generalization capacity. Consequently, these findings provide additional validation for the effectiveness and robustness of the proposed OFGPMA model.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Performance of OFGPMA for predicting different data types.</p>
</caption>
<graphic xlink:href="fgene-16-1643921-g004.tif">
<alt-text content-type="machine-generated">Bar chart comparing AUC and AUPR scores for four categories: miRNA-disease, gene-disease, piRNA-disease, and microbe-disease. AUC bars are pink; AUPR bars are blue. Scores range from 0.80 to 1.00.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3-5">
<label>3.5</label>
<title>The impact of data imbalance on model performance</title>
<p>Previous experiments employed balanced datasets with equal numbers of positive and negative samples for an initial model evaluation. However, model performance could potentially be influenced by variations in the positive-to-negative sample ratio. To more comprehensively assess OFGPMA&#x2019;s robustness under class imbalance, we performed five-fold cross-validation on the starBase dataset, specifically testing performance at positive-negative ratios of 1:1, 1:2, 1:5, and 1:10. A visual representation of the confusion matrix is provided in <xref ref-type="fig" rid="F5">Figure 5</xref>, and detailed performance metrics are tabulated in <xref ref-type="table" rid="T4">Table 4</xref>. Analysis reveals that as the ratio shifts from 1:1 to 1:2, OFGPMA&#x2019;s average AUC exhibits a gradual increase, potentially attributable to the random seed enhancing model performance. By contrast, the AUPR score showed a significant decline, dropping from 0.9105 to 0.8994. The AUPR metric is frequently utilized to assess classifier performance, particularly under imbalanced data conditions. Although AUPR values experience a significant drop, they remain within a practically acceptable range (<xref ref-type="bibr" rid="B21">Ling et al., 2025</xref>; <xref ref-type="bibr" rid="B29">Saito and Rehmsmeier, 2015</xref>). As illustrated in <xref ref-type="fig" rid="F5">Figure 5</xref>, a substantial increase in false negatives coincides with a marginal improvement in accuracy, while both recall and precision exhibit considerable declines. Overall, these results suggest that balanced datasets, featuring an equal ratio of positive to negative samples, yield optimal training outcomes, enabling the model to reach peak predictive accuracy.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>The Confusion matrices calculated under different ratios of positive and negative samples.</p>
</caption>
<graphic xlink:href="fgene-16-1643921-g005.tif">
<alt-text content-type="machine-generated">Four heatmaps compare prediction accuracy for different ratios: 1:1, 1:5, 1:2, and 1:10. Each matrix shows true vs. predicted values with varying color intensities, indicating different probabilities with corresponding numerical values. A color bar to the right indicates the intensity scale from 0 to 1.</alt-text>
</graphic>
</fig>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Performance of OFGPMA under different positive-to-negative ratios on starBase dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left">Evaluation metrics</th>
<th colspan="4" align="center">Positive: Negative sample ratio</th>
</tr>
<tr>
<th align="left">1:1</th>
<th align="left">1:2</th>
<th align="left">1:5</th>
<th align="left">1:10</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">AUC</td>
<td align="left">0.8718</td>
<td align="left">0.8816</td>
<td align="left">0.8753</td>
<td align="left">0.8632</td>
</tr>
<tr>
<td align="left">AUPR</td>
<td align="left">0.9105</td>
<td align="left">0.9087</td>
<td align="left">0.9063</td>
<td align="left">0.8994</td>
</tr>
<tr>
<td align="left">Precision</td>
<td align="left">0.9211</td>
<td align="left">0.9203</td>
<td align="left">0.9184</td>
<td align="left">0.9103</td>
</tr>
<tr>
<td align="left">Recall</td>
<td align="left">0.9015</td>
<td align="left">0.8967</td>
<td align="left">0.8915</td>
<td align="left">0.8834</td>
</tr>
<tr>
<td align="left">F1_score</td>
<td align="left">0.9133</td>
<td align="left">0.9211</td>
<td align="left">0.9033</td>
<td align="left">0.8935</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-6">
<label>3.6</label>
<title>Hyperparameter sensitivity analysis</title>
<p>Hyperparameter sensitivity analyses were performed for OFGPMA under controlled conditions, where non-target parameters remained fixed to isolate performance impacts of critical variables.</p>
<sec id="s3-6-1">
<label>3.6.1</label>
<title>Effect of the learning rate</title>
<p>The learning rate, a critical hyperparameter, governs the magnitude of adjustments applied to model weights during optimization. Its value critically influences both the efficiency of the training process and the ultimate performance of the model. Excessively low learning rates impede gradient updates, extending training duration. Conversely, excessively high learning rates risk inducing gradient explosion, which can prevent model convergence. Consequently, investigating the effect of learning rate variation on the OFGPMA model is highly pertinent. <xref ref-type="fig" rid="F6">Figure 6A</xref> demonstrates a progressive decline in OFGPMA&#x2019;s performance as the learning rate escalates. Experimental findings reveal that a learning rate of 1e-4 yields the optimal model performance, achieving an AUC of 0.8718, AUPR of 0.9105, precision (PREC) of 0.9211, recall (REC) of 0.9015, and F1-score of 0.9133. Therefore, the learning rate for OFGPMA was ultimately fixed at 1e-4.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>OFGPMA performance of AUC, AUPR, PREC and REC under different hyperparameter. <bold>(A)</bold> denotes learning rate, <bold>(B)</bold> indicates hidden size, <bold>(C)</bold> denotes dropout, <bold>(D)</bold> indicates batch size.</p>
</caption>
<graphic xlink:href="fgene-16-1643921-g006.tif">
<alt-text content-type="machine-generated">Four line graphs (A, B, C, D) display performance metrics of a model - AUC, AUPR, PREC, and REC - against various parameters: A) Learning rate, B) Hidden size, C) Dropout, D) Batch size. Each graph plots score on the vertical axis and the parameter on the horizontal axis. Different colored lines represent each metric.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3-6-2">
<label>3.6.2</label>
<title>Effect of the batch size</title>
<p>Batch size represents a crucial hyperparameter in model optimization. While smaller batches can facilitate model convergence, they often constrain training speed and scalability. Conversely, larger batches, despite enabling more efficient utilization of available computational resources and enhancing training throughput, may detrimentally affect model generalization capability [59, 60]. To investigate the influence of batch size on OFGPMA&#x2019;s performance, we evaluated values within the set {32, 64, 96, 128}. Performance metrics, as depicted in <xref ref-type="fig" rid="F6">Figure 6D</xref>, exhibit a declining trend with increasing batch size. A comprehensive analysis of experimental outcomes and model efficacy led to the selection of a batch size of 32 for conducting subsequent experiments on the starBase dataset.</p>
</sec>
<sec id="s3-6-3">
<label>3.6.3</label>
<title>Effect of the hidden size</title>
<p>Furthermore, the dimensionality of latent representations (hidden size) critically influences model behavior. Insufficient hidden dimensions may result in underfitting, whereas excessive dimensions heighten overfitting risks and prolong training duration. To address this, we systematically evaluated OFGPMA&#x2019;s performance across hidden sizes spanning {64, 96, 128, 256}. As evidenced in <xref ref-type="fig" rid="F6">Figure 6B</xref>, the model achieves peak performance on the starBase dataset with a hidden dimension of 64.</p>
</sec>
<sec id="s3-6-4">
<label>3.6.4</label>
<title>Effect of the dropout</title>
<p>As a regularization technique, dropout mitigates overfitting by stochastically deactivating neural units during training. For OFGPMA, dropout rates were evaluated across {0.3, 0.4, 0.5, 0.6, 0.7}, with performance outcomes detailed in <xref ref-type="fig" rid="F6">Figure 6C</xref>. Optimal model efficacy was observed at a dropout probability of 0.3.</p>
</sec>
</sec>
<sec id="s3-7">
<label>3.7</label>
<title>Ablation experiment</title>
<p>The embedding representations for pseudogenes and miRNAs in OFGPMA are learned through two core components: the Optimal Frequency Discovery (OFD) module and the Graph Representation Learning (GRL) module. To assess the contributions of these modules, ablation studies were executed on the starBase dataset. Three model variants are subsequently defined for comparative analysis:<list list-type="simple">
<list-item>
<p>&#x2022; OFGPMA w/o OFD: a variant without the optimal frequency discovery (OFD) module.</p>
</list-item>
<list-item>
<p>&#x2022; OFGPMA-RWR: a variant that incorporating random walk with restart (RWR) for subgraph sampling <italic>in lieu</italic> of the enclosing subgraph extraction strategy.</p>
</list-item>
<list-item>
<p>&#x2022; OFGPMA-ES: a variant that implementing enclosing subgraph extraction as a substitute for random walk with restart (RWR)-based subgraph sampling.</p>
</list-item>
</list>
</p>
<p>As shown in <xref ref-type="fig" rid="F7">Figure 7</xref>, results suggest that the optimal frequency discovery (OFD) module and the graph guidance representation learning (GRL) module are integral components for OFGPMA. Specifically, OFGPMA demonstrates superior performance on every metric. OFGPMA-RWR ranks second overall, while OFGPMA without OFD performs the worst of all models. This might be because the optimal frequency discovery module successfully captured the high-frequency and low-frequency energy information of the graph, thereby significantly enhancing performance and further verifying the effectiveness of OFD. Removing OFD (w/o OFD) led to the largest performance drop, underscoring the importance of frequency analysis. Using only RWR or enclosing subgraphs (OFGPMA-RWR/ES) resulted in intermediate performance, highlighting the value of combining global and local perspectives.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Ablation experiment performance for OFGPMA and three model variants.</p>
</caption>
<graphic xlink:href="fgene-16-1643921-g007.tif">
<alt-text content-type="machine-generated">Bar chart comparing AUC and AUPR scores for four methods: OFGPMA without OFD, OFGPMA-RWR, OFGPMA-ES, and OFGPMA. AUC scores are in cyan and AUPR scores in pink, with values increasing across methods, peaking at OFGPMA.</alt-text>
</graphic>
</fig>
<p>OFGPMA outperforms OFGPMA-RWR and OFGPMA-ES, mainly due to its adoption of a more efficient full-graph information capture strategy, which enriches the structural semantic information. Ablation studies reveal that the newly introduced OFD plays a critical role in OFGPMA&#x2019;s effectiveness. The incorporation of Random Walk with Restart (RWR) and enclosing subgraph extraction also helped boost prediction performance.</p>
</sec>
<sec id="s3-8">
<label>3.8</label>
<title>Case study</title>
<p>To evaluate the performance of the OFGPMA method in predicting pseudogene-miRNA interactions, we randomly selected two widely studied pseudogenes, RPLP0P2 and MTND4P12, from the ground truth of the starBase database. For every pseudogene analyzed, we deliberately masked its known miRNA interactions during testing. The remaining candidate miRNAs were then sorted in descending sequence using OFGPMA&#x2019;s computed prediction scores. Finally, we selected the top-ranked miRNAs and verified their prediction accuracy through the starBase database.</p>
<p>Regarding the pseudogene MTND4P12 (<xref ref-type="table" rid="T5">Table 5</xref>), two prediction errors occurred. This oncogenic pseudogene exhibits dysregulation in cutaneous melanoma, functioning as a competing endogenous RNA (ceRNA) to upregulate the oncogene AURKB [44]. Notably, Hsa-let-7e-5p is a likely regulatory target of MTND4P12, with both entities showing correlated expression patterns in this malignancy.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Evidence identifies the top 10 miRNAs linked to pseudogenes RPLP0P2 and MTND4P12.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Rank</th>
<th colspan="2" align="center">MTND4P12</th>
<th colspan="2" align="center">RPLP0P2</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">1</td>
<td align="left">hsa-let-7e-5p</td>
<td align="left">Confirmed</td>
<td align="left">hsa-miR-34c-5p</td>
<td align="left">Confirmed</td>
</tr>
<tr>
<td align="left">2</td>
<td align="left">hsa-let-7d-5p</td>
<td align="left">Confirmed</td>
<td align="left">hsa-miR-195-5p</td>
<td align="left">Confirmed</td>
</tr>
<tr>
<td align="left">3</td>
<td align="left">hsa-let-7f-5p</td>
<td align="left">Confirmed</td>
<td align="left">hsa-miR-320d</td>
<td align="left">Confirmed</td>
</tr>
<tr>
<td align="left">4</td>
<td align="left">hsa-let-7c-5p</td>
<td align="left">Confirmed</td>
<td align="left">hsa-let-7b-5p</td>
<td align="left">Confirmed</td>
</tr>
<tr>
<td align="left">5</td>
<td align="left">hsa-miR-448</td>
<td align="left">Unconfirmed</td>
<td align="left">hsa-miR-15a-5p</td>
<td align="left">Unconfirmed</td>
</tr>
<tr>
<td align="left">6</td>
<td align="left">hsa-let-7d-5p</td>
<td align="left">Confirmed</td>
<td align="left">hsa-miR-503-5p</td>
<td align="left">Confirmed</td>
</tr>
<tr>
<td align="left">7</td>
<td align="left">hsa-miR-17-5p</td>
<td align="left">Unconfirmed</td>
<td align="left">hsa-miR-3619-5p</td>
<td align="left">Confirmed</td>
</tr>
<tr>
<td align="left">8</td>
<td align="left">hsa-let-7b-5p</td>
<td align="left">Confirmed</td>
<td align="left">hsa-miR-16-5p</td>
<td align="left">Confirmed</td>
</tr>
<tr>
<td align="left">9</td>
<td align="left">hsa-let-7g-5p</td>
<td align="left">Confirmed</td>
<td align="left">hsa-miR-146a-5p</td>
<td align="left">Unconfirmed</td>
</tr>
<tr>
<td align="left">10</td>
<td align="left">hsa-let-7a-5p</td>
<td align="left">Confirmed</td>
<td align="left">hsa-miR-195-5p</td>
<td align="left">Confirmed</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Regarding pseudogene RPLP0P2 (<xref ref-type="table" rid="T5">Table 5</xref>), our model generated three erroneous predictions. This non-coding sequence is implicated in oncogenesis, particularly lung adenocarcinoma and colorectal carcinoma. Prior research indicates that suppressing RPLP0P2 expression reduces malignant cell proliferation and impairs cellular adhesion mechanisms (<xref ref-type="bibr" rid="B3">Chen et al., 2018</xref>; <xref ref-type="bibr" rid="B45">Yuan et al., 2021</xref>).</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s4">
<label>4</label>
<title>Conclusion</title>
<p>This study proposes an Optimal Frequency Graph Representation Learning Approach (OFGPMA) for predicting pseudogenes-miRNAs association. The model consists of two core modules: the optimal frequency discovery module and the graph representation learning module. In the optimal frequency discovery module, the high-frequency and low-frequency energy information of the given pseudogene-miRNA bipartite graph is extracted through Rayleigh quotient pooling and Chebyshev pooling. These high- and low-frequency spectral components are subsequently integrated into a unified graph representation, amplifying the representational capacity of the graph neural network (GNN). Next, in the graph representation learning module, we extract local closed subgraphs for pseudogenes and global random walk restart (RWR) information for miRNAs based on the fused graph. Subsequently, the extracted closed subgraphs and global graphs are input into a two-layer graph convolutional network (GCN) to obtain node representations. Additionally, to align the high-frequency and low-frequency energy information, a loss function between the high-frequency and low-frequency energy information is introduced to meet the requirements of specific biological hypotheses. Pseudogene-miRNA interaction probabilities are derived from the synthesized representations via MLP transformation. Validation on the starBase dataset confirms OFGPMA&#x2019;s significant performance advantage. Furthermore, case investigations reveal OFGPMA&#x2019;s predictive power extends to undocumented pseudogene-miRNA relationships, multiple of which show starBase-documented biological validation.The advantages of OFGPMA are mainly reflected in the following three aspects: First, by learning graph information at different frequencies, it greatly enhances the representation learning ability of GNN; second, by combining local closed subgraphs and global RWR to extract topological structure information of the graph, it requires neither domain expertise nor external datasets, significantly boosting the model&#x2019;s scalability; third, experimental results show that OFGPMA exhibits superior transfer generalization ability in predicting the associations between miRNAs and other biological entities, providing great potential for its application in other related fields. Despite these achievements, there are still some issues that need to be addressed. Existing datasets documenting pseudogene-miRNA interactions remain sparse, constraining model interpretability and predictive performance. Additionally, the current model only considers the structural information in the pseudogene-miRNA network and ignores the roles of other biomolecules closely related to pseudogenes and miRNAs (such as genes and transcription factors). In future work, incorporating these biomarkers could enable development of more comprehensive biological knowledge graphs, capturing deeper semantic relationships to enhance prediction accuracy of pseudogene-miRNA interactions.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s12">Supplementary Material</xref>, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec sec-type="ethics-statement" id="s6">
<title>Ethics statement</title>
<p>Ethical approval was not required for the study involving humans in accordance with the local legislation and institutional requirements. Written informed consent to participate in this study was not required from the participants or the participants&#x2019; legal guardians/next of kin in accordance with the national legislation and the institutional requirements.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>YZ: Data curation, Writing &#x2013; review and editing, Conceptualization, Writing &#x2013; original draft, Formal Analysis. LX: Validation, Supervision, Writing &#x2013; review and editing, Project administration. YL: Project administration, Data curation, Formal Analysis, Resources, Writing &#x2013; review and editing.</p>
</sec>
<ack>
<title>Acknowledgements</title>
<p>The authors also thank to lab members for assistance.</p>
</ack>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The authors declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s12">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2025.1643921/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2025.1643921/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table2.docx" id="SM1" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table1.xlsx" id="SM2" mimetype="application/xlsx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<fn-group>
<fn fn-type="custom" custom-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/778584/overview">Pu-Feng Du</ext-link>, Tianjin University, China</p>
</fn>
<fn fn-type="custom" custom-type="reviewed-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1572964/overview">Massimo La Rosa</ext-link>, National Research Council (CNR), Italy</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1668746/overview">Morteza Kouhsar</ext-link>, University of Exeter, United Kingdom</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bouritsas</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Frasca</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zafeiriou</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bronstein</surname>
<given-names>M. M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Improving graph neural network expressivity <italic>via</italic> subgraph isomorphism counting</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell.</source> <volume>45</volume>, <fpage>657</fpage>&#x2013;<lpage>668</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2022.3154319</pub-id>
<pub-id pub-id-type="pmid">35201983</pub-id>
</mixed-citation>
</ref>
<ref id="B2">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Carninci</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Kasukawa</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Katayama</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gough</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Frith</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Maeda</surname>
<given-names>N.</given-names>
</name>
<etal/>
</person-group> (<year>2005</year>). <article-title>Molecular biology: the transcriptional landscape of the Mammalian genome</article-title>. <source>Science</source> <volume>309</volume>, <fpage>1559</fpage>&#x2013;<lpage>1563</lpage>. <pub-id pub-id-type="doi">10.1126/science.1112014</pub-id>
<pub-id pub-id-type="pmid">16141072</pub-id>
</mixed-citation>
</ref>
<ref id="B3">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Qu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Guan</surname>
<given-names>N. N.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J. Q.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Predicting miRNA-disease association based on inductive matrix completion</article-title>. <source>Bioinformatics</source> <volume>34</volume>, <fpage>4256</fpage>&#x2013;<lpage>4265</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty503</pub-id>
<pub-id pub-id-type="pmid">29939227</pub-id>
</mixed-citation>
</ref>
<ref id="B4">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Qin</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>PUTransGCN: identification of piRNA-disease associations based on attention encoding graph convolutional network and positive unlabelled learning</article-title>. <source>Brief. Bioinform</source> <volume>25</volume>, <fpage>bbae144</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbae144</pub-id>
<pub-id pub-id-type="pmid">38581419</pub-id>
</mixed-citation>
</ref>
<ref id="B5">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dai</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Duan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Predicting miRNA-disease associations using an ensemble learning framework with resampling method</article-title>. <source>Brief. Bioinform</source> <volume>23</volume>, <fpage>bbab543</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbab543</pub-id>
<pub-id pub-id-type="pmid">34929742</pub-id>
</mixed-citation>
</ref>
<ref id="B6">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dong</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Thanou</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Toni</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Bronstein</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Frossard</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Graph signal processing for machine learning: a review and new perspectives</article-title>. <source>IEEE Signal Process Mag.</source> <volume>37</volume>, <fpage>117</fpage>&#x2013;<lpage>127</lpage>. <pub-id pub-id-type="doi">10.1109/MSP.2020.3014591</pub-id>
</mixed-citation>
</ref>
<ref id="B7">
<mixed-citation publication-type="web">
<person-group person-group-type="author">
<name>
<surname>Dong</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Rayleigh quotient graph neural networks for graph-level anomaly detection</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/2310.02861">http://arxiv.org/abs/2310.02861</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B8">
<mixed-citation publication-type="web">
<person-group person-group-type="author">
<name>
<surname>Du</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Moura</surname>
<given-names>J. M. F.</given-names>
</name>
<name>
<surname>Kar</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Topology adaptive graph convolutional networks</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/1710.10370">http://arxiv.org/abs/1710.10370</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B9">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Frasca</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Bevilacqua</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Bronstein</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Maron</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Understanding and extending subgraph GNNs by rethinking their symmetries</article-title>.</mixed-citation>
</ref>
<ref id="B10">
<mixed-citation publication-type="web">
<person-group person-group-type="author">
<name>
<surname>Gavili</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.-P.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>On the shift operator, graph frequency and optimal filtering in graph signal processing</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/1511.03512">http://arxiv.org/abs/1511.03512</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B11">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Grover</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Leskovec</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Node2vec: scalable feature learning for networks</article-title>,&#x201d; in <source>Proceedings of the ACM SIGKDD international conference on knowledge discovery and data mining</source> (<publisher-name>Association for Computing Machinery</publisher-name>), <fpage>855</fpage>&#x2013;<lpage>864</lpage>. <pub-id pub-id-type="doi">10.1145/2939672.2939754</pub-id>
</mixed-citation>
</ref>
<ref id="B12">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Pang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>C. W.</given-names>
</name>
<name>
<surname>Vetro</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Graph signal processing for geometric data and beyond: theory and applications</article-title>. <source>IEEE Trans. Multimed.</source> <volume>24</volume>, <fpage>3961</fpage>&#x2013;<lpage>3977</lpage>. <pub-id pub-id-type="doi">10.1109/TMM.2021.3111440</pub-id>
</mixed-citation>
</ref>
<ref id="B13">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hydbring</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Badalian-Very</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Clinical applications of microRNAs</article-title>. <source>F1000Res</source> <volume>2</volume>, <fpage>136</fpage>. <pub-id pub-id-type="doi">10.12688/f1000research.2-136.v1</pub-id>
<pub-id pub-id-type="pmid">24627783</pub-id>
</mixed-citation>
</ref>
<ref id="B14">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jin</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Predicting mirna-disease association based on neural inductive matrix completion with graph autoencoders and self-attention mechanism</article-title>. <source>Biomolecules</source> <volume>12</volume>, <fpage>64</fpage>. <pub-id pub-id-type="doi">10.3390/biom12010064</pub-id>
<pub-id pub-id-type="pmid">35053212</pub-id>
</mixed-citation>
</ref>
<ref id="B15">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karreth</surname>
<given-names>F. A.</given-names>
</name>
<name>
<surname>Reschke</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ruocco</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ng</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Chapuy</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>L&#xe9;opold</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>The BRAF pseudogene functions as a competitive endogenous RNA and induces lymphoma <italic>in vivo</italic>
</article-title>. <source>Cell</source> <volume>161</volume>, <fpage>319</fpage>&#x2013;<lpage>332</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2015.02.043</pub-id>
<pub-id pub-id-type="pmid">25843629</pub-id>
</mixed-citation>
</ref>
<ref id="B16">
<mixed-citation publication-type="web">
<person-group person-group-type="author">
<name>
<surname>Kipf</surname>
<given-names>T. N.</given-names>
</name>
<name>
<surname>Welling</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Semi-supervised classification with graph convolutional networks</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/1609.02907">http://arxiv.org/abs/1609.02907</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B17">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>R.-C.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Accuracy of computed eigenvectors <italic>via</italic> optimizing a rayleigh quotient</article-title>. <source>BIT Numer. Math.</source> <volume>44</volume>, <fpage>585</fpage>&#x2013;<lpage>593</lpage>. <pub-id pub-id-type="doi">10.1023/b:bitn.0000046798.28622.67</pub-id>
</mixed-citation>
</ref>
<ref id="B18">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>J. H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Qu</surname>
<given-names>L. H.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J. H.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>StarBase v2.0: decoding miRNA-ceRNA, miRNA-ncRNA and protein-RNA interaction networks from large-scale CLIP-seq data</article-title>. <source>Nucleic Acids Res.</source> <volume>42</volume>, <fpage>D92</fpage>&#x2013;<lpage>D97</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkt1248</pub-id>
<pub-id pub-id-type="pmid">24297251</pub-id>
</mixed-citation>
</ref>
<ref id="B19">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Nie</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>You</surname>
<given-names>Z. H.</given-names>
</name>
<name>
<surname>Bao</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A graph auto-encoder model for miRNA-disease associations prediction</article-title>. <source>Brief. Bioinform</source> <volume>22</volume>, <fpage>bbaa240</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbaa240</pub-id>
<pub-id pub-id-type="pmid">34293850</pub-id>
</mixed-citation>
</ref>
<ref id="B20">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Predicting miRNA&#x2013;Disease associations by combining graph and hypergraph convolutional network</article-title>. <source>Interdiscip. Sci.</source> <volume>16</volume>, <fpage>289</fpage>&#x2013;<lpage>303</lpage>. <pub-id pub-id-type="doi">10.1007/s12539-023-00599-3</pub-id>
<pub-id pub-id-type="pmid">38286905</pub-id>
</mixed-citation>
</ref>
<ref id="B21">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ling</surname>
<given-names>C. X.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>AUC: a better measure than accuracy in comparing learning algorithms</article-title>.</mixed-citation>
</ref>
<ref id="B22">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Miao</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X. L.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Neighborhood regularized logistic matrix factorization for drug-target interaction prediction</article-title>. <source>PLoS Comput. Biol.</source> <volume>12</volume>, <fpage>e1004760</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1004760</pub-id>
<pub-id pub-id-type="pmid">26872142</pub-id>
</mixed-citation>
</ref>
<ref id="B23">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Luo</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>L. P.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>F. X.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Enhancing the prediction of disease-gene associations with multimodal deep learning</article-title>. <source>Bioinformatics</source> <volume>35</volume>, <fpage>3735</fpage>&#x2013;<lpage>3742</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz155</pub-id>
<pub-id pub-id-type="pmid">30825303</pub-id>
</mixed-citation>
</ref>
<ref id="B24">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Genome-wide analysis of pseudogenes reveals HBBP1&#x2019;s human-specific essentiality in erythropoiesis and implication in &#x3b2;-thalassemia</article-title>. <source>Dev. Cell</source> <volume>56</volume>, <fpage>478</fpage>&#x2013;<lpage>493.e11</lpage>. <pub-id pub-id-type="doi">10.1016/j.devcel.2020.12.019</pub-id>
<pub-id pub-id-type="pmid">33476555</pub-id>
</mixed-citation>
</ref>
<ref id="B25">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ortega</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Frossard</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Kovacevic</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Moura</surname>
<given-names>J. M. F.</given-names>
</name>
<name>
<surname>Vandergheynst</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Graph signal processing: overview, challenges, and applications</article-title>. <source>Proc. IEEE</source> <volume>106</volume>, <fpage>808</fpage>&#x2013;<lpage>828</lpage>. <pub-id pub-id-type="doi">10.1109/JPROC.2018.2820126</pub-id>
</mixed-citation>
</ref>
<ref id="B26">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peng</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Predicting anti-cancer drug response based on hypergraph representation learning</article-title>. <source>IEEE Trans. Comput. Biol. Bioinforma.</source>, <fpage>1</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1109/TCBBIO.2025.3535887</pub-id>
<pub-id pub-id-type="pmid">40811276</pub-id>
</mixed-citation>
</ref>
<ref id="B27">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ramakrishna</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Wai</surname>
<given-names>H. T.</given-names>
</name>
<name>
<surname>Scaglione</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A user guide to low-pass graph signal processing and its applications: tools and applications</article-title>. <source>IEEE Signal Process Mag.</source> <volume>37</volume>, <fpage>74</fpage>&#x2013;<lpage>85</lpage>. <pub-id pub-id-type="doi">10.1109/MSP.2020.3014590</pub-id>
</mixed-citation>
</ref>
<ref id="B28">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rutnam</surname>
<given-names>Z. J.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>W. W.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>B. B.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>The pseudogene TUSC2P promotes TUSC2 function by binding multiple microRNAs</article-title>. <source>Nat. Commun.</source> <volume>5</volume>, <fpage>2914</fpage>. <pub-id pub-id-type="doi">10.1038/ncomms3914</pub-id>
<pub-id pub-id-type="pmid">24394498</pub-id>
</mixed-citation>
</ref>
<ref id="B29">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Saito</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Rehmsmeier</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>The precision-recall plot is more informative than the ROC plot when evaluating binary classifiers on imbalanced datasets</article-title>. <source>PLoS One</source> <volume>10</volume>, <fpage>e0118432</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0118432</pub-id>
<pub-id pub-id-type="pmid">25738806</pub-id>
</mixed-citation>
</ref>
<ref id="B30">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Salmena</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Poliseno</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Tay</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kats</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Pandolfi</surname>
<given-names>P. P.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>A ceRNA hypothesis: the rosetta stone of a hidden RNA language?</article-title> <source>Cell</source> <volume>146</volume>, <fpage>353</fpage>&#x2013;<lpage>358</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2011.07.014</pub-id>
<pub-id pub-id-type="pmid">21802130</pub-id>
</mixed-citation>
</ref>
<ref id="B31">
<mixed-citation publication-type="web">
<person-group person-group-type="author">
<name>
<surname>Sandryhaila</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Moura</surname>
<given-names>J. M. F.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Discrete signal processing on graphs: frequency analysis</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/1307.0468">http://arxiv.org/abs/1307.0468</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B32">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Setoyama</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Ling</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Natsugoe</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Calin</surname>
<given-names>G. A.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Non-coding RNAs for medical practice in oncology</article-title>. <source>Keio J. Med.</source> <volume>60</volume>, <fpage>106</fpage>&#x2013;<lpage>113</lpage>. <pub-id pub-id-type="doi">10.2302/kjm.60.106</pub-id>
<pub-id pub-id-type="pmid">22200634</pub-id>
</mixed-citation>
</ref>
<ref id="B33">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Nie</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Pseudogene-expressed RNAs: a new frontier in cancers</article-title>. <source>Tumor Biol.</source> <volume>37</volume>, <fpage>1471</fpage>&#x2013;<lpage>1478</lpage>. <pub-id pub-id-type="doi">10.1007/s13277-015-4482-z</pub-id>
<pub-id pub-id-type="pmid">26662308</pub-id>
</mixed-citation>
</ref>
<ref id="B34">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stiegelbauer</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Perakis</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Deutsch</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ling</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Gerger</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Pichler</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>MicroRNAs as novel predictive biomarkers and therapeutic targets in colorectal cancer</article-title>. <source>World J. Gastroenterol.</source> <volume>20</volume>, <fpage>11727</fpage>&#x2013;<lpage>11735</lpage>. <pub-id pub-id-type="doi">10.3748/wjg.v20.i33.11727</pub-id>
<pub-id pub-id-type="pmid">25206276</pub-id>
</mixed-citation>
</ref>
<ref id="B35">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Swarnkar</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Sim&#xf5;es</surname>
<given-names>S. N.</given-names>
</name>
<name>
<surname>Anura</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Brentani</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chatterjee</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hashimoto</surname>
<given-names>R. F.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Identifying dense subgraphs in protein&#x2013;protein interaction network for gene selection from microarray data</article-title>. <source>Netw. Model. Analysis Health Inf. Bioinforma.</source> <volume>4</volume>, <fpage>1</fpage>&#x2013;<lpage>18</lpage>. <pub-id pub-id-type="doi">10.1007/s13721-015-0104-3</pub-id>
</mixed-citation>
</ref>
<ref id="B36">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Teru</surname>
<given-names>K. K.</given-names>
</name>
<name>
<surname>Denis</surname>
<given-names>E. G.</given-names>
</name>
<name>
<surname>Hamilton</surname>
<given-names>W. L.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Inductive relation prediction by subgraph reasoning</article-title>.</mixed-citation>
</ref>
<ref id="B37">
<mixed-citation publication-type="web">
<person-group person-group-type="author">
<name>
<surname>Veli&#x10d;kovi&#x107;</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Cucurull</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Casanova</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Romero</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Li&#xf2;</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Graph attention networks</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/1710.10903">http://arxiv.org/abs/1710.10903</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B38">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023a</year>). <article-title>BERT-PG: a two-branch associative feature gated filtering network for aspect sentiment classification</article-title>. <source>J. Intell. Inf. Syst.</source> <volume>60</volume>, <fpage>709</fpage>&#x2013;<lpage>730</lpage>. <pub-id pub-id-type="doi">10.1007/s10844-023-00785-1</pub-id>
</mixed-citation>
</ref>
<ref id="B39">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xuan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023b</year>). <article-title>Predicting potential microbe&#x2013;disease associations based on multi-source features and deep learning</article-title>. <source>Brief. Bioinform</source> <volume>24</volume>, <fpage>bbad255</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbad255</pub-id>
<pub-id pub-id-type="pmid">37406190</pub-id>
</mixed-citation>
</ref>
<ref id="B40">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Qiao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhuang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Pang</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2023c</year>). <article-title>MSHGANMDA: meta-subgraphs heterogeneous graph attention network for miRNA-Disease association prediction</article-title>. <source>IEEE J. Biomed. Health Inf.</source> <volume>27</volume>, <fpage>4639</fpage>&#x2013;<lpage>4648</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2022.3186534</pub-id>
<pub-id pub-id-type="pmid">35759606</pub-id>
</mixed-citation>
</ref>
<ref id="B41">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>H. B.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2023d</year>). <article-title>ZeroBind: a protein-specific zero-shot predictor with subgraph matching for drug-target interactions</article-title>. <source>Nat. Commun.</source> <volume>14</volume>, <fpage>7861</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-023-43597-1</pub-id>
<pub-id pub-id-type="pmid">38030641</pub-id>
</mixed-citation>
</ref>
<ref id="B42">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Pei</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Graph neural networks</article-title>.</mixed-citation>
</ref>
<ref id="B43">
<mixed-citation publication-type="web">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Leskovec</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jegelka</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>How powerful are graph neural networks?</article-title> <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/1810.00826">http://arxiv.org/abs/1810.00826</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B44">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>MNESEDA: a prior-guided subgraph representation learning framework for predicting disease-related enhancers</article-title>. <source>Knowl. Based Syst.</source> <volume>294</volume>, <fpage>111734</fpage>. <pub-id pub-id-type="doi">10.1016/j.knosys.2024.111734</pub-id>
</mixed-citation>
</ref>
<ref id="B45">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yuan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Tu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Downregulation of lncRNA RPLP0P2 inhibits cell proliferation, invasion and migration, and promotes apoptosis in colorectal cancer</article-title>. <source>Mol. Med. Rep.</source> <volume>23</volume>, <fpage>309</fpage>. <pub-id pub-id-type="doi">10.3892/mmr.2021.11948</pub-id>
<pub-id pub-id-type="pmid">33649783</pub-id>
</mixed-citation>
</ref>
<ref id="B46">
<mixed-citation publication-type="web">
<person-group person-group-type="author">
<name>
<surname>Zeng</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Xia Facebook</surname>
<given-names>Y. A.</given-names>
</name>
<name>
<surname>Srivastava</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Malevich Facebook</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Kannan</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Decoupling the depth and scope of graph neural networks</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://github.com/facebookresearch/shaDow_GNN">https://github.com/facebookresearch/shaDow_GNN</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B47">
<mixed-citation publication-type="web">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Inductive matrix completion based on graph neural networks</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/1904.12058">http://arxiv.org/abs/1904.12058</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B48">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Link prediction based on graph neural networks</article-title>.</mixed-citation>
</ref>
<ref id="B49">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z. B.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>W. M.</given-names>
</name>
<name>
<surname>Ye</surname>
<given-names>X. G.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y. Y.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>The miR-200 family regulates the epithelial-mesenchymal transition induced by EGF/EGFR in anaplastic thyroid cancer cells</article-title>. <source>Int. J. Mol. Med.</source> <volume>30</volume>, <fpage>856</fpage>&#x2013;<lpage>862</lpage>. <pub-id pub-id-type="doi">10.3892/ijmm.2012.1059</pub-id>
<pub-id pub-id-type="pmid">22797360</pub-id>
</mixed-citation>
</ref>
<ref id="B50">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wan</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>MiRNA-Disease association prediction <italic>via</italic> non-negative matrix factorization based matrix completion</article-title>. <source>Signal Process.</source> <volume>190</volume>, <fpage>108312</fpage>. <pub-id pub-id-type="doi">10.1016/j.sigpro.2021.108312</pub-id>
</mixed-citation>
</ref>
<ref id="B51">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Predicting Pseudogene&#x2013;miRNA associations based on feature fusion and graph auto-encoder</article-title>. <source>Front. Genet.</source> <volume>12</volume>, <fpage>781277</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2021.781277</pub-id>
<pub-id pub-id-type="pmid">34966413</pub-id>
</mixed-citation>
</ref>
</ref-list>
</back>
</article>