<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">884028</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2022.884028</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Molecular Subtyping of Cancer Based on Robust Graph Neural Network and Multi-Omics Data Integration</article-title>
<alt-title alt-title-type="left-running-head">Yin et al.</alt-title>
<alt-title alt-title-type="right-running-head">M-GCN Model for Molecular Subtyping</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Yin</surname>
<given-names>Chaoyi</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1695050/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Cao</surname>
<given-names>Yangkun</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1696033/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Sun</surname>
<given-names>Peishuo</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1557819/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Hengyuan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Li</surname>
<given-names>Zhi</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/710105/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xu</surname>
<given-names>Ying</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/762636/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Sun</surname>
<given-names>Huiyan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/866650/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Artificial Intelligence</institution>, <institution>Jilin University</institution>, <addr-line>Changchun</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Medical Oncology</institution>, <institution>the First Hospital of China Medical University</institution>, <addr-line>Shenyang</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Computational Systems Biology Lab</institution>, <institution>Department of Biochemistry and Molecular Biology and Institute of Bioinformatics</institution>, <institution>University of Georgia</institution>, <addr-line>Athens</addr-line>, <addr-line>GA</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1013477/overview">Jianpeng Sheng</ext-link>, Nanyang Technological University, Singapore</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/703922/overview">Jiazhou Chen</ext-link>, South China University of Technology, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1572964/overview">Massimo La Rosa</ext-link>, National Research Council (CNR), Italy</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Zhi Li, <email>zli@cmu.edu.cn</email>; Huiyan Sun, <email>huiyansun@jlu.edu.cn</email>
</corresp>
<fn fn-type="equal" id="fn1">
<label>
<sup>&#x2020;</sup>
</label>
<p>These authors have contributed equally to this work and share first authorship</p>
</fn>
<fn fn-type="other">
<p>This article was submitted to Computational Genomics, a section of the journal Frontiers in Genetics</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>13</day>
<month>05</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>884028</elocation-id>
<history>
<date date-type="received">
<day>25</day>
<month>02</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>31</day>
<month>03</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Yin, Cao, Sun, Zhang, Li, Xu and Sun.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Yin, Cao, Sun, Zhang, Li, Xu and Sun</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Accurate molecular subtypes prediction of cancer patients is significant for personalized cancer diagnosis and treatments. Large amount of multi-omics data and the advancement of data-driven methods are expected to facilitate molecular subtyping of cancer. Most existing machine learning&#x2013;based methods usually classify samples according to single omics data, fail to integrate multi-omics data to learn comprehensive representations of the samples, and ignore that information transfer and aggregation among samples can better represent them and ultimately help in classification. We propose a novel framework named multi-omics graph convolutional network (M-GCN) for molecular subtyping based on robust graph convolutional networks integrating multi-omics data. We first apply the Hilbert&#x2013;Schmidt independence criterion least absolute shrinkage and selection operator (HSIC Lasso) to select the molecular subtype-related transcriptomic features and then construct a sample&#x2013;sample similarity graph with low noise by using these features. Next, we take the selected gene expression, single nucleotide variants (SNV), and copy number variation (CNV) data as input and learn the multi-view representations of samples. On this basis, a robust variant of graph convolutional network (GCN) model is finally developed to obtain samples&#x2019; new representations by aggregating their subgraphs. Experimental results of breast and stomach cancer demonstrate that the classification performance of M-GCN is superior to other existing methods. Moreover, the identified subtype-specific biomarkers are highly consistent with current clinical understanding and promising to assist accurate diagnosis and targeted drug development.</p>
</abstract>
<kwd-group>
<kwd>molecular subtyping of cancer</kwd>
<kwd>multi-omics data</kwd>
<kwd>feature selection</kwd>
<kwd>graph convolutional networks</kwd>
<kwd>subtype-specific biomarkers</kwd>
</kwd-group>
<contract-sponsor id="cn001">National Natural Science Foundation of China<named-content content-type="fundref-id">10.13039/501100001809</named-content>
</contract-sponsor>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>Cancer is a complex and highly individualized disease with diverse subtypes, and molecular heterogeneity exists among different subtypes of the same cancer type (<xref ref-type="bibr" rid="B16">Gonz&#xe1;lez-Garc&#xed;a et al., 2002</xref>; <xref ref-type="bibr" rid="B44">Shipitsin et al., 2007</xref>). As cancer patients of distinct molecular subtypes usually respond differently to same treatment, so accurate subtype classification can not only assist precision diagnosis but also facilitate effective targeted treatment (<xref ref-type="bibr" rid="B52">Toss and Cristofanilli, 2015</xref>; <xref ref-type="bibr" rid="B30">Lee Y.-M. et al., 2020</xref>).</p>
<p>High-throughput sequencing technologies generate a large amount of multi-omics data (<xref ref-type="bibr" rid="B48">Subramanian et al., 2020</xref>), which promotes the proposal of many computational methods to identify the molecular subtypes of cancer. Some methods focus on similarity network fusion (SNF) to cluster cancer subtypes. <xref ref-type="bibr" rid="B56">Wang et al. (2014</xref>) used SNF to combine similarity networks obtained from mRNA expression, DNA methylation, and microRNA expression data into one network. <xref ref-type="bibr" rid="B8">Chen et al. (2021</xref>) proposed a similarity fusion method to fuse the high-order proximity of different omics data and preserve cluster information of multiple graphs. <xref ref-type="bibr" rid="B60">Xu et al. (2019</xref>) developed a method named high-order path elucidated similarity (HOPES), which integrated the similarity of different data by high-order connected paths. These methods apply unsupervised spectral clustering to identify cancer subtypes without using the additional information of sample labels. With the accumulation of labeled data, some supervised machine learning methods are utilized to learn non-linear associations of samples&#x2019; features and subtype labels (<xref ref-type="bibr" rid="B42">Shieh et al., 2004</xref>; <xref ref-type="bibr" rid="B59">Wu et al., 2017</xref>). <xref ref-type="bibr" rid="B19">Guan et al. (2012)</xref> applied splitting random forest to discover a highly predictive gene set for sample classification. <xref ref-type="bibr" rid="B14">Gao et al. (2019)</xref> utilized transcriptomic data and leveraged feedforward neural networks to build molecular subtyping classifiers. <xref ref-type="bibr" rid="B9">Chen et al. (2020)</xref> proposed a DeepType framework, which performed joint supervised classification, unsupervised clustering, and dimensionality reduction to learn cancer-relevant data representation with the cluster structure. These methods treat each sample as an independent individual and do not take full advantage of the similarity and mutual representation ability between samples.</p>
<p>With the strong representation ability of graph-structured data, graph neural networks (GNNs) have achieved great success and are gradually used in a node classification task. It provides one way to obtain new representations of nodes by combining the connectivity and features of its local neighborhood. Recently, some GNN-based methods have been proposed to predict molecular subtyping of cancer. Rhee et al. developed a GCN-based model to explore the gene&#x2013;gene association and information passing for cancer subtyping (<xref ref-type="bibr" rid="B38">Rhee et al., 2017</xref>). Lee et al. developed a GCN model with attention mechanisms to learn pathway-level representations of cancer samples for their subtype classification (<xref ref-type="bibr" rid="B29">Lee et al., 2020a</xref>). Although GNN are powerful, they are reportedly vulnerable when the skeleton of the graph and nodes&#x2019; feature are mixed with noise (<xref ref-type="bibr" rid="B11">Dai et al., 2018</xref>; <xref ref-type="bibr" rid="B24">Jin et al., 2020</xref>; <xref ref-type="bibr" rid="B62">Zhang and Zitnik, 2020</xref>), so a robust GNN model is necessary for accurately and stably predicting cancer subtypes.</p>
<p>It is well known that abnormal behaviors of cancer cells are the result of a series of gene mutations, gene copy number variation, and gene transcription level changes in key regulatory pathways (<xref ref-type="bibr" rid="B17">Greenman et al., 2007</xref>; <xref ref-type="bibr" rid="B7">Bradner et al., 2017</xref>; <xref ref-type="bibr" rid="B28">Kuijjer et al., 2018</xref>; <xref ref-type="bibr" rid="B34">Memon et al., 2021</xref>). A single type of omics data can only capture part of the biological complexity, whereas integrating multiple types of omics data can provide a more holistic view to better understand the interrelationships of the involved biomolecules and their functions and demonstrably improve the prediction accuracy of patients&#x2019; clinical outcome (<xref ref-type="bibr" rid="B22">Huang et al., 2019</xref>; <xref ref-type="bibr" rid="B45">Singh et al., 2019</xref>). To learn integrative representations of different omics data, Li et al. developed a graph autoencoder model by utilizing the prior knowledge graph and integrating mRNA expression and CNV data (<xref ref-type="bibr" rid="B31">Li et al., 2021</xref>). Lin et al. used multi-omics data and applied deep neural networks to improve the classification accuracy of breast cancer sample (<xref ref-type="bibr" rid="B32">Lin et al., 2020</xref>).</p>
<p>In this study, we propose a novel and general framework M-GCN (<xref ref-type="fig" rid="F1">Figure 1</xref>) for molecular subtyping of cancer. It integrates RNA-seq, SNV, and CNV data and learns the node representation based on a robust GCN model. In order to reduce dimension and eliminate noise of transcriptomic data, we first apply HSIC Lasso to select the molecular subtype-related transcriptomic features, which are further used for constructing sample&#x2013;sample similarity graph, and utilize statistics analysis to find genes with high mutational rates and significant copy number changes. The clean data and purified graph structures are prerequisite for building a robust GNN model. Then we use different non-linear transformations to learn multi-view representations of these three types of data. Furthermore, M-GCN strengthens connections between the new generated features and the graph and assigns weight to edges by layer-wise graph memory based on GNNGUARD (<xref ref-type="bibr" rid="B62">Zhang and Zitnik, 2020</xref>). GNNGUARD is originally developed to purify the graph structure and nodes&#x2019; features to eliminate the effect of possible noise edge message passing of GCN. Next, a robust GCN model is developed to get samples&#x2019; new representations by aggregating their subgraphs for predicting their subtype category. When applying M-GCN to study molecular subtyping of breast cancer and stomach cancer, the experiment results show that the subtype classification performance of M-GCN outperforms other state-of-the-art methods. In addition, we further identify a few specific biomarkers for each molecular subtype, which can potentially contribute to disease diagnosis.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Flowchart of M-GCN. <bold>(A)</bold> Filtered SNV and CNV features and molecular subtype-related transcriptomic features selected by HSIC Lasso are used as the input. Three type-specific non-linear transformation layers are used. The sample&#x2013;sample similarity graph is constructed by molecular subtype-related transcriptomic features. <bold>(B)</bold> Output of non-linear transformations and sample&#x2013;sample similarity graph are used as the input of GCN; convolution process is used for message passing and aggregation among samples; output of the final GCN layer is the prediction of samples&#x2019; subtype category.</p>
</caption>
<graphic xlink:href="fgene-13-884028-g001.tif"/>
</fig>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>Materials and Methods</title>
<sec id="s2-1">
<title>Data Collection and Preprocessing</title>
<p>We collect gene expression, SNV and CNV data, and clinical information of breast cancer (BRCA) and stomach adenocarcinoma (STAD) patients from The Cancer Genome Atlas (TCGA) database (<xref ref-type="bibr" rid="B58">Weinstein et al., 2013</xref>). As shown in <xref ref-type="table" rid="T1">Table 1</xref>, there are <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mn>518</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mn>221</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> samples of BRCA and STAD having all three types of omics data. Specifically, BRCA includes molecular subtypes of estrogen receptor positive (ER&#x2b;), human epidermal growth factor receptor 2 positive (HER2&#x2b;), and triple-negative breast cancer (TNBC) (<xref ref-type="bibr" rid="B55">Vuong et al., 2014</xref>), and STAD includes molecular subtypes of chromosomal instability (CIN), Epstein&#x2013;Barr virus (EBV), microsatellite instability (MSI), and genomically stable (GS) (<xref ref-type="bibr" rid="B5">Bass et al., 2014</xref>), respectively.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Dataset attributes.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Cancer</th>
<th align="center">&#x23;Subtype</th>
<th align="center">&#x23;Samples of each subtype</th>
<th align="center">&#x23;CNV features</th>
<th align="center">&#x23;SNV features</th>
<th align="center">&#x23; Gene expression features</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="3" align="left">BRCA</td>
<td align="left">ER&#x2b;</td>
<td align="char" char=".">386</td>
<td rowspan="3" align="char" char=".">74</td>
<td rowspan="3" align="char" char=".">62</td>
<td rowspan="3" align="char" char=".">124</td>
</tr>
<tr>
<td align="left">HER2&#x2b;</td>
<td align="char" char=".">35</td>
</tr>
<tr>
<td align="left">TNBC</td>
<td align="char" char=".">97</td>
</tr>
<tr>
<td rowspan="4" align="left">STAD</td>
<td align="left">CIN</td>
<td align="char" char=".">107</td>
<td rowspan="4" align="char" char=".">169</td>
<td rowspan="4" align="char" char=".">166</td>
<td rowspan="4" align="char" char=".">128</td>
</tr>
<tr>
<td align="left">EBV</td>
<td align="char" char=".">23</td>
</tr>
<tr>
<td align="left">MSI</td>
<td align="char" char=".">46</td>
</tr>
<tr>
<td align="left">GS</td>
<td align="char" char=".">45</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Genes whose expression values are lower than 10 and 3 are considered as not expressed in BRCA and STAD and then deleted, respectively. As RNA-seq data of these two cancer types are obtained from different platforms in TCGA, we set different cutoffs according to their data distributions. Then, fragments per kilobase of exon per million fragments mapped (FPKM) values of gene expression are normalized with log2-transformation. In each cancer type, a gene&#x2019;s mutation frequency is defined as the number of samples with this mutation divided by the total number of samples. Genes with mutation frequency greater than <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mn>0.03</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> are selected and their SNV data are used as SNV features in BRCA. For STAD, mutation frequency threshold is set as <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mn>0.1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. Similarly, the genes having significant amplifications or deletions rates across cancer samples are selected and their CNV data are used as CNV features. Finally, the number of SNV features and CNV features are <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:mn>62</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:mn>74</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> in BRCA and <inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:mn>166</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:mn>169</mml:mn>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> in STAD, respectively. The details of the datasets are listed in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
</sec>
<sec id="s2-2">
<title>Feature Selection</title>
<p>To obtain molecular subtype-related transcriptome features with low noise for constructing a purified sample&#x2013;sample similarity graph and effective message passing, we apply a supervised non-linear feature selection method HSIC Lasso (<xref ref-type="bibr" rid="B61">Yamada et al., 2014</xref>), which captures non-linear dependency of molecular subtyping labels and genes&#x2019; expression level.</p>
<p>Let <inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi>t</mml:mi>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> denote the supervised data with the <inline-formula id="inf10">
<mml:math id="m10">
<mml:mi>n</mml:mi>
</mml:math>
</inline-formula> samples. <inline-formula id="inf11">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf12">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the gene expression vector and label of <inline-formula id="inf13">
<mml:math id="m13">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula>-<inline-formula id="inf14">
<mml:math id="m14">
<mml:mrow>
<mml:mtext>th</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> sample, respectively. Its optimization goal is as follows:<disp-formula id="e1">
<mml:math id="m15">
<mml:mrow>
<mml:mmultiscripts>
<mml:mo>&#xa0;</mml:mo>
<mml:mprescripts/>
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>&#x3b5;</mml:mi>
<mml:msup>
<mml:mi mathvariant="bold-italic">&#x211d;</mml:mi>
<mml:mi mathvariant="bold-italic">d</mml:mi>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mi mathvariant="bold-italic">n</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mn>2</mml:mn>
</mml:mfrac>
<mml:mo>&#x2225;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="bold-italic">L</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>d</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi mathvariant="bold-italic">l</mml:mi>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi mathvariant="bold-italic">l</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
<mml:msubsup>
<mml:mo>&#x2225;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mi mathvariant="bold-italic">r</mml:mi>
<mml:mi mathvariant="bold-italic">o</mml:mi>
<mml:mi mathvariant="bold-italic">b</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:mstyle>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>&#x2225;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:msub>
<mml:mo>&#x2225;</mml:mo>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mtext>s</mml:mtext>
<mml:mo>.</mml:mo>
<mml:mtext>t</mml:mtext>
<mml:mo>.</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi>&#x3b3;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x3b3;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <inline-formula id="inf15">
<mml:math id="m16">
<mml:mrow>
<mml:mo>&#x2225;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x2225;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is the Frobenius norm, <inline-formula id="inf16">
<mml:math id="m17">
<mml:mrow>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="bold-italic">L</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi mathvariant="bold-italic">&#x393;</mml:mi>
<mml:mi mathvariant="bold-italic">L</mml:mi>
<mml:mi mathvariant="bold-italic">&#x393;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is centered Gram matrices, <inline-formula id="inf17">
<mml:math id="m18">
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x393;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">I</mml:mi>
<mml:mi mathvariant="bold-italic">n</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi mathvariant="bold-italic">n</mml:mi>
</mml:mfrac>
<mml:msub>
<mml:mn>1</mml:mn>
<mml:mi mathvariant="bold-italic">n</mml:mi>
</mml:msub>
<mml:msubsup>
<mml:mn>1</mml:mn>
<mml:mi mathvariant="bold-italic">n</mml:mi>
<mml:mi mathvariant="bold-italic">T</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the centering matrix, <inline-formula id="inf18">
<mml:math id="m19">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">I</mml:mi>
<mml:mi mathvariant="bold-italic">n</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the <inline-formula id="inf19">
<mml:math id="m20">
<mml:mi>n</mml:mi>
</mml:math>
</inline-formula>-dimensional identity matrix, <bold>1</bold>
<sub>
<italic>n</italic>
</sub> is the <inline-formula id="inf21">
<mml:math id="m22">
<mml:mi>n</mml:mi>
</mml:math>
</inline-formula>-dimensional vector that all the elements are ones, <inline-formula id="inf22">
<mml:math id="m23">
<mml:mi>d</mml:mi>
</mml:math>
</inline-formula> is the number of features, <inline-formula id="inf23">
<mml:math id="m24">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the regression coefficient of the <inline-formula id="inf24">
<mml:math id="m25">
<mml:mi>l</mml:mi>
</mml:math>
</inline-formula>-th feature, <inline-formula id="inf25">
<mml:math id="m26">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>l</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi mathvariant="bold-italic">&#x393;</mml:mi>
<mml:msup>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>l</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
<mml:mi mathvariant="bold-italic">&#x393;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the centered Gram matrix, <inline-formula id="inf26">
<mml:math id="m27">
<mml:mrow>
<mml:msubsup>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>l</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>l</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>j</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>l</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf27">
<mml:math id="m28">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> are calculated by kernel functions <inline-formula id="inf28">
<mml:math id="m29">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x27;</mml:mo>
</mml:msup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf29">
<mml:math id="m30">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>y</mml:mi>
<mml:mo>&#x27;</mml:mo>
</mml:msup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf30">
<mml:math id="m31">
<mml:mtext>&#x3bb;</mml:mtext>
</mml:math>
</inline-formula> is the regularization parameter, and <inline-formula id="inf31">
<mml:math id="m32">
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b3;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b3;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is a regression coefficient vector.</p>
</sec>
<sec id="s2-3">
<title>Sample&#x2014;Sample Similarity Graph Construction</title>
<p>Since samples with similar features are more likely to fall into the same category, we first construct a sample&#x2013;sample graph based on similarity of each samples pair. From the perspective of the whole biological system, compared with SNV and CNV, gene expression is the most fundamental level at which the genotype gives rise to the phenotype. So we only use transcriptomic data and apply spearman&#x2019;s correlation to calculate the similarity of each two samples. We select sample pairs whose correlation coefficient (<inline-formula id="inf32">
<mml:math id="m33">
<mml:mi>&#x3c1;</mml:mi>
</mml:math>
</inline-formula>) are greater than the threshold <inline-formula id="inf33">
<mml:math id="m34">
<mml:mi>r</mml:mi>
</mml:math>
</inline-formula> with <inline-formula id="inf34">
<mml:math id="m35">
<mml:mi>p</mml:mi>
</mml:math>
</inline-formula> value less than <inline-formula id="inf35">
<mml:math id="m36">
<mml:mrow>
<mml:mn>0.05</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> and then generate adjacency matrix <inline-formula id="inf36">
<mml:math id="m37">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
<disp-formula id="e2">
<mml:math id="m38">
<mml:mrow>
<mml:msub>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi>&#x3c1;</mml:mi>
<mml:mrow>
<mml:mi>ij</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2265;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>0.05</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>0</mml:mi>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <inline-formula id="inf37">
<mml:math id="m39">
<mml:mn>1</mml:mn>
</mml:math>
</inline-formula> and 0 (<xref ref-type="disp-formula" rid="e2">Eq. 2</xref>) represent that there is and there is not an edge between sample <inline-formula id="inf38">
<mml:math id="m40">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula> and sample <inline-formula id="inf39">
<mml:math id="m41">
<mml:mi>j</mml:mi>
</mml:math>
</inline-formula>, respectively.</p>
<p>In the generated undirected graph <inline-formula id="inf40">
<mml:math id="m42">
<mml:mrow>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi mathvariant="bold-italic">E</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf41">
<mml:math id="m43">
<mml:mi mathvariant="bold-italic">V</mml:mi>
</mml:math>
</inline-formula> and <inline-formula id="inf42">
<mml:math id="m44">
<mml:mi mathvariant="bold-italic">E</mml:mi>
</mml:math>
</inline-formula> denote the sample nodes and edges, respectively. There are <inline-formula id="inf43">
<mml:math id="m45">
<mml:mi>n</mml:mi>
</mml:math>
</inline-formula> samples in the graph. <inline-formula id="inf44">
<mml:math id="m46">
<mml:mi mathvariant="bold-italic">X</mml:mi>
</mml:math>
</inline-formula> &#x3d; [<inline-formula id="inf45">
<mml:math id="m47">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi>e</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>] are the nodes&#x2019; feature matrix, where [] is the concat operation. <inline-formula id="inf46">
<mml:math id="m48">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf47">
<mml:math id="m49">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf48">
<mml:math id="m50">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi>e</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> represent SNV feature matrix, CNV feature matrix, and gene expression feature matrix, respectively. The number of features in each data type is <inline-formula id="inf49">
<mml:math id="m51">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
<inline-formula id="inf50">
<mml:math id="m52">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf51">
<mml:math id="m53">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The features of SNV and CNV are selected by data preprocessing. Gene expression features are obtained by HSIC Lasso.</p>
</sec>
<sec id="s2-4">
<title>GCN Model Integrating Multi-Omics Data for Sample Classification</title>
<p>Some noises could be introduced as there could be some mismatch between the graph and the new concatenate features. According to similarity between the new features, we introduce the idea of a robust variant into the GCN model to mitigate the impact of noises. Specifically, we apply GNNGUARD, which originally is a defense method against adversarial attacks. It improves robustness of GCN models by detecting fake edges of graph structure and removes or reduces their weights in message passing of GCN. GNNGUARD is implemented by neighbor importance estimation and layer-wise graph memory. We use GNNGUARD here to strengthen connections between the new features and the graph by assigning weight to edges. In addition, our framework can be more robust using dirty data with noises.</p>
<sec>
<title>Multi-Omics Data Features Transformation</title>
<p>To improve samples&#x2019; feature representations, we adopt different non-linear transformations to separately project gene expression features, SNV features, and CNV features into their feature space, and then concatenate them together. The projected latent feature matrix <inline-formula id="inf52">
<mml:math id="m54">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is as follows:<disp-formula id="e3">
<mml:math id="m55">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mi mathvariant="normal">0</mml:mi>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mtext>&#x3c3;</mml:mtext>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi mathvariant="italic">m</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi mathvariant="italic">m</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mtext>&#x3c3;</mml:mtext>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mtext>&#x3c3;</mml:mtext>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi mathvariant="italic">e</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi mathvariant="italic">e</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>where <inline-formula id="inf53">
<mml:math id="m56">
<mml:mi>&#x3c3;</mml:mi>
</mml:math>
</inline-formula> is the ReLU activation function, <inline-formula id="inf54">
<mml:math id="m57">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x27;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf55">
<mml:math id="m58">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">&#x211d;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#xd7;</mml:mo>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mn>2</mml:mn>
<mml:mo>&#x27;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf56">
<mml:math id="m59">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi>e</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">&#x211d;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mn>3</mml:mn>
<mml:mo>&#x27;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> represent the learnable non-linear transformation matrix of SNV, CNV, and gene expression data, respectively. <inline-formula id="inf57">
<mml:math id="m60">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mn>0</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf58">
<mml:math id="m61">
<mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x27;</mml:mo>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mn>2</mml:mn>
<mml:mo>&#x27;</mml:mo>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mn>3</mml:mn>
<mml:mo>&#x27;</mml:mo>
</mml:msubsup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the final output multi-view representations of samples.</p>
</sec>
<sec>
<title>Neighbor Importance Estimation</title>
<p>To quantify the relevance between node <inline-formula id="inf59">
<mml:math id="m62">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula> and node <inline-formula id="inf60">
<mml:math id="m63">
<mml:mi>j</mml:mi>
</mml:math>
</inline-formula> for successful message passing of GCN, GNNGUARD evaluates the importance weight of each edge <inline-formula id="inf61">
<mml:math id="m64">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> in each layer based on similarity measure between nodes&#x2019; representations. The similarity <inline-formula id="inf62">
<mml:math id="m65">
<mml:mrow>
<mml:msubsup>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is defined as follows based on the hypothesis that similar nodes are more likely to interact with each other:<disp-formula id="e4">
<mml:math id="m66">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">s</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mi mathvariant="bold-italic">j</mml:mi>
</mml:mrow>
<mml:mi mathvariant="bold-italic">k</mml:mi>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">h</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mi mathvariant="bold-italic">k</mml:mi>
</mml:msubsup>
<mml:mo>&#x2299;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">h</mml:mi>
<mml:mi mathvariant="bold-italic">j</mml:mi>
<mml:mi mathvariant="bold-italic">k</mml:mi>
</mml:msubsup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mo>&#x2225;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">h</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mi mathvariant="bold-italic">k</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mo>&#x2225;</mml:mo>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2225;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">h</mml:mi>
<mml:mi mathvariant="bold-italic">j</mml:mi>
<mml:mi mathvariant="bold-italic">k</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mo>&#x2225;</mml:mo>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>where <inline-formula id="inf63">
<mml:math id="m67">
<mml:mrow>
<mml:msubsup>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the cosine similarity between <inline-formula id="inf64">
<mml:math id="m68">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula> and its neighbor <inline-formula id="inf65">
<mml:math id="m69">
<mml:mrow>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> in the <inline-formula id="inf66">
<mml:math id="m70">
<mml:mi>k</mml:mi>
</mml:math>
</inline-formula>-th layer of GCN, <inline-formula id="inf67">
<mml:math id="m71">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">h</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">&#x211d;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf68">
<mml:math id="m72">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">h</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">&#x211d;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> denote the representations of node <inline-formula id="inf69">
<mml:math id="m73">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula> and node <inline-formula id="inf70">
<mml:math id="m74">
<mml:mi>j</mml:mi>
</mml:math>
</inline-formula> in the <inline-formula id="inf71">
<mml:math id="m75">
<mml:mi>k</mml:mi>
</mml:math>
</inline-formula>-th layer of GCN, <inline-formula id="inf72">
<mml:math id="m76">
<mml:mo>&#x2299;</mml:mo>
</mml:math>
</inline-formula> is dot product, <inline-formula id="inf73">
<mml:math id="m77">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the dimension of <inline-formula id="inf74">
<mml:math id="m78">
<mml:mrow>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">h</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> (or <inline-formula id="inf75">
<mml:math id="m79">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">h</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>), and <inline-formula id="inf76">
<mml:math id="m80">
<mml:mrow>
<mml:mo>&#x2225;</mml:mo>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mo>&#x2225;</mml:mo>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is the L-2 norm. Node similarity <inline-formula id="inf77">
<mml:math id="m81">
<mml:mrow>
<mml:msubsup>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is normalized at the node-level within <inline-formula id="inf78">
<mml:math id="m82">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula>&#x2019;s neighborhood as follows:<disp-formula id="e5">
<mml:math id="m83">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mtext>&#x3a3;</mml:mtext>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msubsup>
<mml:mi>N</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2217;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>&#xd7;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>N</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>N</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x2260;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>N</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>where <inline-formula id="inf79">
<mml:math id="m84">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is an importance weight between node <inline-formula id="inf80">
<mml:math id="m85">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula> and node <inline-formula id="inf81">
<mml:math id="m86">
<mml:mi>j</mml:mi>
</mml:math>
</inline-formula> in the <inline-formula id="inf82">
<mml:math id="m87">
<mml:mi>k</mml:mi>
</mml:math>
</inline-formula>-th layer, <inline-formula id="inf83">
<mml:math id="m88">
<mml:mrow>
<mml:msubsup>
<mml:mi>N</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2217;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> represents <inline-formula id="inf84">
<mml:math id="m89">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula>&#x2019;s neighborhood (excluding node <inline-formula id="inf85">
<mml:math id="m90">
<mml:mrow>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>), and <inline-formula id="inf86">
<mml:math id="m91">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>N</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>&#x3a3;</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msubsup>
<mml:mi>N</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2217;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2225;</mml:mo>
<mml:msubsup>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mo>&#x2225;</mml:mo>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The noises can be defensed by using important weights on the basis of reducing the weight of dissimilar nodes. Edge pruning probability for edge <inline-formula id="inf87">
<mml:math id="m92">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is calculated by a binary indicator <inline-formula id="inf88">
<mml:math id="m93">
<mml:mrow>
<mml:msub>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>: <inline-formula id="inf89">
<mml:math id="m94">
<mml:mrow>
<mml:mtext>&#x3c3;</mml:mtext>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">c</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> as follows:<disp-formula id="e6">
<mml:math id="m95">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">&#x3c3;</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">c</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mi mathvariant="bold-italic">W</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">c</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3c;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>where <inline-formula id="inf90">
<mml:math id="m96">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">c</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is a characteristic vector in the <inline-formula id="inf91">
<mml:math id="m97">
<mml:mi>k</mml:mi>
</mml:math>
</inline-formula>-th layer of GCN which describes edge <inline-formula id="inf92">
<mml:math id="m98">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf93">
<mml:math id="m99">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the learnable parameter, <inline-formula id="inf94">
<mml:math id="m100">
<mml:mtext>&#x3c3;</mml:mtext>
</mml:math>
</inline-formula> is a non-linear transformation, and <inline-formula id="inf95">
<mml:math id="m101">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is a pre-defined threshold. We update importance weight <inline-formula id="inf96">
<mml:math id="m102">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula id="inf97">
<mml:math id="m103">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and prune edges with <xref ref-type="disp-formula" rid="e7">Eq. 7</xref> in order to ignore perturbed edge.<disp-formula id="e7">
<mml:math id="m104">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>&#x3c3;</mml:mtext>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">c</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
</p>
</sec>
<sec>
<title>Layer-Wise Graph Memory</title>
<p>Neighbor importance estimation and edge pruning change the structure of graph. Because the weighted graph changes in each layer, for a stable training to keep partial memory of the weighted graph structure from the <inline-formula id="inf98">
<mml:math id="m105">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>-th layer for the <inline-formula id="inf99">
<mml:math id="m106">
<mml:mi>k</mml:mi>
</mml:math>
</inline-formula>-th layer, GNNGUARD introduces a trick called layer-wise graph memory (<xref ref-type="fig" rid="F2">Figure 2</xref>). The layer-wise graph memory is defined as follows:<disp-formula id="e8">
<mml:math id="m107">
<mml:mrow>
<mml:msubsup>
<mml:mtext>&#x3c6;</mml:mtext>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:msubsup>
<mml:mtext>&#x3c6;</mml:mtext>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>where <inline-formula id="inf100">
<mml:math id="m108">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is a learnable parameter and <inline-formula id="inf101">
<mml:math id="m109">
<mml:mrow>
<mml:msubsup>
<mml:mtext>&#x3c6;</mml:mtext>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> denotes weight for edge <inline-formula id="inf102">
<mml:math id="m110">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> in the <inline-formula id="inf103">
<mml:math id="m111">
<mml:mi>k</mml:mi>
</mml:math>
</inline-formula>-th layer of GCN.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Illustration of GNNGUARD. An example of one node (orange node) is chosen to demonstrate the process of layer-wise graph memory. <bold>(A)</bold> Message passing in <italic>i</italic>&#x2019;s local neighborhood in the <italic>k</italic>-th layer of GCN. <bold>(B)</bold> Thickness of the gray arrow represents the weight in the message passing of GCN. The weight between node <inline-formula id="inf104">
<mml:math id="m112">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and node <inline-formula id="inf105">
<mml:math id="m113">
<mml:mi>j</mml:mi>
</mml:math>
</inline-formula> is utilized to determine message passing between nodes, such as strengthening message or blocking message. To stably train model, the <italic>k</italic>-th layer weight coefficient keeps a partial memory of the <inline-formula id="inf106">
<mml:math id="m114">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>-th layer.</p>
</caption>
<graphic xlink:href="fgene-13-884028-g002.tif"/>
</fig>
</sec>
<sec>
<title>Node Aggregation with Multi-View Representations Based on GCN</title>
<p>To learn comprehensive representations of sample nodes and multi-omics data, a multilayered graph convolutional network (<xref ref-type="bibr" rid="B26">Kipf and Welling, 2016</xref>) based on the message passing is defined as follows:<disp-formula id="e9">
<mml:math id="m115">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold-italic">k</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mtext>&#x3c3;</mml:mtext>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mi mathvariant="bold-italic">k</mml:mi>
</mml:msup>
<mml:msup>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mi mathvariant="bold-italic">k</mml:mi>
</mml:msup>
<mml:msup>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi mathvariant="bold-italic">k</mml:mi>
</mml:msup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>where <inline-formula id="inf107">
<mml:math id="m116">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="bold-italic">D</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mn>2</mml:mn>
</mml:mfrac>
</mml:mrow>
</mml:msup>
<mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msup>
<mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="bold-italic">D</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mn>2</mml:mn>
</mml:mfrac>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> represents the normalized Laplacian of the weighted graph in <italic>k</italic>-th layer, <inline-formula id="inf108">
<mml:math id="m117">
<mml:mrow>
<mml:msubsup>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msubsup>
<mml:mtext>A</mml:mtext>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:msubsup>
<mml:mrow>
<mml:mtext>&#xa0;&#x3c6;</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is recalculated at each layer to update the adjacency matrix, <inline-formula id="inf109">
<mml:math id="m118">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="bold-italic">I</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the adjacency matrix with added self-connections, <inline-formula id="inf110">
<mml:math id="m119">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>D</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</inline-formula> is the degree matrix, <inline-formula id="inf111">
<mml:math id="m120">
<mml:mi mathvariant="bold-italic">W</mml:mi>
</mml:math>
</inline-formula> is a layer-specific learnable weight matrix from training, <inline-formula id="inf112">
<mml:math id="m121">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mn>0</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> represents the input of the first GCN layer, <inline-formula id="inf113">
<mml:math id="m122">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the input of the <inline-formula id="inf114">
<mml:math id="m123">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>-th layer, and <inline-formula id="inf115">
<mml:math id="m124">
<mml:mrow>
<mml:msup>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the comprehensive representation by aggregating neighbor features <inline-formula id="inf116">
<mml:math id="m125">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. The activation function softmax is used in the last graph convolutional layer to calculate the probability <inline-formula id="inf117">
<mml:math id="m126">
<mml:mrow>
<mml:mtext mathvariant="bold-italic">P</mml:mtext>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> of which molecular subtyping of each sample belongs to <xref ref-type="disp-formula" rid="e10">Eq. 10</xref>.<disp-formula id="e10">
<mml:math id="m127">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mi mathvariant="bold-italic">P</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>softmax</mml:mtext>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mi mathvariant="bold-italic">k</mml:mi>
</mml:msup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>where <inline-formula id="inf118">
<mml:math id="m128">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the output of final graph convolutional layer <inline-formula id="inf119">
<mml:math id="m129">
<mml:mi>k</mml:mi>
</mml:math>
</inline-formula>, and <inline-formula id="inf120">
<mml:math id="m130">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the prediction probability vector of the sample node <inline-formula id="inf121">
<mml:math id="m131">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula>.</p>
</sec>
<sec>
<title>Loss and Optimization</title>
<p>Cross-entropy is used as the loss function of our model:<disp-formula id="e11">
<mml:math id="m132">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>&#x2112;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mi>1</mml:mi>
<mml:mi>n</mml:mi>
</mml:mfrac>
<mml:msub>
<mml:mtext>&#x3a3;</mml:mtext>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msubsup>
<mml:mtext>&#x3a3;</mml:mtext>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>1</mml:mi>
</mml:mrow>
<mml:mi>V</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">P</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>where <inline-formula id="inf122">
<mml:math id="m133">
<mml:mi>V</mml:mi>
</mml:math>
</inline-formula> is the number of molecular subtypes, <inline-formula id="inf123">
<mml:math id="m134">
<mml:mi>n</mml:mi>
</mml:math>
</inline-formula> is the number of total samples, <inline-formula id="inf124">
<mml:math id="m135">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the ground truth label of <inline-formula id="inf125">
<mml:math id="m136">
<mml:mrow>
<mml:msup>
<mml:mi>i</mml:mi>
<mml:mo>&#xa0;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>-th sample, and <inline-formula id="inf126">
<mml:math id="m137">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the probability score that sample <inline-formula id="inf127">
<mml:math id="m138">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula> in molecular subtype <inline-formula id="inf128">
<mml:math id="m139">
<mml:mi>v</mml:mi>
</mml:math>
</inline-formula>. Adam is used to minimize the loss function (<xref ref-type="bibr" rid="B25">Kingma and Ba, 2014</xref>).</p>
</sec>
<sec>
<title>New Sample Prediction</title>
<p>When predicting which molecular subtype a new sample is, we first add it into dataset and the sample&#x2013;sample similarity graph according to <xref ref-type="disp-formula" rid="e2">Eq. 2</xref>. The new data is <inline-formula id="inf129">
<mml:math id="m140">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#x2208;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf130">
<mml:math id="m141">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf131">
<mml:math id="m142">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf132">
<mml:math id="m143">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the number of selected features of SNV data, CNV data, and gene expression data, respectively. The new graph is <inline-formula id="inf133">
<mml:math id="m144">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">&#x211d;</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. The projected latent feature matrix <inline-formula id="inf134">
<mml:math id="m145">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mrow>
<mml:mo>&#x27;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is obtained from <xref ref-type="disp-formula" rid="e3">Eq. 3</xref>. Therefore, given <inline-formula id="inf135">
<mml:math id="m146">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf136">
<mml:math id="m147">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, we can predict cancer subtype of the new sample by <xref ref-type="disp-formula" rid="e2">Eqs 2</xref>&#x2013;<xref ref-type="disp-formula" rid="e11">11</xref>.</p>
</sec>
<sec>
<title>Experiment settings</title>
<p>We implement M-GCN using the deep learning framework of PyTorch and train <inline-formula id="inf137">
<mml:math id="m148">
<mml:mrow>
<mml:mn>500</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> epochs for M-GCN with a learning rate of <inline-formula id="inf138">
<mml:math id="m149">
<mml:mrow>
<mml:mn>0.0001</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. The dropout rate is set as <inline-formula id="inf139">
<mml:math id="m150">
<mml:mrow>
<mml:mn>0.4</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> to avoid overfitting. We set <inline-formula id="inf140">
<mml:math id="m151">
<mml:mrow>
<mml:mn>0.82</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf141">
<mml:math id="m152">
<mml:mrow>
<mml:mn>0.79</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> as spearman correlation coefficient thresholds of BRCA and STAD, respectively. The similarity threshold parameter (<inline-formula id="inf142">
<mml:math id="m153">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) in neighbor importance estimation is set to 0.1 and 0.25 on BRCA and STAD, respectively. For BRCA, after transformation, the data dimensions of SNV (<inline-formula id="inf143">
<mml:math id="m154">
<mml:mrow>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x27;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>), CNV (<inline-formula id="inf144">
<mml:math id="m155">
<mml:mrow>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mn>2</mml:mn>
<mml:mo>&#x27;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>), and gene expression (<inline-formula id="inf145">
<mml:math id="m156">
<mml:mrow>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mn>3</mml:mn>
<mml:mo>&#x27;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>) are set as <inline-formula id="inf146">
<mml:math id="m157">
<mml:mrow>
<mml:mn>25</mml:mn>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf147">
<mml:math id="m158">
<mml:mrow>
<mml:mn>20</mml:mn>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf148">
<mml:math id="m159">
<mml:mrow>
<mml:mn>60</mml:mn>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> respectively. M-GCN model has two GCN layers in which the number of neurons in first and second hidden layer are <inline-formula id="inf149">
<mml:math id="m160">
<mml:mrow>
<mml:mn>32</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf150">
<mml:math id="m161">
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> respectively. As for STAD, we set <inline-formula id="inf151">
<mml:math id="m162">
<mml:mrow>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x27;</mml:mo>
</mml:msubsup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf152">
<mml:math id="m163">
<mml:mrow>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mn>2</mml:mn>
<mml:mo>&#x27;</mml:mo>
</mml:msubsup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf153">
<mml:math id="m164">
<mml:mrow>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mn>3</mml:mn>
<mml:mo>&#x27;</mml:mo>
</mml:msubsup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and the number of neurons in first and second hidden layer of GCN are <inline-formula id="inf154">
<mml:math id="m165">
<mml:mrow>
<mml:mn>35</mml:mn>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf155">
<mml:math id="m166">
<mml:mrow>
<mml:mn>40</mml:mn>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf156">
<mml:math id="m167">
<mml:mrow>
<mml:mn>65</mml:mn>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf157">
<mml:math id="m168">
<mml:mrow>
<mml:mn>32</mml:mn>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf158">
<mml:math id="m169">
<mml:mrow>
<mml:mn>4</mml:mn>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> respectively.</p>
</sec>
</sec>
<sec id="s2-5">
<title>Evaluation Metrics</title>
<p>We perform <inline-formula id="inf159">
<mml:math id="m170">
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>-fold cross-validation to evaluate the performance of M-GCN in molecular subtyping tasks of BRCA and STAD. The samples are divided into ten groups according to stratified sampling, nine of which are used for training data and one for test data in turn. In each training process, we select transcriptomic features by using HSIC Lasso on gene expression data of training samples (see <xref ref-type="disp-formula" rid="e1">Eq. 1</xref>) and then construct a separate sample&#x2013;sample similarity graph. In order to evaluate the performance of the method comprehensively, we take several evaluation metrics, that is, <inline-formula id="inf160">
<mml:math id="m171">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> (<inline-formula id="inf161">
<mml:math id="m172">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>), <inline-formula id="inf162">
<mml:math id="m173">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <italic>recall</italic>, <inline-formula id="inf163">
<mml:math id="m174">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf164">
<mml:math id="m175">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf165">
<mml:math id="m176">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:msub>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf166">
<mml:math id="m177">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, which are calculated as follows:<disp-formula id="e12">
<mml:math id="m178">
<mml:mrow>
<mml:mi mathvariant="italic">accuracy</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi mathvariant="italic">N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">T</mml:mi>
<mml:mi mathvariant="italic">P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="italic">T</mml:mi>
<mml:mi mathvariant="italic">N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="italic">F</mml:mi>
<mml:mi mathvariant="italic">P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="italic">F</mml:mi>
<mml:mi mathvariant="italic">N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(12)</label>
</disp-formula>
<disp-formula id="e13">
<mml:math id="m179">
<mml:mrow>
<mml:mi mathvariant="italic">precision</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="italic">T</mml:mi>
<mml:mi mathvariant="italic">P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">T</mml:mi>
<mml:mi mathvariant="italic">P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="italic">F</mml:mi>
<mml:mi mathvariant="italic">P</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(13)</label>
</disp-formula>
<disp-formula id="e14">
<mml:math id="m180">
<mml:mrow>
<mml:mi mathvariant="italic">recall</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="italic">T</mml:mi>
<mml:mi mathvariant="italic">P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">T</mml:mi>
<mml:mi mathvariant="italic">P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="italic">F</mml:mi>
<mml:mi mathvariant="italic">N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(14)</label>
</disp-formula>
<disp-formula id="e15">
<mml:math id="m181">
<mml:mrow>
<mml:mi mathvariant="italic">F</mml:mi>
<mml:mi mathvariant="italic">1</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="italic">score</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>2</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:mi mathvariant="italic">precision</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:mi mathvariant="italic">recall</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">precision</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="italic">recall</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(15)</label>
</disp-formula>
<disp-formula id="e16">
<mml:math id="m182">
<mml:mrow>
<mml:mi mathvariant="italic">Precisio</mml:mi>
<mml:msub>
<mml:mi mathvariant="italic">n</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">macro</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mfrac>
<mml:mi>1</mml:mi>
<mml:mi mathvariant="italic">v</mml:mi>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>1</mml:mi>
</mml:mrow>
<mml:mi>v</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mi mathvariant="italic">precisio</mml:mi>
<mml:msub>
<mml:mi mathvariant="italic">n</mml:mi>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mstyle>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(16)</label>
</disp-formula>
<disp-formula id="e17">
<mml:math id="m183">
<mml:mrow>
<mml:mi mathvariant="italic">Recal</mml:mi>
<mml:msub>
<mml:mi mathvariant="italic">l</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">macro</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mfrac>
<mml:mi>1</mml:mi>
<mml:mi>v</mml:mi>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="italic">c</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>1</mml:mi>
</mml:mrow>
<mml:mi mathvariant="bold-italic">v</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mi mathvariant="italic">recal</mml:mi>
<mml:msub>
<mml:mi mathvariant="italic">l</mml:mi>
<mml:mi mathvariant="italic">c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mstyle>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(17)</label>
</disp-formula>
<disp-formula id="e18">
<mml:math id="m184">
<mml:mrow>
<mml:mi mathvariant="italic">F</mml:mi>
<mml:mi>1</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="italic">scor</mml:mi>
<mml:msub>
<mml:mi mathvariant="italic">e</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">macro</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>2</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">macro</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">R</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">macro</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">macro</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">R</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">macro</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(18)</label>
</disp-formula>where true positive (<inline-formula id="inf167">
<mml:math id="m185">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) is an outcome where the model correctly predicts the positive class, and true negative (<inline-formula id="inf168">
<mml:math id="m186">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) is an outcome where the model correctly predicts the negative class. For a multi-class classification task, as long as it is not a positive class, we define it as a negative class. False positive (<inline-formula id="inf169">
<mml:math id="m187">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) is an outcome where the model incorrectly predicts the positive class, and false negative (<inline-formula id="inf170">
<mml:math id="m188">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) is an outcome where the model incorrectly predicts the negative class. <inline-formula id="inf171">
<mml:math id="m189">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf172">
<mml:math id="m190">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf173">
<mml:math id="m191">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf174">
<mml:math id="m192">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> are the most commonly used evaluation indexes for classification performance based on the above <inline-formula id="inf175">
<mml:math id="m193">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf176">
<mml:math id="m194">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf177">
<mml:math id="m195">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf178">
<mml:math id="m196">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Considering the evaluation bias caused by an unbalanced sample size in multi-classification task, <inline-formula id="inf179">
<mml:math id="m197">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf180">
<mml:math id="m198">
<mml:mrow>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:msub>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf181">
<mml:math id="m199">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are finally used for evaluation. They are weighted average of <inline-formula id="inf182">
<mml:math id="m200">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf183">
<mml:math id="m201">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf184">
<mml:math id="m202">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> on each category, with each category being equally weighted.</p>
</sec>
<sec id="s2-6">
<title>Identification of Specific Genes of Each Molecular Subtype and Functional Enrichment Analysis</title>
<p>Since selected transcriptomic features have the potential to classify samples, we further identify the specific genes of each molecular subtype. We first take z-score normalization on the expression matrix of selected genes in order to make the genes&#x2019; specificity comparable between samples. Then, in every subtype, we calculate the mean value of each gene and sort the genes in descending order. Finally, we select top 10 genes of each subtype as specific markers excluding the genes that are present in at least two subtypes.</p>
<p>In order to understand biological function of each certain gene set, we perform biological process (BP) and Kyoto Encyclopedia of Genes and Genomes (KEGG) pathways enrichment analysis on top 40 subtype-specific genes. The R package &#x201c;clusterProfiler&#x201d; is used.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<sec id="s3-1">
<title>Subtype Classification Performance of M-GCN on BRCA and STAD</title>
<p>To demonstrate the performance of our method, we compare the performance of M-GCN with six commonly used or advanced methods on STAD and BRCA molecular subtyping including traditional machine learning&#x2013;based methods, neural network&#x2013;based method, and a GCN-based method:<list list-type="bullet">
<list-item>
<p> K-nearest neighbor classifier (KNN), random forest (RF), support vector machine classifier (SVM), and Gaussian naive Bayes (GNB) are traditional machine learning methods and we utilize gene expression, SNV, and CNV data as features.</p>
</list-item>
<list-item>
<p> DeepCC is a neural network-based method which utilizes transcriptomic data and leverages feedforward neural networks to classify molecular subtypes.</p>
</list-item>
<list-item>
<p> Li&#x2019;s method is a GCN-based molecular subtyping method which integrates CNV data and gene expression data.</p>
</list-item>
</list>
</p>
<p>For BRCA, our framework M-GCN achieves best performance (<xref ref-type="fig" rid="F3">Figure 3</xref>). M-GCN achieves the highest averaged <inline-formula id="inf185">
<mml:math id="m203">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of <inline-formula id="inf186">
<mml:math id="m204">
<mml:mrow>
<mml:mn>94</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>%, which is <inline-formula id="inf187">
<mml:math id="m205">
<mml:mrow>
<mml:mn>1.5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>% higher than the second best method RF, <inline-formula id="inf188">
<mml:math id="m206">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mn>2.5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>% better than SVM and DeepCC, <inline-formula id="inf189">
<mml:math id="m207">
<mml:mn>3</mml:mn>
</mml:math>
</inline-formula>% higher than GNB, and <inline-formula id="inf190">
<mml:math id="m208">
<mml:mrow>
<mml:mn>4.8</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>% and <inline-formula id="inf191">
<mml:math id="m209">
<mml:mrow>
<mml:mn>6.7</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>% better than Li&#x2019;s method and KNN, respectively. Under the <inline-formula id="inf192">
<mml:math id="m210">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> index, RF outperforms others and M-GCN ranks second. For <inline-formula id="inf193">
<mml:math id="m211">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:msub>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf194">
<mml:math id="m212">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> indexes, M-GCN has the significantly advantage. Overall, KNN has the worst performance.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Prediction performance under four evaluation metrics of seven methods in the BRCA dataset. Pink bar represents the final performance of M-GCN; purple bar, orange bar, yellow bar, and green bar refer to the performance of KNN, RF, SVM, and GNB, respectively. Light blue bar is for the performance of DeepCC, and dark blue bar is for the performance of Li&#x2019;s method.</p>
</caption>
<graphic xlink:href="fgene-13-884028-g003.tif"/>
</fig>
<p>Furthermore, we analyze the detailed results of subtype classification. As shown in <xref ref-type="table" rid="T2">Table 2</xref>, M-GCN achieves the best performance in diagnosis of ER&#x2b; subtype patients, where <inline-formula id="inf195">
<mml:math id="m213">
<mml:mrow>
<mml:mn>95.9</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>% samples can be accurately predicted. By comparison, HER2&#x2b; patients are relatively hard to predict. Through <inline-formula id="inf196">
<mml:math id="m214">
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>-fold cross-validation, there are average <inline-formula id="inf197">
<mml:math id="m215">
<mml:mn>6</mml:mn>
</mml:math>
</inline-formula> out of <inline-formula id="inf198">
<mml:math id="m216">
<mml:mrow>
<mml:mn>30</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> samples are wrongly predicted as TNBC.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Classification results of M-GCN on each subtype of BRCA.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="center">Ratio predicted as ER&#x2b; (%)</th>
<th align="center">Ratio predicted as HER2&#x2b; (%)</th>
<th align="center">Ratio predicted as TNBC (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">ER&#x2b;</td>
<td align="char" char=".">
<bold>95.9</bold>
</td>
<td align="char" char=".">0.51</td>
<td align="char" char=".">3.59</td>
</tr>
<tr>
<td align="left">HER2&#x2b;</td>
<td align="char" char=".">0</td>
<td align="char" char=".">
<bold>80</bold>
</td>
<td align="char" char=".">20</td>
</tr>
<tr>
<td align="left">TNBC</td>
<td align="char" char=".">7</td>
<td align="char" char=".">3</td>
<td align="char" char=".">
<bold>90</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The meaning of the bold values provided in Tables 2 and 3 is &#x201c;the highest prediction ratio in each subtype&#x201d;.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>For the subtype classification task with more classes and smaller sample size, our method still performs best on STAD than other methods in all metrics (<xref ref-type="fig" rid="F4">Figure 4</xref>). The performance of neural network&#x2013;based method DeepCC ranks second, which ignores the sample&#x2013;sample graph structure information. These traditional machine learning&#x2013;based methods have better scores in four metrics by utilizing multi-omics data. Compared with the results in BRCA, Li&#x2019;s method has the largest decline of performance in STAD. According to the detailed classification results of each subtype by M-GCN under <inline-formula id="inf199">
<mml:math id="m217">
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>-fold cross-validation (<xref ref-type="table" rid="T3">Table 3</xref>), M-GCN has a <inline-formula id="inf200">
<mml:math id="m218">
<mml:mrow>
<mml:mn>100</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>% predictive power for EBV stomach cancer, and a <inline-formula id="inf201">
<mml:math id="m219">
<mml:mrow>
<mml:mn>90</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>% probability of correctly predicting the MSI type. However, GS is relatively hard to predict, especially not easily distinguishable from CIN.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Prediction performance under four evaluation metrics of seven methods in the STAD dataset. Pink bar represents the final performance of M-GCN; purple bar, orange bar, yellow bar, and green bar refer to the performance of KNN, RF, SVM, and GNB, respectively. Light blue bar is for the performance of DeepCC, and dark blue bar is for the performance of Li&#x2019;s method.</p>
</caption>
<graphic xlink:href="fgene-13-884028-g004.tif"/>
</fig>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Classification results of M-GCN on each subtype of STAD.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="center">Ratio predicted as CIN (%)</th>
<th align="center">Ratio predicted as EBV (%)</th>
<th align="center">Ratio predicted as MSI (%)</th>
<th align="center">Ratio predicted as GS (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">CIN</td>
<td align="char" char=".">
<bold>93.64</bold>
</td>
<td align="char" char=".">0</td>
<td align="char" char=".">2.72</td>
<td align="char" char=".">3.64</td>
</tr>
<tr>
<td align="left">EBV</td>
<td align="char" char=".">0</td>
<td align="char" char=".">
<bold>100</bold>
</td>
<td align="char" char=".">0</td>
<td align="char" char=".">0</td>
</tr>
<tr>
<td align="left">MSI</td>
<td align="char" char=".">6</td>
<td align="char" char=".">0</td>
<td align="char" char=".">
<bold>90</bold>
</td>
<td align="char" char=".">4</td>
</tr>
<tr>
<td align="left">GS</td>
<td align="char" char=".">20</td>
<td align="char" char=".">4</td>
<td align="char" char=".">6</td>
<td align="char" char=".">
<bold>70</bold>
</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Our framework M-GCN achieves best performance in BRCA and STAD molecular subtypes. The number of samples in BRCA is greater than that in STAD, and all of these methods in BRCA have good accuracy. The machine learning&#x2013;based methods such as RF, SVM, and GNB have a significant difference between BRCA and STAD tasks. In addition, these methods in traditional machine learning&#x2013;based methods are shallow and cannot learn the deep and complex representations of sample nodes. The performance of the neural network&#x2013;based method DeepCC is higher than most of these machine learning&#x2013;based methods, which shows the deep and non-linear representation are important. Li&#x2019;s method may suitable for the task with more samples. M-GCN still has the better scores in evaluation metrics than the multiple omic-based methods by utilizing the cleaned structure information and message passing of sample nodes.</p>
</sec>
<sec id="s3-2">
<title>Contribution of Each Element to Molecular Subtype Classification in STAD</title>
<p>After assessing the performance compared with other methods, we conduct three ablation experiments to evaluate the contributions of feature selection step, SNV data, and CNV data in STAD, respectively (<xref ref-type="fig" rid="F5">Figure 5</xref>). The basic idea of ablation experiment is to learn the framework by removing parts of it and studying its performance. In the first ablation experiment, without feature selection, we use all the gene expression features to construct sample&#x2013;sample similarity graph and take them as the transcriptomic feature for training the GCN-based molecular subtyping model. Under this setting, the prediction performance decreases by <inline-formula id="inf202">
<mml:math id="m220">
<mml:mn>7</mml:mn>
</mml:math>
</inline-formula>%, <inline-formula id="inf203">
<mml:math id="m221">
<mml:mrow>
<mml:mn>8.9</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>%, <inline-formula id="inf204">
<mml:math id="m222">
<mml:mrow>
<mml:mn>12.1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>%, and <inline-formula id="inf205">
<mml:math id="m223">
<mml:mrow>
<mml:mn>10.6</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>% in terms of <inline-formula id="inf206">
<mml:math id="m224">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf207">
<mml:math id="m225">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf208">
<mml:math id="m226">
<mml:mrow>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:msub>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf209">
<mml:math id="m227">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> when compared with M-GCN. In the second ablation experiment, we exclude SNV features from the input data. <inline-formula id="inf210">
<mml:math id="m228">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf211">
<mml:math id="m229">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf212">
<mml:math id="m230">
<mml:mrow>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:msub>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf213">
<mml:math id="m231">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> of the new trained molecular subtyping model reduce <inline-formula id="inf214">
<mml:math id="m232">
<mml:mrow>
<mml:mn>3.5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>%, <inline-formula id="inf215">
<mml:math id="m233">
<mml:mrow>
<mml:mn>2.5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>%, <inline-formula id="inf216">
<mml:math id="m234">
<mml:mrow>
<mml:mn>2.6</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>%, and <inline-formula id="inf217">
<mml:math id="m235">
<mml:mrow>
<mml:mn>2.5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>%, respectively. In the last ablation experiment, excluding CNV features from the input data, the model&#x2019;s performance has dropped <inline-formula id="inf218">
<mml:math id="m236">
<mml:mrow>
<mml:mn>2.2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>%, <inline-formula id="inf219">
<mml:math id="m237">
<mml:mrow>
<mml:mn>1.2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>%, <inline-formula id="inf220">
<mml:math id="m238">
<mml:mrow>
<mml:mn>1.7</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>%, and <inline-formula id="inf221">
<mml:math id="m239">
<mml:mrow>
<mml:mn>1.4</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>% in <inline-formula id="inf222">
<mml:math id="m240">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf223">
<mml:math id="m241">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf224">
<mml:math id="m242">
<mml:mrow>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:msub>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf225">
<mml:math id="m243">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> metrics.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Results of ablation experiment of M-GCN in STAD. Pink bar represents the final performance of M-GCN, orange bar represents the performance of M-GCN without feature selection, green bar is for the performance of M-GCN without SNV features, and blue bar is for the performance of M-GCN without CNV features.</p>
</caption>
<graphic xlink:href="fgene-13-884028-g005.tif"/>
</fig>
<p>Overall, the results of ablation experiments in STAD demonstrate that feature selection, SNV data, and CNV data are essential. Especially, feature selection makes a more significant contribution. One possible reason for this is that selected subtype-related features can help learn good representations of sample and reduce noise of the sample&#x2013;sample graph.</p>
</sec>
<sec id="s3-3">
<title>Contribution of Each Element to Molecular Subtype Classification in BRCA</title>
<p>Similarly, to explore contributions of feature selection, SNV, and CNV data for molecular subtyping of BRCA, we also perform ablation experiments. The results of three ablation experiments are shown in <xref ref-type="fig" rid="F6">Figure 6</xref>. Without the feature selection, the prediction performance decreases by <inline-formula id="inf226">
<mml:math id="m244">
<mml:mn>4</mml:mn>
</mml:math>
</inline-formula>%, <inline-formula id="inf227">
<mml:math id="m245">
<mml:mrow>
<mml:mn>24.1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>%, and <inline-formula id="inf228">
<mml:math id="m246">
<mml:mrow>
<mml:mn>25.4</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>% in terms of <inline-formula id="inf229">
<mml:math id="m247">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf230">
<mml:math id="m248">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:msub>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf231">
<mml:math id="m249">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> respectively. Under <inline-formula id="inf232">
<mml:math id="m250">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> index, ablation experiment outperforms M-GCN. Without SNV as the input, the prediction ability reduces by <inline-formula id="inf233">
<mml:math id="m251">
<mml:mrow>
<mml:mn>0.3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>%, <inline-formula id="inf234">
<mml:math id="m252">
<mml:mrow>
<mml:mn>1.6</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>%, <inline-formula id="inf235">
<mml:math id="m253">
<mml:mrow>
<mml:mn>0.9</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>%, and <inline-formula id="inf236">
<mml:math id="m254">
<mml:mrow>
<mml:mn>1.2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>% for <inline-formula id="inf237">
<mml:math id="m255">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf238">
<mml:math id="m256">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf239">
<mml:math id="m257">
<mml:mrow>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:msub>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf240">
<mml:math id="m258">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> metrics. Without CNV as the input, the model&#x2019;s performance has dropped <inline-formula id="inf241">
<mml:math id="m259">
<mml:mrow>
<mml:mn>0.2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>%, <inline-formula id="inf242">
<mml:math id="m260">
<mml:mrow>
<mml:mn>1.0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>%, <inline-formula id="inf243">
<mml:math id="m261">
<mml:mrow>
<mml:mn>0.1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>%, and <inline-formula id="inf244">
<mml:math id="m262">
<mml:mrow>
<mml:mn>0.5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>% in terms of <inline-formula id="inf245">
<mml:math id="m263">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf246">
<mml:math id="m264">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf247">
<mml:math id="m265">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:msub>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf248">
<mml:math id="m266">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> respectively.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Results of ablation experiment of M-GCN in BRCA. Pink bar represents the final performance of M-GCN, orange bar represents the performance of M-GCN without feature selection, green bar is for the performance of M-GCN without SNV features, and blue bar is for the performance of M-GCN without CNV features.</p>
</caption>
<graphic xlink:href="fgene-13-884028-g006.tif"/>
</fig>
</sec>
<sec id="s3-4">
<title>Biomarkers of Each Subtype of BRCA and Their Functions</title>
<p>On the basis of selected transcriptomic features that could accurately classify the breast cancer samples into various molecular subtypes, we further obtained the subtype-specific genes.</p>
<p>We identify ten genes with highest specificity score of each subtype, the gene lists are shown in <xref ref-type="table" rid="T4">Table 4</xref>. These identified biomarkers can significantly distinguish samples of different subtypes with the normalized gene expression by z-score transformation (<xref ref-type="fig" rid="F7">Figure 7</xref>). Among these genes, two thirds of them have been extensively studied. For example, Robinson et al. suggested that activating mutations in ESR1 were a key mechanism in acquired endocrine resistance in breast cancer therapy (<xref ref-type="bibr" rid="B39">Robinson et al., 2013</xref>). In addition, the specific biomarkers of ER&#x2b; subtype, ESR1 (<xref ref-type="bibr" rid="B39">Robinson et al., 2013</xref>; <xref ref-type="bibr" rid="B46">Spoerke et al., 2016</xref>), AGR3 (<xref ref-type="bibr" rid="B15">Garczyk et al., 2015</xref>), GATA3 (<xref ref-type="bibr" rid="B10">Ciocca et al., 2009</xref>), PCSK6 (<xref ref-type="bibr" rid="B54">Venables et al., 2008</xref>), BCAS1 (<xref ref-type="bibr" rid="B13">Fenne et al., 2013</xref>), PMAIP1 (<xref ref-type="bibr" rid="B37">Putnik et al., 2012</xref>), GPR77 (<xref ref-type="bibr" rid="B63">Zhu et al., 2021</xref>), and SCGB2A2 (<xref ref-type="bibr" rid="B18">Guan et al., 2003</xref>) have been demonstrated to be associated with breast cancer development or prognosis. For HER2&#x2b; subtype, Prat et al. have found that HER2&#x2b; patients are highly sensitive to ERBB2-targeted therapy (<xref ref-type="bibr" rid="B36">Prat et al., 2020</xref>). In addition, existing studies have reported that ERBB2 (<xref ref-type="bibr" rid="B33">Lucci et al., 2010</xref>; <xref ref-type="bibr" rid="B2">Alcal&#xe1;-Corona et al., 2018</xref>; <xref ref-type="bibr" rid="B36">Prat et al., 2020</xref>), STARD3 (<xref ref-type="bibr" rid="B40">Sahlberg et al., 2013</xref>; <xref ref-type="bibr" rid="B53">Vassilev et al., 2015</xref>; <xref ref-type="bibr" rid="B2">Alcal&#xe1;-Corona et al., 2018</xref>), GRB7 (<xref ref-type="bibr" rid="B33">Lucci et al., 2010</xref>; <xref ref-type="bibr" rid="B35">Natrajan et al., 2010</xref>; <xref ref-type="bibr" rid="B40">Sahlberg et al., 2013</xref>; <xref ref-type="bibr" rid="B2">Alcal&#xe1;-Corona et al., 2018</xref>; <xref ref-type="bibr" rid="B50">Tang et al., 2019</xref>), C17orf37 (<xref ref-type="bibr" rid="B35">Natrajan et al., 2010</xref>), PGAP3 (<xref ref-type="bibr" rid="B2">Alcal&#xe1;-Corona et al., 2018</xref>), PSMD3 (<xref ref-type="bibr" rid="B40">Sahlberg et al., 2013</xref>), and DUSP10 (<xref ref-type="bibr" rid="B33">Lucci et al., 2010</xref>) played an important role in the development and progression of breast cancer. For the TNBC subtype, although understanding of the identified subtype-specific genes is less than other two types, the roles of DGCR5 (<xref ref-type="bibr" rid="B23">Jiang et al., 2020</xref>), RAD51L1 (<xref ref-type="bibr" rid="B47">Stevens et al., 2011</xref>), and TTLL4 (<xref ref-type="bibr" rid="B3">Arnold et al., 2020</xref>) in breast cancer are well studied.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Specific biomarkers of each BRCA subtype and their enrichment pathways. The listed biomarkers rank in descending order from high to low specific score.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Molecular subtypes</th>
<th align="center">Biomarker</th>
<th align="center">Pathway and <italic>p</italic>-value</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="10" align="left">ER&#x2b;</td>
<td align="left">ESR1</td>
<td rowspan="10" align="center">Response to estradiol (<italic>p-</italic>value &#x3d; 1.09E-02)</td>
</tr>
<tr>
<td align="left">AGR3</td>
</tr>
<tr>
<td align="left">GATA3</td>
</tr>
<tr>
<td align="left">PCSK6</td>
</tr>
<tr>
<td align="left">FLJ45983</td>
</tr>
<tr>
<td align="left">BCAS1</td>
</tr>
<tr>
<td align="left">PMAIP1</td>
</tr>
<tr>
<td align="left">GPR77</td>
</tr>
<tr>
<td align="left">SCGB2A2</td>
</tr>
<tr>
<td align="left">C10orf82</td>
</tr>
<tr>
<td rowspan="10" align="left">HER2&#x2b;</td>
<td align="left">ERBB2</td>
<td rowspan="10" align="center">ERBB2 signaling pathway (<italic>p-</italic>value &#x3d; 7.52E-04)</td>
</tr>
<tr>
<td align="left">STARD3</td>
</tr>
<tr>
<td align="left">GRB7</td>
</tr>
<tr>
<td align="left">C17orf37</td>
</tr>
<tr>
<td align="left">CRISP3</td>
</tr>
<tr>
<td align="left">SERHL2</td>
</tr>
<tr>
<td align="left">PGAP3</td>
</tr>
<tr>
<td align="left">PSMD3</td>
</tr>
<tr>
<td align="left">IDH1</td>
</tr>
<tr>
<td align="left">DUSP10</td>
</tr>
<tr>
<td rowspan="10" align="left">TNBC</td>
<td align="left">MFI2</td>
<td rowspan="10" align="center">Sequestering of actin monomers (<italic>p-</italic>value &#x3d; 6.36E-05)</td>
</tr>
<tr>
<td align="left">TFCP2L1</td>
</tr>
<tr>
<td align="left">DGCR5</td>
</tr>
<tr>
<td align="left">C6orf162</td>
</tr>
<tr>
<td align="left">DCLRE1C</td>
</tr>
<tr>
<td align="left">FAM90A1</td>
</tr>
<tr>
<td align="left">RAD51L1</td>
</tr>
<tr>
<td align="left">TTLL4</td>
</tr>
<tr>
<td align="left">TM4SF1</td>
</tr>
<tr>
<td align="left">ESYT3</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Heatmap of the z-score normalized gene expression of the molecular subtype-specific biomarker genes in BRCA. Green bar, pink bar, and blue bar at the top represent ER&#x2b;, HER2&#x2b;, and TNBC subtype, respectively.</p>
</caption>
<graphic xlink:href="fgene-13-884028-g007.tif"/>
</fig>
<p>Moreover, specific genes of ER&#x2b;, HER2&#x2b;, and TNBC are significantly enriched in biological processes of response to estradiol, ERBB2 signaling pathway, and sequestering of actin monomers, respectively. Some of the findings are also highly consistent with current understandings. Daniel et al. have found that estrogen were important drivers of breast cancer proliferation and PR-B expression increased breast cancer cell growth in response to estradiol (<xref ref-type="bibr" rid="B12">Daniel et al., 2015</xref>). Shah et al. reported that HER2&#x2b; subtype of breast cancer is associated with gene amplification and/or protein overexpression of ERBB2, which leads to aggressive tumor growth and poor clinical outcome (<xref ref-type="bibr" rid="B4">Arora et al., 2008</xref>; <xref ref-type="bibr" rid="B41">Shah and Osipo, 2016</xref>). Other enriched pathways of subtype-specific genes of BRCA are listed in <xref ref-type="sec" rid="s11">Supplementary Table S1</xref>.</p>
</sec>
<sec id="s3-5">
<title>Biomarkers of Each Subtype of STAD and Their Functions</title>
<p>Compared with BRCA, the current understanding of subtype markers and biological mechanisms of STAD is much less and our analysis is expected to provide more insight. From the gene expression heatmap across all the samples, it can be concluded that biomarkers of STAD perform well in distinguishing EBV, GS, and MSI (<xref ref-type="fig" rid="F8">Figure 8</xref>). Through functional enrichment analysis, we find genes in CIN are usually enriched in regulation of cellular response to insulin stimulus, response to radiation, and telomere maintenance. Telomere maintenance in cancer cells is often accompanied by activated telomerase to protect genetically damaged DNA from normal cell senescence or apoptosis (<xref ref-type="bibr" rid="B6">Basu et al., 2013</xref>). Moreover, we also identify the specific gene CCNE1, which was reported as one of potential targets in the CIN subtype (<xref ref-type="bibr" rid="B57">Wang et al., 2019</xref>). For the EBV subtype, we infer their specific genes mainly involve in cilium organization and Herpes simplex virus 1 infection. It is well known that EBV is a gamma-herpes virus, and EBV subtyping accounts for nearly 10% of gastric carcinomas (<xref ref-type="bibr" rid="B43">Shinozaki-Ushiku et al., 2015</xref>). Identified specific genes of MSI are related to regulation of microtubule cytoskeleton organization and positive regulation of I-kappaB kinase/NF-kappaB signaling. Gullo et al. analyzed 55 differentially expressed genes in microsatellite unstable cases and found these genes associated with microtubule cytoskeleton organization (<xref ref-type="bibr" rid="B20">Gullo et al., 2018</xref>). Identified specific genes of GS are enriched in the biological process of fatty acid oxidation, protein targeting to peroxisome, and AMPK signaling pathway. He et al. discovered that mesenchymal stem cells promoted stemness and chemoresistance in stomach cancer cells through fatty acid oxidation (<xref ref-type="bibr" rid="B21">He et al., 2019</xref>). Detailed pathways related with molecular subtyping of STAD are listed in <xref ref-type="table" rid="T5">Table 5</xref> and <xref ref-type="sec" rid="s11">Supplementary Table S2</xref>. As the identified biomarkers by our method for breast cancer are greatly consistent with the current clinical consensus, we infer that the predicted biomarkers for STAD are also promising to provide guidance for researchers on the further studies of stomach cancer.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Heatmap of the z-score normalized gene expression of the molecular subtype-specific biomarker genes in STAD. Green bar, pink bar, blue bar, and purple bar at the top represent CIN, EBV, GS, and MSI subtype, respectively.</p>
</caption>
<graphic xlink:href="fgene-13-884028-g008.tif"/>
</fig>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Specific biomarkers of each STAD subtype and their enrichment pathways. The listed biomarkers rank in descending order from high to low specific score.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Molecular subtypes</th>
<th align="center">Biomarker</th>
<th align="center">Pathway and <italic>p</italic>-value</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">CIN</td>
<td align="left">DDX27, OPN3, F11R, MORC2, TMEM117, CCNE1, SMG5, CIB2, RPL39L, ZNF480</td>
<td align="left">Regulation of cellular response to insulin stimulus (<italic>p</italic>-value &#x3d; 2.43E-03), response to radiation (<italic>p</italic>-value &#x3d; 1.28E-02), and telomere maintenance (<italic>p</italic>-value &#x3d; 1.37E-02)</td>
</tr>
<tr>
<td rowspan="2" align="left">EBV</td>
<td rowspan="2" align="left">ATP2C1, BCL2L1, SYT13, ZNF8, EPB41L1, ZNF486, C12orf75, IQCB1, HABP2, RCN1</td>
<td align="left">Cilium organization (<italic>p</italic>-value &#x3d; 1.88E-02)</td>
</tr>
<tr>
<td align="left">Herpes simplex virus 1 infection (<italic>p</italic>-value &#x3d; 8.63E-04)</td>
</tr>
<tr>
<td align="left">MSI</td>
<td align="left">TMEM52, DNAJA4, DAZAP2, PAK6, MIB2, KCMF1, RAD51C, PARP3, ATP5A1, METRN</td>
<td align="left">Regulation of microtubule cytoskeleton organization (<italic>p</italic>-value &#x3d; 6.63E-03); positive regulation of I-kappaB kinase/NF-kappaB signaling (<italic>p</italic>-value &#x3d; 1.14E-02)</td>
</tr>
<tr>
<td rowspan="2" align="left">GS</td>
<td rowspan="2" align="left">MLYCD, CISD1, PRMT2, CRABP2, ALPL, ECHDC2, C2CD4B, CAMTA2, SH3BP5, IRS2</td>
<td align="left">Fatty acid oxidation (<italic>p</italic>-value &#x3d; 3.30E-06); protein targeting to peroxisome (<italic>p</italic>-value &#x3d; 2.10E-03)</td>
</tr>
<tr>
<td align="left">AMPK signaling pathway (<italic>p</italic>-value &#x3d; 3.13E-03)</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<p>The generation of large amounts of multi-omics data and development of deep learning methods offer a more effective mean to study the personalized diagnosis and treatment options of complex diseases, such as cancer (<xref ref-type="bibr" rid="B1">Ades et al., 2017</xref>; <xref ref-type="bibr" rid="B27">Krzyszczyk et al., 2018</xref>). In this study, we propose a new framework M-GCN for molecular subtyping of cancer, which is empowered by integrated multi-omics data and a robust graph convolutional network. In two case studies, that are molecular subtyping of breast and stomach cancer, M-GCN achieves best classification performance under almost all the metrics when compared with six advanced methods. As we all known, although GCN is a powerful end-to-end model, it usually ignores the noise of data and graph which makes GCN unstable. M-GCN first learns subtype-related features to denoise data and construct a relatively pure sample&#x2013;sample similarity graph. HSIC Lasso, which is recognized as an effective feature selection method, is used in our study. Furthermore, M-GCN assigns higher weights to similar nodes and utilizes layer-wise graph memory to limit the network to improve the robustness of the model based on GNNGUARD. To learn multi-view representations of multi-omics data, M-GCN then re-maps denoised three types of data into their feature spaces. Furthermore, to fuse multi-view representations of multi-omics data, M-GCN utilizes information transfer among samples in the same class and over different classes, respectively. In addition these three types of data, in the future, other omics data will be added to our framework.</p>
<p>Ablation experiments demonstrate that subtype-dependent feature selection contributes most to the improvement of classification performance of cancer molecular subtypes. Furthermore, we verify the stability of the feature selection process to ensure that obtained features are reliable. When shuffling the samples and using 90% of them to perform feature selection, we find the intersection of features picked out by the 10 rounds of feature selection processes are very large for both BRCA and STAD. This is extremely beneficial to train a stable GCN-based model.</p>
<p>On the basis of subtype-related features, we further identify a few subtype-specific features which can potentially be used for diagnostic biomarkers. In our study, ESR1, ERBB2, and MFI2 are predicted as the subtype-specific biomarkers because they have the highest specificity scores for ER&#x2b;, HER2&#x2b;, and TNBC samples, respectively. It is worth noting that ESR1 and ERBB2 are well accepted markers for ER&#x2b; and HER2&#x2b; breast subtype, indicating our prediction is highly consistent with current understanding. Although MFI2 has not been demonstrated as biomarkers of TNBC subtype by wet lab and clinical experiments, its encoding protein shares sequence similarity and iron-binding properties with members of the transferrin superfamily. Public studies have demonstrated these iron-binding properties serve iron uptake and promote cell proliferation, and high expression of these proteins are associated with the decreased overall survival of patients in many cancer types (<xref ref-type="bibr" rid="B51">Torti and Torti, 2013</xref>; <xref ref-type="bibr" rid="B49">Sun et al., 2018</xref>). As we know, TNBC patients show the poorest prognosis with a low survival time compared with other types of breast cancer. Moreover, DGCR5, having the third highest score under our prediction, reportedly incudes tumorigenesis of triple-negative breast cancer by affecting the Wnt/&#x3b2;-catenin signaling pathway. Overall, our study can accurately identify the subtype-specific biomarkers which are helpful to personalized diagnosis. So far, for many cancer types, there are still not effect means to predict their molecular subtyping. Our method is expected to be an important tool for effectively predicting molecular typing with very few genes. Moreover, the proposed framework can be used for other tasks, such as prediction of cancer staging and grading.</p>
</sec>
<sec sec-type="conclusion" id="s5">
<title>Conclusion</title>
<p>Large amount of multi-omics data generated by rapid development of high-throughput technologies has enabled data-driven methods to apply in molecular subtyping of cancer. We proposed a robust GCN-based framework M-GCN for molecular subtyping of cancer by integrating gene expression, SNV, and CNV data. In addition to comprehensive information of individual samples, M-GCN fully considers message aggregation among samples for subtype classification. Compared with other six advanced computational methods, M-GCN achieves the best classification performance for molecular subtyping of breast and stomach cancer. Through ablation experiments, we demonstrate subtype-related transcriptomics features obtained by HSIC Lasso method highly contribute to sample classification, which is probably because the selected features eliminate data noise and facilitate the construction of purified graph. On the basis of the graph structure constructed by HSIC Lasso, M-GCN further strengthens connections between new features and the graph by assigning weights. By assigning higher weights, M-GCN aims to successfully pass message in GCN. Furthermore, the identified molecular subtype-specific marker of breast cancer is highly consistent with clinical cognition, so the predicted biomarkers of stomach cancer are promising to be used for molecular typing diagnosis of patients, filling in the current gap.</p>
</sec>
</body>
<back>
<sec id="s6">
<title>Data Availability Statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s11">Supplementary Material</xref>, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s7">
<title>Author Contributions</title>
<p>CY and YC: data processing, methodology, experiments, program, discussion, and writing&#x2014;review and editing. PS: data processing, gene selection, and figures. HZ: baseline implementation. ZL and YX: guidance on biological knowledge and writing&#x2014;review. HS: resources, methodology, experiments, program, discussion, writing&#x2014;review and editing, and funding acquisition. All authors have read and agreed to the published version of the manuscript.</p>
</sec>
<sec id="s8">
<title>Funding</title>
<p>This study was supported by the National Natural Science Foundation of China (61902144).</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors, and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ack>
<p>The key data of this research was downloaded from The Cancer Genome Atlas (TCGA). We thank Yujie Gu of Jilin University for his help in method design.</p>
</ack>
<sec id="s11">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2022.884028/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2022.884028/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table2.XLSX" id="SM1" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table1.XLSX" id="SM2" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ades</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Tryfonidis</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zardavas</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>The Past and Future of Breast Cancer Treatment-From the Papyrus to Individualised Treatment Approaches</article-title>. <source>ecancer</source> <volume>11</volume>, <fpage>746</fpage>. <pub-id pub-id-type="doi">10.3332/ecancer.2017.746</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alcal&#xe1;-Corona</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Espinal-Enr&#xed;quez</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>de Anda-J&#xe1;uregui</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Hern&#xe1;ndez-Lemus</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>The Hierarchical Modular Structure of HER2&#x2b; Breast Cancer Network</article-title>. <source>Front. Physiol.</source> <volume>9</volume>, <fpage>1423</fpage>. <pub-id pub-id-type="doi">10.3389/fphys.2018.01423</pub-id> </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arnold</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Schattschneider</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Blechner</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Krisp</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Schl&#xfc;ter</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Schweizer</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Tubulin Tyrosine Ligase like 4 (TTLL4) Overexpression in Breast Cancer Cells Is Associated with Brain Metastasis and Alters Exosome Biogenesis</article-title>. <source>J. Exp. Clin. Cancer Res.</source> <volume>39</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>15</lpage>. <pub-id pub-id-type="doi">10.1186/s13046-020-01712-w</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arora</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Cuevas</surname>
<given-names>B. D.</given-names>
</name>
<name>
<surname>Russo</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Johnson</surname>
<given-names>G. L.</given-names>
</name>
<name>
<surname>Trejo</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Persistent Transactivation of EGFR and ErbB2/HER2 by Protease-Activated Receptor-1 Promotes Breast Carcinoma Cell Invasion</article-title>. <source>Oncogene</source> <volume>27</volume> (<issue>32</issue>), <fpage>4434</fpage>&#x2013;<lpage>4445</lpage>. <pub-id pub-id-type="doi">10.1038/onc.2008.84</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bass</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Thorsson</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Shmulevich</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Reynolds</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Miller</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bernard</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Comprehensive Molecular Characterization of Gastric Adenocarcinoma</article-title>. <source>Nature</source> <volume>513</volume> (<issue>7517</issue>), <fpage>202</fpage>&#x2013;<lpage>209</lpage>. <pub-id pub-id-type="doi">10.1038/nature13480</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Basu</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Skinner</surname>
<given-names>H. G.</given-names>
</name>
<name>
<surname>Litzelman</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Vanderboom</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Baichoo</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Boardman</surname>
<given-names>L. A.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Telomeres and Telomere Dynamics: Relevance to Cancers of the GI Tract</article-title>. <source>Expert Rev. Gastroenterol. Hepatol.</source> <volume>7</volume> (<issue>8</issue>), <fpage>733</fpage>&#x2013;<lpage>748</lpage>. <pub-id pub-id-type="doi">10.1586/17474124.2013.848790</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bradner</surname>
<given-names>J. E.</given-names>
</name>
<name>
<surname>Hnisz</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Young</surname>
<given-names>R. A.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Transcriptional Addiction in Cancer</article-title>. <source>Cell</source> <volume>168</volume> (<issue>4</issue>), <fpage>629</fpage>&#x2013;<lpage>643</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2016.12.013</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Rong</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Tao</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Similarity Fusion via Exploiting High Order Proximity for Cancer Subtyping</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinformatics</source>, <fpage>1</fpage>. <pub-id pub-id-type="doi">10.1109/tcbb.2021.3139597</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Goodison</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Deep-learning Approach to Identifying Cancer Subtypes Using High-Dimensional Genomic Data</article-title>. <source>Bioinformatics</source> <volume>36</volume> (<issue>5</issue>), <fpage>1476</fpage>&#x2013;<lpage>1483</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz769</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ciocca</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Daskalakis</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Ciocca</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Ruiz-Orrico</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Palazzo</surname>
<given-names>J. P.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>The Significance of GATA3 Expression in Breast Cancer: a 10-year Follow-Up Study</article-title>. <source>Hum. Pathol.</source> <volume>40</volume> (<issue>4</issue>), <fpage>489</fpage>&#x2013;<lpage>495</lpage>. <pub-id pub-id-type="doi">10.1016/j.humpath.2008.09.010</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Dai</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). &#x201c;<article-title>Adversarial Attack on Graph Structured Data</article-title>,&#x201d; in <source>Proceedings of the 35th International Conference on Machine Learning</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Jennifer</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Andreas</surname>
<given-names>K.</given-names>
</name>
</person-group>. (<publisher-loc>Stockholmsm&#x00E4;ssan, Stockholm, Sweden</publisher-loc>: <publisher-name>PMLR</publisher-name>). </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Daniel</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Gaviglio</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Knutson</surname>
<given-names>T. P.</given-names>
</name>
<name>
<surname>Ostrander</surname>
<given-names>J. H.</given-names>
</name>
<name>
<surname>D&#x27;Assoro</surname>
<given-names>A. B.</given-names>
</name>
<name>
<surname>Ravindranathan</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Progesterone Receptor-B Enhances Estrogen Responsiveness of Breast Cancer Cells via Scaffolding PELP1- and Estrogen Receptor-Containing Transcription Complexes</article-title>. <source>Oncogene</source> <volume>34</volume> (<issue>4</issue>), <fpage>506</fpage>&#x2013;<lpage>515</lpage>. <pub-id pub-id-type="doi">10.1038/onc.2013.579</pub-id> </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fenne</surname>
<given-names>I. S.</given-names>
</name>
<name>
<surname>Helland</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Fl&#xe5;geng</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Dankel</surname>
<given-names>S. N.</given-names>
</name>
<name>
<surname>Mellgren</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Sagen</surname>
<given-names>J. V.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Downregulation of Steroid Receptor Coactivator-2 Modulates Estrogen-Responsive Genes and Stimulates Proliferation of Mcf-7 Breast Cancer Cells</article-title>. <source>PloS one</source> <volume>8</volume> (<issue>7</issue>), <fpage>e70096</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0070096</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fessler</surname>
<given-names>E.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>DeepCC: a Novel Deep Learning-Based Framework for Cancer Molecular Subtype Classification</article-title>. <source>Oncogenesis</source> <volume>8</volume> (<issue>9</issue>), <fpage>1</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1038/s41389-019-0157-8</pub-id> </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Garczyk</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>von Stillfried</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Antonopoulos</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Hartmann</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Schrauder</surname>
<given-names>M. G.</given-names>
</name>
<name>
<surname>Fasching</surname>
<given-names>P. A.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>AGR3 in Breast Cancer: Prognostic Impact and Suitable Serum-Based Biomarker for Early Cancer Detection</article-title>. <source>PloS one</source> <volume>10</volume> (<issue>4</issue>), <fpage>e0122106</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0122106</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gonz&#xe1;lez-Garc&#xed;a</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Sol&#xe9;</surname>
<given-names>R. V.</given-names>
</name>
<name>
<surname>Costa</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Metapopulation Dynamics and Spatial Heterogeneity in Cancer</article-title>. <source>Proc. Natl. Acad. Sci. U.S.A.</source> <volume>99</volume> (<issue>20</issue>), <fpage>13085</fpage>&#x2013;<lpage>13089</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.202139299</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Greenman</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Stephens</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Dalgliesh</surname>
<given-names>G. L.</given-names>
</name>
<name>
<surname>Hunter</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bignell</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2007</year>). <article-title>Patterns of Somatic Mutation in Human Cancer Genomes</article-title>. <source>Nature</source> <volume>446</volume> (<issue>7132</issue>), <fpage>153</fpage>&#x2013;<lpage>158</lpage>. <pub-id pub-id-type="doi">10.1038/nature05610</pub-id> </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guan</surname>
<given-names>X.-f.</given-names>
</name>
<name>
<surname>Hamedani</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Adeyinka</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Walker</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Kemp</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Murphy</surname>
<given-names>L. C.</given-names>
</name>
<etal/>
</person-group> (<year>2003</year>). <article-title>Relationship between Mammaglobin Expression and Estrogen Receptor Status in Breast Tumors</article-title>. <source>Endo</source> <volume>21</volume> (<issue>3</issue>), <fpage>245</fpage>&#x2013;<lpage>250</lpage>. <pub-id pub-id-type="doi">10.1385/ENDO:21:3:245</pub-id> </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chance</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Barnholtz-Sloan</surname>
<given-names>J. S.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Splitting Random forest (SRF) for Determining Compact Sets of Genes that Distinguish between Cancer Subtypes</article-title>. <source>J. Clin. Bioinformatics</source> <volume>2</volume> (<issue>1</issue>), <fpage>13</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1186/2043-9113-2-13</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gullo</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Carvalho</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Martins</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Lemos</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Monteiro</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ferreira</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>The Transcriptomic Landscape of Gastric Cancer: Insights into Epstein-Barr Virus Infected and Microsatellite Unstable Tumors</article-title>. <source>Ijms</source> <volume>19</volume> (<issue>7</issue>), <fpage>2079</fpage>. <pub-id pub-id-type="doi">10.3390/ijms19072079</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>He</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Q.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>MSC-regulated lncRNA MACC1-AS1 Promotes Stemness and Chemoresistance through Fatty Acid Oxidation in Gastric Cancer</article-title>. <source>Oncogene</source> <volume>38</volume> (<issue>23</issue>), <fpage>4637</fpage>&#x2013;<lpage>4654</lpage>. <pub-id pub-id-type="doi">10.1038/s41388-019-0747-0</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xiang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Johnson</surname>
<given-names>T. S.</given-names>
</name>
<name>
<surname>Helm</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>C. Y.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>SALMON: Survival Analysis Learning with Multi-Omics Neural Networks on Breast Cancer</article-title>. <source>Front. Genet.</source> <volume>10</volume>, <fpage>166</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2019.00166</pub-id> </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Long Non-coding RNA DGCR5 Incudes Tumorigenesis of Triple-Negative Breast Cancer by Affecting Wnt/&#x3b2;-Catenin Signaling Pathway</article-title>. <source>J. BUON</source> <volume>25</volume> (<issue>2</issue>), <fpage>702</fpage>&#x2013;<lpage>708</lpage>. </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jin</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ji</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Aggarwal</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Adversarial Attacks and Defenses on Graphs: A Review, A Tool and Empirical Studies</article-title>. <comment>Available at: https://ui.adsabs.harvard.edu/abs/2020arXiv200300653J (Accessed March 01, 2020)</comment>. </citation>
</ref>
<ref id="B25">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kingma</surname>
<given-names>D. P.</given-names>
</name>
<name>
<surname>Ba</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>Adam: A Method for Stochastic Optimization</article-title>,&#x201d; in <conf-name>International Conference on Learning Representations</conf-name>, <conf-loc>San Diego, United States</conf-loc>, <fpage>1</fpage>&#x2013;<lpage>15</lpage>. </citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kipf</surname>
<given-names>T. N.</given-names>
</name>
<name>
<surname>Welling</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Semi-Supervised Classification with Graph Convolutional Networks</article-title>. <comment>Available at: https://ui.adsabs.harvard.edu/abs/2016arXiv160902907K (Accessed September 01, 2016)</comment>. </citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Krzyszczyk</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Acevedo</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Davidoff</surname>
<given-names>E. J.</given-names>
</name>
<name>
<surname>Timmins</surname>
<given-names>L. M.</given-names>
</name>
<name>
<surname>Marrero-Berrios</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Patel</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>The Growing Role of Precision and Personalized Medicine for Cancer Treatment</article-title>. <source>Technology</source> <volume>06</volume> (<issue>03n04</issue>), <fpage>79</fpage>&#x2013;<lpage>100</lpage>. <pub-id pub-id-type="doi">10.1142/S2339547818300020</pub-id> </citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kuijjer</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>Paulson</surname>
<given-names>J. N.</given-names>
</name>
<name>
<surname>Salzman</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Quackenbush</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Cancer Subtype Identification Using Somatic Mutation Data</article-title>. <source>Br. J. Cancer</source> <volume>118</volume> (<issue>11</issue>), <fpage>1492</fpage>&#x2013;<lpage>1501</lpage>. <pub-id pub-id-type="doi">10.1038/s41416-018-0109-7</pub-id> </citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lim</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Sung</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020a</year>). <article-title>Cancer Subtype Classification and Modeling by Pathway Attention and Propagation</article-title>. <source>Bioinformatics</source> <volume>36</volume> (<issue>12</issue>), <fpage>3818</fpage>&#x2013;<lpage>3824</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa203</pub-id> </citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname>
<given-names>Y.-M.</given-names>
</name>
<name>
<surname>Oh</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Go</surname>
<given-names>J.-H.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Choi</surname>
<given-names>S.-Y.</given-names>
</name>
</person-group> (<year>2020b</year>). <article-title>Molecular Subtypes of Triple-Negative Breast Cancer: Understanding of Subtype Categories and Clinical Implication</article-title>. <source>Genes Genom</source> <volume>42</volume>, <fpage>1381</fpage>&#x2013;<lpage>1387</lpage>. <pub-id pub-id-type="doi">10.1007/s13258-020-01014-7</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Nabavi</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Cancer Molecular Subtype Classification by Graph Convolutional Networks on Multi-Omics Data</article-title>,&#x201d; in <conf-name>Proceedings of the 12th ACM Conference on Bioinformatics, Computational Biology, and Health Informatics</conf-name>, <conf-loc>Gainesville, FL</conf-loc>, <conf-date>August 2021</conf-date>. <pub-id pub-id-type="doi">10.1145/3459930.3469542</pub-id> </citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Classifying Breast Cancer Subtypes Using Deep Neural Networks Based on Multi-Omics Data</article-title>. <source>Genes</source> <volume>11</volume> (<issue>8</issue>), <fpage>888</fpage>. <pub-id pub-id-type="doi">10.3390/genes11080888</pub-id> </citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lucci</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Orlandi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Triulzi</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Tagliabue</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Balsari</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Villa-Moruzzi</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Expression Profile of Tyrosine Phosphatases in HER2 Breast Cancer Cells and Tumors</article-title>. <source>Cell Oncol</source> <volume>32</volume> (<issue>5-6</issue>), <fpage>361</fpage>&#x2013;<lpage>372</lpage>. <pub-id pub-id-type="doi">10.3233/CLO-2010-0520</pub-id> </citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Memon</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Gill</surname>
<given-names>M. B.</given-names>
</name>
<name>
<surname>Papachristou</surname>
<given-names>E. K.</given-names>
</name>
<name>
<surname>Ochoa</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>D&#x2019;Santos</surname>
<given-names>C. S.</given-names>
</name>
<name>
<surname>Miller</surname>
<given-names>M. L.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Copy Number Aberrations Drive Kinase Rewiring, Leading to Genetic Vulnerabilities in Cancer</article-title>. <source>Cel Rep.</source> <volume>35</volume> (<issue>7</issue>), <fpage>109155</fpage>. <pub-id pub-id-type="doi">10.1016/j.celrep.2021.109155</pub-id> </citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Natrajan</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Weigelt</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Mackay</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Geyer</surname>
<given-names>F. C.</given-names>
</name>
<name>
<surname>Grigoriadis</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>D. S. P.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>An Integrative Genomic and Transcriptomic Analysis Reveals Molecular Pathways and Networks Regulated by Copy Number Aberrations in Basal-like, HER2 and Luminal Cancers</article-title>. <source>Breast Cancer Res. Treat.</source> <volume>121</volume> (<issue>3</issue>), <fpage>575</fpage>&#x2013;<lpage>589</lpage>. <pub-id pub-id-type="doi">10.1007/s10549-009-0501-3</pub-id> </citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Prat</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Pascual</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>De Angelis</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Gutierrez</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Llombart-Cussac</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>HER2-enriched Subtype and ERBB2 Expression in HER2-Positive Breast Cancer Treated with Dual HER2 Blockade</article-title>. <source>JNCI: J. Natl. Cancer Inst.</source> <volume>112</volume> (<issue>1</issue>), <fpage>46</fpage>&#x2013;<lpage>54</lpage>. <pub-id pub-id-type="doi">10.1093/jnci/djz042</pub-id> </citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Putnik</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Gustafsson</surname>
<given-names>J.-&#xc5;.</given-names>
</name>
<name>
<surname>Dahlman-Wright</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Global Identification of Genes Regulated by Estrogen Signaling and Demethylation in MCF-7 Breast Cancer Cells</article-title>. <source>Biochem. biophysical Res. Commun.</source> <volume>426</volume> (<issue>1</issue>), <fpage>26</fpage>&#x2013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1016/j.bbrc.2012.08.007</pub-id> </citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rhee</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Seo</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Hybrid Approach of Relation Network and Localized Graph Convolutional Filtering for Breast Cancer Subtype Classification</article-title>. <comment>Available at: https://ui.adsabs.harvard.edu/abs/2017arXiv171105859R (Accessed November 01, 2017)</comment>. </citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Robinson</surname>
<given-names>D. R.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y.-M.</given-names>
</name>
<name>
<surname>Vats</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Lonigro</surname>
<given-names>R. J.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Activating ESR1 Mutations in Hormone-Resistant Metastatic Breast Cancer</article-title>. <source>Nat. Genet.</source> <volume>45</volume> (<issue>12</issue>), <fpage>1446</fpage>&#x2013;<lpage>1451</lpage>. <pub-id pub-id-type="doi">10.1038/ng.28210.1038/ng.2823</pub-id> </citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sahlberg</surname>
<given-names>K. K.</given-names>
</name>
<name>
<surname>Hongisto</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Edgren</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>M&#xe4;kel&#xe4;</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Hellstr&#xf6;m</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Due</surname>
<given-names>E. U.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>The HER2 Amplicon Includes Several Genes Required for the Growth and Survival of HER2 Positive Breast Cancer Cells</article-title>. <source>Mol. Oncol.</source> <volume>7</volume> (<issue>3</issue>), <fpage>392</fpage>&#x2013;<lpage>401</lpage>. <pub-id pub-id-type="doi">10.1016/j.molonc.2012.10.012</pub-id> </citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shah</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Osipo</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Cancer Stem Cells and HER2 Positive Breast Cancer: The story So Far</article-title>. <source>Genes Dis.</source> <volume>3</volume> (<issue>2</issue>), <fpage>114</fpage>&#x2013;<lpage>123</lpage>. <pub-id pub-id-type="doi">10.1016/j.gendis.2016.02.002</pub-id> </citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shieh</surname>
<given-names>G. S.</given-names>
</name>
<name>
<surname>Bai</surname>
<given-names>C.-H.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Identify Breast Cancer Subtypes by Gene Expression Profiles</article-title>. <source>J. Data Sci.</source> <volume>2</volume> (<issue>2</issue>), <fpage>165</fpage>&#x2013;<lpage>175</lpage>. </citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shinozaki-Ushiku</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kunita</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Fukayama</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Update on Epstein-Barr Virus and Gastric Cancer (Review)</article-title>. <source>Int. J. Oncol.</source> <volume>46</volume> (<issue>4</issue>), <fpage>1421</fpage>&#x2013;<lpage>1434</lpage>. <pub-id pub-id-type="doi">10.3892/ijo.2015.2856</pub-id> </citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shipitsin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Campbell</surname>
<given-names>L. L.</given-names>
</name>
<name>
<surname>Argani</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Weremowicz</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bloushtain-Qimron</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Yao</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2007</year>). <article-title>Molecular Definition of Breast Tumor Heterogeneity</article-title>. <source>Cancer cell</source> <volume>11</volume> (<issue>3</issue>), <fpage>259</fpage>&#x2013;<lpage>273</lpage>. <pub-id pub-id-type="doi">10.1016/j.ccr.2007.01.013</pub-id> </citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Singh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shannon</surname>
<given-names>C. P.</given-names>
</name>
<name>
<surname>Gautier</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Rohart</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Vacher</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Tebbutt</surname>
<given-names>S. J.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>DIABLO: an Integrative Approach for Identifying Key Molecular Drivers from Multi-Omics Assays</article-title>. <source>Bioinformatics</source> <volume>35</volume> (<issue>17</issue>), <fpage>3055</fpage>&#x2013;<lpage>3062</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty1054</pub-id> </citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Spoerke</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Gendreau</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Walter</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Qiu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wilson</surname>
<given-names>T. R.</given-names>
</name>
<name>
<surname>Savage</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Heterogeneity and Clinical Significance of ESR1 Mutations in ER-Positive Metastatic Breast Cancer Patients Receiving Fulvestrant</article-title>. <source>Nat. Commun.</source> <volume>7</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1038/ncomms11579</pub-id> </citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stevens</surname>
<given-names>K. N.</given-names>
</name>
<name>
<surname>Vachon</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Slager</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lesnick</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Olswold</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Common Breast Cancer Susceptibility Loci Are Associated with Triple-Negative Breast Cancer</article-title>. <source>Cancer Res.</source> <volume>71</volume> (<issue>19</issue>), <fpage>6240</fpage>&#x2013;<lpage>6249</lpage>. <pub-id pub-id-type="doi">10.1158/0008-5472.CAN-11-1266</pub-id> </citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Subramanian</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Verma</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jere</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Anamika</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Multi-omics Data Integration, Interpretation, and its Application</article-title>. <source>Bioinform Biol. Insights</source> <volume>14</volume>, <fpage>117793221989905</fpage>. <pub-id pub-id-type="doi">10.1177/1177932219899051</pub-id> </citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sheng</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Fenton Reactions Drive Nucleotide and ATP Syntheses in Cancer</article-title>. <source>J. Mol. Cel. Biol.</source> <volume>10</volume> (<issue>5</issue>), <fpage>448</fpage>&#x2013;<lpage>459</lpage>. <pub-id pub-id-type="doi">10.1093/jmcb/mjy039</pub-id> </citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Epigenetically Altered miR-193a-3p P-romotes HER2 P-ositive B-reast C-ancer A-ggressiveness by T-argeting GRB7</article-title>. <source>Int. J. Mol. Med.</source> <volume>43</volume> (<issue>6</issue>), <fpage>2352</fpage>&#x2013;<lpage>2360</lpage>. <pub-id pub-id-type="doi">10.3892/ijmm.2019.4167</pub-id> </citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Torti</surname>
<given-names>S. V.</given-names>
</name>
<name>
<surname>Torti</surname>
<given-names>F. M.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Iron and Cancer: More Ore to Be Mined</article-title>. <source>Nat. Rev. Cancer</source> <volume>13</volume> (<issue>5</issue>), <fpage>342</fpage>&#x2013;<lpage>355</lpage>. <pub-id pub-id-type="doi">10.1038/nrc3495</pub-id> </citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Toss</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cristofanilli</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Molecular Characterization and Targeted Therapeutic Approaches in Breast Cancer</article-title>. <source>Breast Cancer Res.</source> <volume>17</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1186/s13058-015-0560-9</pub-id> </citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vassilev</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Sihto</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>H&#xf6;ltt&#xe4;-Vuori</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ilola</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lundin</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Elevated Levels of StAR-Related Lipid Transfer Protein 3 Alter Cholesterol Balance and Adhesiveness of Breast Cancer Cells</article-title>. <source>Am. J. Pathol.</source> <volume>185</volume> (<issue>4</issue>), <fpage>987</fpage>&#x2013;<lpage>1000</lpage>. <pub-id pub-id-type="doi">10.1016/j.ajpath.2014.12.018</pub-id> </citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Venables</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Klinck</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Bramard</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Inkel</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Dufresne-Martin</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Koh</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2008</year>). <article-title>Identification of Alternative Splicing Markers for Breast Cancer</article-title>. <source>Cancer Res.</source> <volume>68</volume> (<issue>22</issue>), <fpage>9525</fpage>&#x2013;<lpage>9531</lpage>. <pub-id pub-id-type="doi">10.1158/0008-5472.CAN-08-1769</pub-id> </citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vuong</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Simpson</surname>
<given-names>P. T.</given-names>
</name>
<name>
<surname>Green</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Cummings</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Lakhani</surname>
<given-names>S. R.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Molecular Classification of Breast Cancer</article-title>. <source>Virchows Arch.</source> <volume>465</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.1007/s00428-014-1593-7</pub-id> </citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Mezlini</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Demir</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Fiume</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Tu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Brudno</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Similarity Network Fusion for Aggregating Data Types on a Genomic Scale</article-title>. <source>Nat. Methods</source> <volume>11</volume> (<issue>3</issue>), <fpage>333</fpage>&#x2013;<lpage>337</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.2810</pub-id> </citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Molecular Classification of Gastric Adenocarcinoma</article-title>. <source>Gastroenterol. Res.</source> <volume>12</volume> (<issue>6</issue>), <fpage>275</fpage>&#x2013;<lpage>282</lpage>. <pub-id pub-id-type="doi">10.14740/gr1187</pub-id> </citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Weinstein</surname>
<given-names>J. N.</given-names>
</name>
<name>
<surname>Creighton</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Collisson</surname>
<given-names>E. A.</given-names>
</name>
<name>
<surname>Mills</surname>
<given-names>G. B.</given-names>
</name>
<name>
<surname>Shaw</surname>
<given-names>K. R. M.</given-names>
</name>
<name>
<surname>Ozenberger</surname>
<given-names>B. A.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>The Cancer Genome Atlas Pan-Cancer Analysis Project</article-title>. <source>Nat. Genet.</source> <volume>45</volume> (<issue>10</issue>), <fpage>1113</fpage>&#x2013;<lpage>1120</lpage>. <pub-id pub-id-type="doi">10.1038/ng.2764</pub-id> </citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>A Pathways-Based Prediction Model for Classifying Breast Cancer Subtypes</article-title>. <source>Oncotarget</source> <volume>8</volume> (<issue>35</issue>), <fpage>58809</fpage>&#x2013;<lpage>58822</lpage>. <pub-id pub-id-type="doi">10.18632/oncotarget.18544</pub-id> </citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Simultaneous Interrogation of Cancer Omics to Identify Subtypes with Significant Clinical Differences</article-title>. <source>Front. Genet.</source> <volume>10</volume>, <fpage>236</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2019.00236</pub-id> </citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yamada</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Jitkrittum</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Sigal</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xing</surname>
<given-names>E. P.</given-names>
</name>
<name>
<surname>Sugiyama</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>High-dimensional Feature Selection by Feature-wise Kernelized Lasso</article-title>. <source>Neural Comput.</source> <volume>26</volume> (<issue>1</issue>), <fpage>185</fpage>&#x2013;<lpage>207</lpage>. <pub-id pub-id-type="doi">10.1162/NECO_a_00537</pub-id> </citation>
</ref>
<ref id="B62">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zitnik</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Gnnguard: Defending Graph Neural Networks against Adversarial Attacks</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>33</volume>, <fpage>9263</fpage>&#x2013;<lpage>9275</lpage>. </citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>An Integrated Analysis of C5AR2 Related to Malignant Properties and Immune Infiltration of Breast Cancer</article-title>. <source>Front. Oncol.</source> <volume>11</volume>, <fpage>736725</fpage>. <pub-id pub-id-type="doi">10.3389/fonc.2021.736725</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>