<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1466825</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2024.1466825</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Multi-fusion strategy network-guided cancer subtypes discovering based on multi-omics data</article-title>
<alt-title alt-title-type="left-running-head">Liu et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2024.1466825">10.3389/fgene.2024.1466825</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Jian</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1353121/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xue</surname>
<given-names>Xinzheng</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2862681/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wen</surname>
<given-names>Pengbo</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1142428/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Song</surname>
<given-names>Qian</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yao</surname>
<given-names>Jun</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Ge</surname>
<given-names>Shuguang</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Information and Control Engineering</institution>, <institution>China University of Mining and Technology</institution>, <addr-line>Xuzhou</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>School of Medical Information and Engineering</institution>, <institution>Xuzhou Medical University</institution>, <addr-line>Xuzhou</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Gynecology and Obstetrics, Taizhou Cancer Hospital</institution>, <addr-line>Wenling</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Department of Colorectal Surgery, Taizhou Cancer Hospital</institution>, <addr-line>Wenling</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/13286/overview">Yan Cui</ext-link>, University of Tennessee Health Science Center (UTHSC), United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/832518/overview">Junwei Luo</ext-link>, Henan Polytechnic University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1896965/overview">Hyo Young Choi</ext-link>, University of Tennessee Health Science Center (UTHSC), United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Shuguang Ge, <email>gesgcumt17@163.com</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>14</day>
<month>11</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1466825</elocation-id>
<history>
<date date-type="received">
<day>18</day>
<month>07</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>04</day>
<month>11</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Liu, Xue, Wen, Song, Yao and Ge.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Liu, Xue, Wen, Song, Yao and Ge</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>The combination of next-generation sequencing technology and Cancer Genome Atlas (TCGA) data provides unprecedented opportunities for the discovery of cancer subtypes. Through comprehensive analysis and in-depth analysis of the genomic data of a large number of cancer patients, researchers can more accurately identify different cancer subtypes and reveal their molecular heterogeneity.</p>
</sec>
<sec>
<title>Methods</title>
<p>In this paper, we propose the SMMSN (Self-supervised Multi-fusion Strategy Network) model for the discovery of cancer subtypes. SMMSN can not only fuse multi-level data representations of single omics data by Graph Convolutional Network (GCN) and Stacked Autoencoder Network (SAE), but also achieve the organic fusion of multi- -omics data through multiple fusion strategies. In response to the problem of lack label information in multi-omics data, SMMSN propose to use dual self-supervise method to cluster cancer subtypes from the integrated data.</p>
</sec>
<sec>
<title>Results</title>
<p>We conducted experiments on three labeled and five unlabeled multi-omics datasets to distinguish potential cancer subtypes. Kaplan Meier survival curves and other results showed that SMMSN can obtain cancer subtypes with significant differences.</p>
</sec>
<sec>
<title>Discussion</title>
<p>In the case analysis of Glioblastoma Multiforme (GBM) and Breast Invasive Carcinoma (BIC), we conducted survival time and age distribution analysis, drug response analysis, differential expression analysis, functional enrichment analysis on the predicted cancer subtypes. The research results showed that SMMSN can discover clinically meaningful cancer subtypes.</p>
</sec>
</abstract>
<kwd-group>
<kwd>cancer subtypes discovering</kwd>
<kwd>multi-omics data</kwd>
<kwd>clustering</kwd>
<kwd>deep learning</kwd>
<kwd>fusion strategy</kwd>
</kwd-group>
<contract-sponsor id="cn001">National Natural Science Foundation of China<named-content content-type="fundref-id">10.13039/501100001809</named-content>
</contract-sponsor>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Genomics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Cancer is a heterogeneous disease characterized by diverse pathogenic mechanisms and clinical features (<xref ref-type="bibr" rid="B35">Wang et al., 2023</xref>). Research has shown that genomic alterations, such as copy number variations and somatic mutations, can lead to cancer development (<xref ref-type="bibr" rid="B41">Xu et al., 2023</xref>). Due to high heterogeneity, patients with similar phenotypes often exhibit different genomic changes, resulting in varied symptoms among cancer subtypes, which significantly impacts clinical diagnosis and prognosis (<xref ref-type="bibr" rid="B17">Jin et al., 2023</xref>). A major focus in current cancer research is predicting molecular subtypes using multi-omics data (<xref ref-type="bibr" rid="B18">Livesey et al., 2023</xref>; <xref ref-type="bibr" rid="B7">Chen et al., 2023</xref>). Classifying cancer subtypes can enhance our understanding of cancer pathogenesis and aid in personalized treatment approaches (<xref ref-type="bibr" rid="B28">Sosinsky et al., 2024</xref>).</p>
<p>Early research on cancer subtype discovery primarily concentrated on single omics data, such as gene expression data, using general clustering algorithms (<xref ref-type="bibr" rid="B25">Rappoport and Shamir, 2018</xref>). However, with the rapid accumulation of diverse omics data and the development of extensive cancer genome databases, the field has evolved significantly. One notable resource is The Cancer Genome Atlas (TCGA) (<xref ref-type="bibr" rid="B1">Akbani et al., 2014</xref>; <xref ref-type="bibr" rid="B2">Baird and Roychoudhuri, 2024</xref>), which has extensively studied multi-omics data from various cancer types across numerous patient samples. This wealth of sequencing data offers unprecedented opportunities to utilize multi-omics approaches for the discovery of cancer subtypes, paving the way for more precise and comprehensive cancer research and treatment strategies.</p>
<p>Researchers have proposed various methods for predicting cancer subtypes using multi-omics data. The simplest approach involves concatenating different biological data to form a single input matrix, followed by applying general clustering methods to identify cancer subtypes. For instance, <xref ref-type="bibr" rid="B39">Wu et al. (2015)</xref> introduced a comprehensive probability model called LRAcluster, based on low-rank approximation, to swiftly mine the shared main features across multiple omics data types. However, such methods often overlook differences in distribution and dimensionality among omics data, making it challenging to accurately characterize the input features. To address this, more sophisticated clustering strategies have been developed that consider the unique characteristics of each data source. The iCluster model (<xref ref-type="bibr" rid="B26">Shen et al., 2009</xref>) assumes that each omics dataset contains latent variables and employs a sparse method for gene selection and clustering. However, iCluster is limited to clustering continuous data types. Building on this, <xref ref-type="bibr" rid="B21">Mo et al. (2013)</xref> proposed iClusterPlus, an algorithm capable of jointly modeling multiple types of omics data, including continuous, count, and binary data. Additionally, Shi et al. designed the PFA algorithm (<xref ref-type="bibr" rid="B27">Shi et al., 2017</xref>), which maps each type of omics data to its corresponding low-dimensional space and performs automated information alignment and bias correction to achieve global pattern fusion in the feature space. These advancements offer more accurate and nuanced approaches to cancer subtype prediction, leveraging the full potential of multi-omics data.</p>
<p>The approaches mentioned primarily emphasize the representational characteristics of omics data while neglecting the structural insights that can illuminate similarities among patients, which are crucial for effective data learning. Spectral clustering (<xref ref-type="bibr" rid="B19">Luxburg, 2007</xref>) stands out as a method that captures such structural features by constructing graphs from data samples and leveraging graph-based clustering. Building on spectral clustering, various data integration algorithms have been developed. For instance, <xref ref-type="bibr" rid="B34">Wang et al. (2014)</xref> introduced the SNF method, which establishes similarity networks for diverse omics data types and integrates these networks using non-linear fusion techniques, thereby exploiting the complementary nature of the data. Expanding on these concepts, <xref ref-type="bibr" rid="B20">Ma and Zhang (2017)</xref> proposed the ANF method, which constructs K-nearest neighbor (KNN) networks for different omics datasets. These individual networks are then amalgamated into a unified fusion network using a random walk approach. To address the optimization challenges of spectral clustering, <xref ref-type="bibr" rid="B43">Yu et al. (2019)</xref> employed a linear search technique on the Stiefel manifold space, culminating in the MVCMO algorithm designed specifically for clustering multi-omics data. These advancements not only enhance our ability to extract meaningful insights from omics data but also underscore the importance of structural information for more robust data analysis and learning.</p>
<p>Deep learning has rapidly emerged as a research hotspot in the field of Artificial Intelligence (AI), especially in image data processing. Many deep learning-based methods for processing omics data have also been proposed to address the problem of cancer subtype discovery. <xref ref-type="bibr" rid="B6">Chen et al. (2020)</xref> proposed the DeepType algorithm for cancer classification, which combines supervised learning, unsupervised learning, and dimensionality reduction to learn data representations with clustering structures. <xref ref-type="bibr" rid="B38">Way and Greene (2018)</xref> utilized Variational Autoencoders (VAE) to compress gene expression features, thereby uncovering biologically relevant latent spaces. <xref ref-type="bibr" rid="B40">Xu et al. (2019)</xref> employed a Stacked Autoencoder (SAE) model to learn high-level representations of each omics data type, integrating these representations into an autoencoder layer to achieve a complex representation. They then used a Deep Flexible Neural Forest (DFNForest) model to classify the samples. These methods leverage deep learning to extract high-level feature representations from omics data and predict cancer subtypes based on these learned features. However, they often do not utilize the structural information inherent in omics data, which can be crucial for a more comprehensive understanding and prediction of cancer subtypes.</p>
<p>Graph Convolutional Networks (GCNs) (<xref ref-type="bibr" rid="B31">Thomas and Kipf, 2017</xref>) extend Convolutional Neural Networks (CNNs) to graph structures from the perspective of spectral theory (<xref ref-type="bibr" rid="B4">Bruna et al., 2013</xref>) (<xref ref-type="bibr" rid="B9">Defferrard et al., 2016</xref>). GCNs integrate the connectivity and characteristics of graph-structured data, and it has been demonstrated that GCNs and their variants (<xref ref-type="bibr" rid="B13">Hamilton et al., 2017</xref>; <xref ref-type="bibr" rid="B32">Veli&#x10d;kovi&#x107; et al., 2017</xref>; <xref ref-type="bibr" rid="B8">Dai et al., 2018</xref>; <xref ref-type="bibr" rid="B5">Chen et al., 2017</xref>) significantly outperform Multi-Layer Perceptron (MLP) networks and traditional graph learning methods (<xref ref-type="bibr" rid="B29">Tang et al., 2015</xref>; <xref ref-type="bibr" rid="B24">Perozzi et al., 2014</xref>; <xref ref-type="bibr" rid="B12">Grover and Leskovec, 2016</xref>). To obtain high-level representations and fully utilize the spatial structure characteristics of omics data, we propose a new multi-omics deep clustering algorithm for discovering cancer subtypes, called Self-supervised Multi-fusion Strategy Network (SMMSN). SMMSN utilizes GCNs and SAEs to achieve the fusion of representation and structural information. It introduces various multi-omics data fusion strategies, ultimately achieving clustering through a self-supervised mechanism. This approach ensures efficient integration and utilization of information within and between omics data, leading to more accurate and insightful cancer subtype discovery.</p>
<p>The main contributions of our work are as follows.<list list-type="simple">
<list-item>
<p>(1) Integration of Structured and Representation Information. We introduce a novel method for integrating both structured and representation information within omics data. This approach aims to comprehensively harness and effectively learn the diverse and rich information inherent in multi-omics datasets.</p>
</list-item>
<list-item>
<p>(2) Multi-omics Data Fusion. We present two distinct methods for fusing multi-omics data: error reconstruction fusion and adaptive weighting network fusion. These methods are tailored to different aspects of data representation fusion, offering versatile strategies adapted to specific data characteristics.</p>
</list-item>
<list-item>
<p>(3) Dual Self-supervised Learning. We design a dual self-supervised learning module to perform unsupervised training on fused representations. By leveraging a self-supervised loss function, SMMSN enables the discovery of cancer subtypes from multi-omics fusion data without the need for real labels.</p>
</list-item>
<list-item>
<p>(4) Experimental Validation and Clinical Relevance. Experimental results compared with other algorithms and Kaplan-Meier survival curves demonstrated that SMMSN effectively distinguishes cancer subtypes with significant survival differences. In our analysis of Glioblastoma Multiforme (GBM) and Breast Invasive Carcinoma (BIC), the findings underscored SMMSN&#x2019;s capability to discover clinically relevant cancer subtypes.</p>
</list-item>
</list>
</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<p>The framework of our SMMSN for cancer subtype discovery based on multi-omics (Take DNA methylation data and mRNA expression data, for example,) is shown in <xref ref-type="fig" rid="F1">Figure 1</xref>. SMMSN contains four main modules: A) Date Representation, B) Information Fusion Learning, C) Multi-omics Fusion, D) Dual Self-supervised Learning. The general clustering process of SMMSN is presented as follows.<list list-type="simple">
<list-item>
<p>
<bold>&#x2460; Date Representation Module</bold>. For the <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th omics data <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, a KNN graph <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is constructed to obtain the structure information. At the same time, the feature representation is initialized and taken as input to the SAE network.</p>
</list-item>
<list-item>
<p>
<bold>&#x2461; Information Fusion Learning Module</bold>. Based on the KNN graph <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, a multi-layer GCN model is used to obtain the high-order structure representation <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, which is the output of the <inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> layer in the neural network. At the same time, SAE is used to learn the feature representation <inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> of the omics data by using <inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Then <inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf10">
<mml:math id="m10">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> are combined to obtain a joint representation <inline-formula id="inf11">
<mml:math id="m11">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> that contains both high-level structural information and feature information. The output of the SAE is <inline-formula id="inf12">
<mml:math id="m12">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>l</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, and the output of the GCN is <inline-formula id="inf13">
<mml:math id="m13">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>l</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> which is obtained by <inline-formula id="inf14">
<mml:math id="m14">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>. In this way, the structure information and feature information can be introduced into the deep clustering model through <inline-formula id="inf15">
<mml:math id="m15">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>l</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>
<bold>&#x2462; Multi-omics Fusion Module</bold>. According to the characteristics of different data representations, two data fusion methods are proposed to integrate the information of multiple omics data. For the GCN network output <inline-formula id="inf16">
<mml:math id="m16">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>l</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, an adaptive weighting network is designed to obtain GCN fusion representation <inline-formula id="inf17">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. For the SAE network output <inline-formula id="inf18">
<mml:math id="m18">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>l</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, an error reconstruction method is proposed to obtain SAE fusion representation <inline-formula id="inf19">
<mml:math id="m19">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>
<bold>&#x2463; Dual Self-supervised Learning Module</bold>. A dual self-supervised module is used to jointly learn <inline-formula id="inf20">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf21">
<mml:math id="m21">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> to achieve end-to-end training of the entire model. Firstly, the probability distribution matrix <inline-formula id="inf22">
<mml:math id="m22">
<mml:mrow>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> containing the sample clustering information is calculated according to <inline-formula id="inf23">
<mml:math id="m23">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Through learning high-confidence distribution to make the data representation closer to the cluster center and the target probability distribution matrix <inline-formula id="inf24">
<mml:math id="m24">
<mml:mrow>
<mml:mi mathvariant="bold-italic">P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is obtained. We use the softmax function to perform multi-classification on <inline-formula id="inf25">
<mml:math id="m25">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and obtain the probability distribution matrix <inline-formula id="inf26">
<mml:math id="m26">
<mml:mrow>
<mml:mi mathvariant="bold-italic">G</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Finally, <inline-formula id="inf27">
<mml:math id="m27">
<mml:mrow>
<mml:mi mathvariant="bold-italic">P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is used to perform supervised training on the probability distribution matrices <inline-formula id="inf28">
<mml:math id="m28">
<mml:mrow>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf29">
<mml:math id="m29">
<mml:mrow>
<mml:mi mathvariant="bold-italic">G</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
</list>
</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The framework of our proposed SMMSN model for cancer subtype discovery based on multi-omics (Take DNA methylation data and mRNA expression data, for example,). SMMSN contains four main modules: <bold>(A)</bold> Date Representation, <bold>(B)</bold> Representation Fusion Learning, <bold>(C)</bold> Multi-omics Fusion, <bold>(D)</bold> Dual Self-supervised Learning.</p>
</caption>
<graphic xlink:href="fgene-15-1466825-g001.tif"/>
</fig>
<p>After the iteration is completed, the probability distribution matrix <inline-formula id="inf30">
<mml:math id="m30">
<mml:mrow>
<mml:mi mathvariant="bold-italic">G</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> contains both the feature representation information and structure information of the data. Therefore, the cluster label <inline-formula id="inf31">
<mml:math id="m31">
<mml:mrow>
<mml:mi mathvariant="bold-italic">Y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is calculated according to <inline-formula id="inf32">
<mml:math id="m32">
<mml:mrow>
<mml:mi mathvariant="bold-italic">G</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<sec id="s2-1">
<title>2.1 Data representation module</title>
<p>Given multiple omics datasets <inline-formula id="inf33">
<mml:math id="m33">
<mml:mrow>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi>V</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf34">
<mml:math id="m34">
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the number of datasets, <inline-formula id="inf35">
<mml:math id="m35">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the <inline-formula id="inf36">
<mml:math id="m36">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th omics data in <inline-formula id="inf37">
<mml:math id="m37">
<mml:mrow>
<mml:mi mathvariant="bold-italic">X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf38">
<mml:math id="m38">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents that the <inline-formula id="inf39">
<mml:math id="m39">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th omics data has <inline-formula id="inf40">
<mml:math id="m40">
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> genes (features), and <inline-formula id="inf41">
<mml:math id="m41">
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the number of patients (samples). Prior to implementing our SMMSN model, we carried out several preprocessing steps to address outliers within the multi-omics data. First, any patient with more than 20% missing information in a particular data type was excluded from analysis. Similarly, biological features (such as mRNA expression) with over 20% missing values across all patients were also removed. Additionally, normalization was applied using the following formula:<disp-formula id="e1">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mtext>Var</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>In <xref ref-type="disp-formula" rid="e1">Equation 1</xref> <inline-formula id="inf42">
<mml:math id="m43">
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is any biological feature, <inline-formula id="inf43">
<mml:math id="m44">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the corresponding feature after normalization, <inline-formula id="inf44">
<mml:math id="m45">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf45">
<mml:math id="m46">
<mml:mrow>
<mml:mtext>Var</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represent the mean and variance of <inline-formula id="inf46">
<mml:math id="m47">
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, respectively.</p>
<p>The aim of data representation module is to construct the input of GCNs and SAEs. For GCNs, we use the adjacency matrices constructed from the original data matrices of different omics as input. Since the adjacency matrix represents the relationship information between patient samples, and the number of patients is consistent across all omics data, the input matrix for each omics data in the GCN is of size <inline-formula id="inf47">
<mml:math id="m48">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
<italic>,</italic> where <italic>N</italic> is the number of patients. For SAEs, the input feature dimensions of different omics data can vary, but after being compressed by the encoder, the encoded representations of each type of omics data can be mapped to the same latent space dimension. This means that although the input features of the omics data differ, their output feature dimensions can be aligned through the encoder. In this way, even if the original feature dimensions of different omics data are inconsistent, the autoencoder can compress them into feature representations of the same dimension, allowing these features to be processed consistently in subsequent fusion operations.</p>
<p>Therefore, we take the matrix after the initialization of the omics data as the SAE input, and the <italic>v</italic>th omics data is still represented by <inline-formula id="inf48">
<mml:math id="m49">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. A KNN graph is constructed as the input of GCN based on each omics data. For each sample of each omics data, we select its top-<italic>K</italic> similar samples as neighbors to calculate the similarity between it and each neighbor, and then construct the similarity matrix <inline-formula id="inf49">
<mml:math id="m50">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">S</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. We use the heat kernel method to construct the KNN graph, and the similarity between the two samples <inline-formula id="inf50">
<mml:math id="m51">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf51">
<mml:math id="m52">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> can be written as<disp-formula id="e2">
<mml:math id="m53">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">S</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msubsup>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>j</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>In <xref ref-type="disp-formula" rid="e2">Equation 2</xref> <inline-formula id="inf52">
<mml:math id="m54">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents heat kernel parameter. Then the top-<italic>K</italic> similar samples of each omics data are defined as neighbors to form the adjacency matrix <inline-formula id="inf53">
<mml:math id="m55">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</sec>
<sec id="s2-2">
<title>2.2 Information fusion learning module</title>
<p>This subsection contains three processes: GCN learning, SAE learning and information fusion learning. The whole information fusion learning process of single omics data can be found in <xref ref-type="fig" rid="F2">Figure 2</xref> (Take DNA Methylation data for example).</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Information fusion learning process of single omics data (Take DNA Methylation data, for example,) by combining SAE and GCN model.</p>
</caption>
<graphic xlink:href="fgene-15-1466825-g002.tif"/>
</fig>
<sec id="s2-2-1">
<title>2.2.1 Stacked autoencoder learning</title>
<p>It is critical to learn effective feature representation in clustering tasks. Compared with traditional methods, deep learning methods can extract more advanced data feature representations and are widely used in various fields. In order to extract the high-level feature representation of omics data, we use the Stacked Autoencoder (SAE) model with the strongest generalization performance to learn the original omics data. The training process of SAE model can be found in <xref ref-type="fig" rid="F2">Figure 2</xref> (See SAE Model).</p>
<p>Suppose there are <inline-formula id="inf54">
<mml:math id="m56">
<mml:mrow>
<mml:mi mathvariant="script">l</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> layers in the SAE. In the encoder stage, when SAE is used to learn omics data <inline-formula id="inf55">
<mml:math id="m57">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, the learning of the <inline-formula id="inf56">
<mml:math id="m58">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th layer is written as <inline-formula id="inf57">
<mml:math id="m59">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>
<disp-formula id="e3">
<mml:math id="m60">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3d5;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mmultiscripts>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mprescripts/>
<mml:none/>
<mml:mi mathvariant="bold-italic">e</mml:mi>
</mml:mmultiscripts>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:mmultiscripts>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mi>e</mml:mi>
</mml:mmultiscripts>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>In <xref ref-type="disp-formula" rid="e3">Equation 3</xref> <inline-formula id="inf58">
<mml:math id="m61">
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the activation function of the full connection layer. Here we use the LeakyRELU activation function. <inline-formula id="inf59">
<mml:math id="m62">
<mml:mrow>
<mml:mmultiscripts>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mi>e</mml:mi>
</mml:mmultiscripts>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf60">
<mml:math id="m63">
<mml:mrow>
<mml:mmultiscripts>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mi>e</mml:mi>
</mml:mmultiscripts>
</mml:mrow>
</mml:math>
</inline-formula> are the weight matrix and bias of the <inline-formula id="inf61">
<mml:math id="m64">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th layer in the encoder, respectively. When the encoder starts learning, the feature representation is initialized as: <inline-formula id="inf62">
<mml:math id="m65">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>In the decoder stage, the input data is reconstructed through multiple fully connected layers, which can be written as<disp-formula id="e4">
<mml:math id="m66">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3d5;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mmultiscripts>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mi>d</mml:mi>
</mml:mmultiscripts>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:mmultiscripts>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mi>d</mml:mi>
</mml:mmultiscripts>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>In <xref ref-type="disp-formula" rid="e4">Equation 4</xref> <inline-formula id="inf63">
<mml:math id="m67">
<mml:mrow>
<mml:mmultiscripts>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mi>d</mml:mi>
</mml:mmultiscripts>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf64">
<mml:math id="m68">
<mml:mrow>
<mml:mmultiscripts>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mi>d</mml:mi>
</mml:mmultiscripts>
</mml:mrow>
</mml:math>
</inline-formula> are the parameters of <inline-formula id="inf65">
<mml:math id="m69">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th layer in the decoder.</p>
<p>The final output <inline-formula id="inf66">
<mml:math id="m70">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="script">l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the output <inline-formula id="inf67">
<mml:math id="m71">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>v</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> of SAE: <inline-formula id="inf68">
<mml:math id="m72">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>v</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="script">l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>. We hope that <inline-formula id="inf69">
<mml:math id="m73">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>v</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> can reconstruct the original omics data <inline-formula id="inf70">
<mml:math id="m74">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> as much as possible, and then use the following loss function in <xref ref-type="disp-formula" rid="e5">Equation 5</xref> for SAE model training<disp-formula id="e5">
<mml:math id="m75">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>v</mml:mi>
<mml:mi>V</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msubsup>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>v</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
</sec>
<sec id="s2-2-2">
<title>2.2.2 Graph convolutional network learning</title>
<p>SAE can learn the advanced feature representation of omics data, but it does not consider the structural information among omics data samples. We introduce Graph Convolutional Network (GCN) to learn the structural representation of each omics data. The training process of GCN model can be found in <xref ref-type="fig" rid="F2">Figure 2</xref> (See GCN Model).</p>
<p>For omics data <inline-formula id="inf71">
<mml:math id="m76">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, GCN learns the structural representation <inline-formula id="inf72">
<mml:math id="m77">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> of the <inline-formula id="inf73">
<mml:math id="m78">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th layer through the following convolution operations<disp-formula id="e6">
<mml:math id="m79">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="double-struck">C</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>v</mml:mi>
</mml:msub>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>where <inline-formula id="inf74">
<mml:math id="m80">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the weight matrix of <inline-formula id="inf75">
<mml:math id="m81">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>-th layer. <inline-formula id="inf76">
<mml:math id="m82">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>v</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="bold-italic">I</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf77">
<mml:math id="m83">
<mml:mrow>
<mml:mi mathvariant="bold-italic">I</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is an identity matrix. According to <xref ref-type="disp-formula" rid="e6">Equation 6</xref>, GCN can learn the representation <inline-formula id="inf78">
<mml:math id="m84">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> of the next layer through <inline-formula id="inf79">
<mml:math id="m85">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf80">
<mml:math id="m86">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and the adjacency matrix <inline-formula id="inf81">
<mml:math id="m87">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>v</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</sec>
<sec id="s2-2-3">
<title>2.2.3 Information fusion learning</title>
<p>The information fusion learning process of single omics data by combining SAE and GCN model can be found in <xref ref-type="fig" rid="F2">Figure 2</xref>. Considering both <inline-formula id="inf82">
<mml:math id="m88">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf83">
<mml:math id="m89">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, we can obtain a joint representation <inline-formula id="inf84">
<mml:math id="m90">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> with more effective information through the following formula<disp-formula id="e7">
<mml:math id="m91">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b5;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b5;</mml:mi>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>where <inline-formula id="inf85">
<mml:math id="m92">
<mml:mrow>
<mml:mi>&#x3b5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the balance parameter used to balance the relationship between the two representations <inline-formula id="inf86">
<mml:math id="m93">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf87">
<mml:math id="m94">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>. For simplicity, we set it to 0.5. Through <xref ref-type="disp-formula" rid="e7">Equation 7</xref>, we have realized the connection between SAE and GCN network. And <inline-formula id="inf88">
<mml:math id="m95">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> contains both feature representation information and structure representation information.</p>
<p>Next, we need to learn the <inline-formula id="inf89">
<mml:math id="m96">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th layer representation <inline-formula id="inf90">
<mml:math id="m97">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> of GCN. At this time, <inline-formula id="inf91">
<mml:math id="m98">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is taken as the input of GCN. Then we have<disp-formula id="e8">
<mml:math id="m99">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="double-struck">C</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>v</mml:mi>
</mml:msub>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>
</p>
<p>In the traditional GCN model, after the multi-layer graph convolution operation is adopted, the characteristics of different nodes tend to be homogenized, that is, the characteristics of all nodes within the same connected component are almost the same. This is the so-called over-smoothing phenomenon. The representation information learned by the SAE in each layer is very different, and in <xref ref-type="disp-formula" rid="e8">Equation 8</xref>, the joint representation <inline-formula id="inf92">
<mml:math id="m100">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> contains both the feature information learned and the structured information learned. Therefore, the existence of <xref ref-type="disp-formula" rid="e7">Equation 7</xref> can alleviate the over-smoothing problem of GCN.</p>
<p>It is worth noting that the input data matrix <inline-formula id="inf93">
<mml:math id="m101">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> of the first layer can be calculated by using omics data <inline-formula id="inf94">
<mml:math id="m102">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf95">
<mml:math id="m103">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> can be defined by <xref ref-type="disp-formula" rid="e9">Equation 9</xref>
<disp-formula id="e9">
<mml:math id="m104">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="double-struck">C</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>v</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>
</p>
<p>The final output of GCN is determined according to <xref ref-type="disp-formula" rid="e10">Equation 10</xref>
<disp-formula id="e10">
<mml:math id="m105">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="script">l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="double-struck">C</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>v</mml:mi>
</mml:msub>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="script">l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="script">l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>
</p>
</sec>
</sec>
<sec id="s2-3">
<title>2.3 Multi-omics fusion module</title>
<p>After learning the feature representation and structural representation of any kind of omics data, in order to realize the further clustering task, it is necessary to fuse multi-omics data representations. Based on the different characteristics of omics data representations, we propose two multi-omics data fusion ideas: adaptive weighting network fusion and error reconstruction fusion, to implement Feature Representation Fusion (FRF) and Structural Information Fusion (SIF), respectively. The detailed fusion process can be found in <xref ref-type="fig" rid="F3">Figure 3</xref>.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Graphical illustration of two multi-omics data fusion strategies and dual self-supervised learning.</p>
</caption>
<graphic xlink:href="fgene-15-1466825-g003.tif"/>
</fig>
<p>For the GCN output <inline-formula id="inf96">
<mml:math id="m106">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="script">l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> of each omics data, we connect them in series and propose an adaptive weighting network for fusion to obtain a fusion representation <inline-formula id="inf97">
<mml:math id="m107">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
<disp-formula id="e11">
<mml:math id="m108">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="script">l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2225;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="script">l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mo>&#x2225;</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>&#x2225;</mml:mo>
</mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi>V</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="script">l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>where <inline-formula id="inf98">
<mml:math id="m109">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mrow>
<mml:mi>V</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is a weight matrix that needs to be learned in the fusion process. Since <inline-formula id="inf99">
<mml:math id="m110">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="script">l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> contains the structural information of the omics data, it is necessary to consider the correlation between the samples in the fusion process. Therefore, in <xref ref-type="disp-formula" rid="e11">Equation 11</xref>, we first connect <inline-formula id="inf100">
<mml:math id="m111">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="script">l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> of each omics data to form an overall joint matrix, and then use <inline-formula id="inf101">
<mml:math id="m112">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> to perform weighted learning, so that the adaptive weighting of all samples of all omics data is realized. After obtaining <inline-formula id="inf102">
<mml:math id="m113">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">G</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, we use the softmax function to perform multiple classifications to obtain a probability distribution matrix <inline-formula id="inf103">
<mml:math id="m114">
<mml:mrow>
<mml:mi mathvariant="bold-italic">G</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf104">
<mml:math id="m115">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="bold-italic">G</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the probability that the sample <inline-formula id="inf105">
<mml:math id="m116">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> belongs to category <inline-formula id="inf106">
<mml:math id="m117">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>For the SAE output <inline-formula id="inf107">
<mml:math id="m118">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="script">l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> of each omics data, we propose an error reconstruction fusion method to obtain a fusion representation <inline-formula id="inf108">
<mml:math id="m119">
<mml:mrow>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. First, <inline-formula id="inf109">
<mml:math id="m120">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="script">l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is initialized, and then it is learned according to the following loss function<disp-formula id="e12">
<mml:math id="m121">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>V</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msubsup>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="script">l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
<label>(12)</label>
</disp-formula>
</p>
<p>Following <xref ref-type="disp-formula" rid="e12">Equation 12</xref>, our method can learn the fusion representation with the smallest error of all omics data feature representation through this data reconstruction idea.</p>
</sec>
<sec id="s2-4">
<title>2.4 Dual self-supervised learning module</title>
<p>Traditional SAE and GCN are unsupervised learning and semi-supervised learning algorithms respectively, which cannot be directly applied to clustering problems. In this paper, the dual self-supervised method is used to uniformly train the multi-omics data fusion representation learned by SAE and GCN to realize the clustering task. Graphical illustration of dual self-supervised learning is given in <xref ref-type="fig" rid="F3">Figure 3</xref>.</p>
<p>Firstly, K-means algorithm is adopted to cluster the fusion representation <inline-formula id="inf110">
<mml:math id="m122">
<mml:mrow>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of SAE, and get <inline-formula id="inf111">
<mml:math id="m123">
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> initial cluster centers, where <inline-formula id="inf112">
<mml:math id="m124">
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the number of clusters. For the <inline-formula id="inf113">
<mml:math id="m125">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th sample <inline-formula id="inf114">
<mml:math id="m126">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> (the <inline-formula id="inf115">
<mml:math id="m127">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th row of <inline-formula id="inf116">
<mml:math id="m128">
<mml:mrow>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) and the <inline-formula id="inf117">
<mml:math id="m129">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th cluster center <inline-formula id="inf118">
<mml:math id="m130">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> of <inline-formula id="inf119">
<mml:math id="m131">
<mml:mrow>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, we use the student&#x2019;s <italic>t</italic> distribution in <xref ref-type="disp-formula" rid="e13">Equation 13</xref> (<xref ref-type="bibr" rid="B11">Dunnett and Sobel, 1954</xref>) to measure the similarity between them (<xref ref-type="bibr" rid="B30">Tao et al., 2019</xref>; <xref ref-type="bibr" rid="B36">Wang et al., 2018</xref>)<disp-formula id="e13">
<mml:math id="m132">
<mml:mrow>
<mml:msub>
<mml:mi>q</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>/</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:mfrac>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:msup>
<mml:mi>j</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:munder>
</mml:mstyle>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
<mml:msup>
<mml:mi>j</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>/</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:mfrac>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(13)</label>
</disp-formula>where <inline-formula id="inf120">
<mml:math id="m133">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the degree of freedom of student&#x2019;s <italic>t</italic> distribution, <inline-formula id="inf121">
<mml:math id="m134">
<mml:mrow>
<mml:msub>
<mml:mi>q</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the probability that the <inline-formula id="inf122">
<mml:math id="m135">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th sample is allocated to the <inline-formula id="inf123">
<mml:math id="m136">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th cluster center. The probability distribution matrix of all sample assignments can be denoted as <inline-formula id="inf124">
<mml:math id="m137">
<mml:mrow>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf125">
<mml:math id="m138">
<mml:mrow>
<mml:msub>
<mml:mi>q</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>Then we optimize <inline-formula id="inf126">
<mml:math id="m139">
<mml:mrow>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> by learning high-confidence assignments to make the data representation closer to the cluster center. In <xref ref-type="disp-formula" rid="e14">Equation 14</xref>, the target distribution matrix <inline-formula id="inf127">
<mml:math id="m140">
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="bold-italic">P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> can be obtained according to <inline-formula id="inf128">
<mml:math id="m141">
<mml:mrow>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
<disp-formula id="e14">
<mml:math id="m142">
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mi>q</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>/</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:msup>
<mml:mi>j</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:munder>
</mml:mstyle>
<mml:msubsup>
<mml:mi>q</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>/</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:msup>
<mml:mi>j</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(14)</label>
</disp-formula>where <inline-formula id="inf129">
<mml:math id="m143">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>q</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. In <inline-formula id="inf130">
<mml:math id="m144">
<mml:mrow>
<mml:mi mathvariant="bold-italic">P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, all assignments have higher confidence.</p>
<p>In order to minimize the loss between <inline-formula id="inf131">
<mml:math id="m145">
<mml:mrow>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf132">
<mml:math id="m146">
<mml:mrow>
<mml:mi mathvariant="bold-italic">P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, KL divergence is used as the loss function<disp-formula id="e15">
<mml:math id="m147">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>KL</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="bold-italic">P</mml:mi>
<mml:mo>&#x2225;</mml:mo>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>i</mml:mi>
</mml:munder>
</mml:mstyle>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>j</mml:mi>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>q</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(15)</label>
</disp-formula>
</p>
<p>
<xref ref-type="disp-formula" rid="e15">Equation 15</xref> can make the data representation closer to the cluster center, which is conducive to data clustering. <inline-formula id="inf133">
<mml:math id="m148">
<mml:mrow>
<mml:mi mathvariant="bold-italic">P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is calculated by <inline-formula id="inf134">
<mml:math id="m149">
<mml:mrow>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and the update of <inline-formula id="inf135">
<mml:math id="m150">
<mml:mrow>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> needs to rely on <inline-formula id="inf136">
<mml:math id="m151">
<mml:mrow>
<mml:mi mathvariant="bold-italic">P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Therefore, this is a self-supervised learning mechanism.</p>
<p>We also perform self-supervised learning on the fusion representation of GCN. Since we have obtained the probability distribution matrix <inline-formula id="inf137">
<mml:math id="m152">
<mml:mrow>
<mml:mi mathvariant="bold-italic">G</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of GCN output, we can directly use <inline-formula id="inf138">
<mml:math id="m153">
<mml:mrow>
<mml:mi mathvariant="bold-italic">P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf139">
<mml:math id="m154">
<mml:mrow>
<mml:mi mathvariant="bold-italic">G</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to perform supervised learning. That is<disp-formula id="e16">
<mml:math id="m155">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>KL</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="bold-italic">P</mml:mi>
<mml:mo>&#x2225;</mml:mo>
<mml:mi mathvariant="bold-italic">G</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>i</mml:mi>
</mml:munder>
</mml:mstyle>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>j</mml:mi>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(16)</label>
</disp-formula>
</p>
<p>Through the above-mentioned dual self-supervised learning mechanism, the target distribution <inline-formula id="inf140">
<mml:math id="m156">
<mml:mrow>
<mml:mi mathvariant="bold-italic">P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> conducts supervised learning on <inline-formula id="inf141">
<mml:math id="m157">
<mml:mrow>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf142">
<mml:math id="m158">
<mml:mrow>
<mml:mi mathvariant="bold-italic">G</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> respectively in <xref ref-type="disp-formula" rid="e15">Equations 15</xref>, <xref ref-type="disp-formula" rid="e16">16</xref>, so that the fusion output representations of GCN and SAE are unified under the same optimization framework. After iteration and update, the final training results tend to be consistent.</p>
<p>In conclusion, the overall loss function of the proposed SMMSN framework is defined as <xref ref-type="disp-formula" rid="e17">Equation 17</xref>
<disp-formula id="e17">
<mml:math id="m159">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(17)</label>
</disp-formula>where <inline-formula id="inf143">
<mml:math id="m160">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf144">
<mml:math id="m161">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf145">
<mml:math id="m162">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are hyperparameters used to balance different loss functions.</p>
<p>Since the final output <inline-formula id="inf146">
<mml:math id="m163">
<mml:mrow>
<mml:mi mathvariant="bold-italic">G</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of SMMSN model contains both the representation information and structure information of the data, in <xref ref-type="disp-formula" rid="e18">Equation 18</xref>, we use <inline-formula id="inf147">
<mml:math id="m164">
<mml:mrow>
<mml:mi mathvariant="bold-italic">G</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to achieve clustering. Then the label <inline-formula id="inf148">
<mml:math id="m165">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="bold-italic">Y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of sample <inline-formula id="inf149">
<mml:math id="m166">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> can be calculated by the following formula<disp-formula id="e18">
<mml:math id="m167">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>arg</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:munder>
<mml:mi>max</mml:mi>
<mml:mi>j</mml:mi>
</mml:munder>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(18)</label>
</disp-formula>where <inline-formula id="inf150">
<mml:math id="m168">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="bold-italic">G</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</sec>
</sec>
<sec sec-type="results|discussion" id="s3">
<title>3 Results and discussion</title>
<p>In the experimental phase, we validated the effectiveness of our proposed algorithm using two major categories of real-world cancer multi-omics datasets. First, we conducted experiments on three labeled cancer multi-omics datasets to verify the SMMSN by assessing the accuracy of the clustering results. Secondly, we tested the performance of the SMMSN on five unlabeled cancer multi-omics datasets through survival analysis and validated the biological significance of the cancer subtypes identified by the SMMSN through multidimensional analysis on two cancer cases.</p>
<sec id="s3-1">
<title>3.1 Multi-omics datasets description</title>
<sec id="s3-1-1">
<title>3.1.1 The labeled real-world cancer multi-omics datasets</title>
<p>To demonstrate the effectiveness of SMMSN, we applied it to clustering tasks on three labeled real-world cancer multi-omics datasets. These datasets include the ROSMAP dataset for Alzheimer&#x2019;s disease (AD) patients and normal control (NC) classification, the Low Grade Glioma (LGG) dataset for Grade 2 and Grade 3 classification in low-grade glioma, and the Pan Kidney Cohort (KIPAN) dataset for the classification of three kidney cancer types: Chromophobe Renal Cell Carcinoma (KICH), Clear Renal Cell Carcinoma (KIRC), and Papillary Renal Cell Carcinoma (KIRP) (<xref ref-type="bibr" rid="B37">Wang et al., 2021</xref>). The ROSMAP dataset is composed of ROS and MAP, both of which are longitudinal clinical-pathologic cohort studies of AD from Rush University (<xref ref-type="bibr" rid="B3">Bennett et al., 2012</xref>; <xref ref-type="bibr" rid="B10">De Jager et al., 2018</xref>). It is available through the AMP-AD Knowledge Portal (<ext-link ext-link-type="uri" xlink:href="https://adknowledgeportal.synapse.org/">https://adknowledgeportal.synapse.org/</ext-link>) (<xref ref-type="bibr" rid="B14">Hodes and Buckholtz, 2016</xref>). The omics data for LGG and KIPAN were obtained from TCGA via Broad GDAC Firehose (<ext-link ext-link-type="uri" xlink:href="https://gdac.broadinstitute.org/">https://gdac.broadinstitute.org/</ext-link>). For each dataset, we used three types of omics data (i.e., mRNA expression data, DNA methylation data, and miRNA expression data) for clustering to provide comprehensive and complementary information about the diseases. Only samples with matched omics data were included for each data type. Below are detailed descriptions of the datasets.<list list-type="simple">
<list-item>
<p>&#x2022; ROSMAP: 55889 genes for mRNA expression, 23788 genes for DNA methylation, 309 genes for miRNA expression, 351 patients (NC: 169 patients, AD: 182 patients).</p>
</list-item>
<list-item>
<p>&#x2022; LGG: 20531 genes for mRNA expression, 20114 genes for DNA methylation, 548 genes for miRNA expression, 510 patients (Grade 2: 246 patients, Grade 3: 264 patients).</p>
</list-item>
<list-item>
<p>&#x2022; KIPAN: 20531 genes for mRNA expression, 20111 genes for DNA methylation, 445genes for miRNA expression, 658 patients (KICH: 66 patients, KIRC: 318 patients, KIRP: 274 patients).</p>
</list-item>
</list>
</p>
</sec>
<sec id="s3-1-2">
<title>3.1.2 The unlabeled real-world cancer multi-omics datasets</title>
<p>To further validate the efficacy of SMMSN for cancer subtype discovery, it is used to process multiple omics data sourced from TCGA, as preprocessed by <xref ref-type="bibr" rid="B34">Wang et al. (2014)</xref>. Our study encompassed five distinct cancer types: Breast Invasive Carcinoma (BIC), Glioblastoma Multiforme (GBM), Lung Squamous Cell Carcinoma (LSCC), Kidney Renal Clear Cell Carcinoma (KRCCC), and Colon Adenocarcinoma (COAD). For each cancer type, we analyzed three types of omics data obtained from different platforms: mRNA expression, DNA methylation, and miRNA expression. Detailed descriptions of these multi-omics datasets for the five cancer types are provided below.<list list-type="simple">
<list-item>
<p>&#x2022; GBM: 12,042 genes for mRNA expression, 1,305 genes for DNA methylation, 534 genes for miRNA expression, 213 patients.</p>
</list-item>
<list-item>
<p>&#x2022; BIC: 17,814 genes for mRNA expression, 23,094 genes for DNA methylation, 354 genes for miRNA expression, 105 patients.</p>
</list-item>
<list-item>
<p>&#x2022; KRCCC: 17,899 genes for mRNA expression, 24,960 genes for DNA methylation, 329 genes for miRNA expression, 122 patients.</p>
</list-item>
<list-item>
<p>&#x2022; LSCC: 12,042 genes for mRNA expression, 23,074 genes for DNA methylation, 352 genes for miRNA expression, 106 patients.</p>
</list-item>
<list-item>
<p>&#x2022; COAD: 17,814 genes for mRNA expression, 23,088 genes for DNA methylation, 312 genes for miRNA expression, 92 patients.</p>
</list-item>
</list>
</p>
</sec>
</sec>
<sec id="s3-2">
<title>3.2 Experiment settings</title>
<sec id="s3-2-1">
<title>3.2.1 Evaluation indicator</title>
<p>For labeled cancer multi-omics data, we used the Accuracy (ACC) for evaluation to validate the clustering results. ACC quantifies the consistency between the clustering results and the true labels, and its calculation formula is defined as <xref ref-type="disp-formula" rid="e19">Equation 19</xref>:<disp-formula id="e19">
<mml:math id="m169">
<mml:mrow>
<mml:mtext>ACC</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mi>&#x3b4;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mtext>map</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(19)</label>
</disp-formula>where <inline-formula id="inf151">
<mml:math id="m170">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the true label, <inline-formula id="inf152">
<mml:math id="m171">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the label assigned by the clustering methods, <inline-formula id="inf153">
<mml:math id="m172">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the indicator function, which equals 1 when <inline-formula id="inf154">
<mml:math id="m173">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>map</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, and 0 otherwise. <inline-formula id="inf155">
<mml:math id="m174">
<mml:mrow>
<mml:mtext>map</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents an optimal mapping function that best matches the clustering labels to the true labels. By calculating ACC, we can intuitively evaluate the clustering performance.</p>
<p>For unlabeled datasets, this study performs survival analysis on cancer subtypes identified through clustering to assess survival disparities among sample groups derived from the proposed algorithm. In statistical analysis, hypothesis testing, such as the Cox Log-rank Test (CLT) (<xref ref-type="bibr" rid="B15">Hosmer et al., 2000</xref>), is employed to quantify differences in survival curves. CLT is a non-parametric method commonly used to evaluate whether variations in survival between subtypes are significant. A lower <italic>p</italic>-value from this test suggests stronger evidence against the null hypothesis, indicating substantial differences in survival outcomes that are unlikely to be due to chance alone. Additionally, the Kaplan-Meier estimation method (<xref ref-type="bibr" rid="B15">Hosmer et al., 2000</xref>) is utilized to derive survival functions and construct Kaplan-Meier survival curves. These curves plot the survival rate on the <italic>y</italic>-axis against time from the start of observation to the last recorded time point on the <italic>x</italic>-axis. They visually illustrate how the event (e.g., survival or recurrence) unfolds over time for different cancer subtypes, providing insight into their respective prognostic outcomes.</p>
</sec>
<sec id="s3-2-2">
<title>3.2.2 Comparison methods</title>
<p>For comparison purposes, we included five established traditional multi-view clustering algorithms known for their efficacy in cancer subtype prediction: PFA (<xref ref-type="bibr" rid="B27">Shi et al., 2017</xref>), SNF (<xref ref-type="bibr" rid="B34">Wang et al., 2014</xref>), ANF (<xref ref-type="bibr" rid="B20">Ma and Zhang, 2017</xref>), and MVCMO (<xref ref-type="bibr" rid="B43">Yu et al., 2019</xref>). Two deep learning-based cancer subtypes discovering methods, Subtype-Former (<xref ref-type="bibr" rid="B42">Yang et al., 2022</xref>) and Subtype-DCC (<xref ref-type="bibr" rid="B46">Zhao et al., 2023</xref>), are also taken as the competing methods. These algorithms are widely recognized in the field for their ability to integrate diverse data sources and identify meaningful subtypes within cancer datasets.</p>
</sec>
<sec id="s3-2-3">
<title>3.2.3 Experimental parameter settings</title>
<p>The deep learning algorithms involved in this study were implemented using the popular deep learning framework PyTorch 3.9, and all experiments were conducted on an NVIDIA GeForce RTX 4080 GPU with 32&#xa0;GB RAM, Core I7-12700K. To evaluate the performance of the models and comparison methods, each experiment was run five times, and the average accuracy score along with the standard deviation was reported to ensure the robustness and comparability of the results. The parameter settings for the deep learning models are as follows.<list list-type="simple">
<list-item>
<p>&#x2022; For the SMMSN algorithm, the network output dimension was set to 100, and the adjustable adjacency matrix parameter <italic>k</italic> was defined as 40. The Adam optimizer was used during training, with an initial learning rate of 1 &#xd7; 10&#x207b;&#x2074; and a decay factor of 1 &#xd7; 10&#x207b;<sup>1</sup>&#x2075;. The model was trained for 500 epochs.</p>
</list-item>
<list-item>
<p>&#x2022; For the Subtype-DCC algorithm, the feature dimension was set to 256, the batch size to 64, and the number of training epochs to 600. The Adam optimizer was used with automatic learning rate adjustment, starting with an initial learning rate of 1.95 &#xd7; 10&#x207b;&#x2074;. The instance-level and cluster-level temperature parameters were set to 0.5 and 1.0, respectively.</p>
</list-item>
<list-item>
<p>&#x2022; For the Subtype-Former algorithm, the Adam optimizer was also used, with an initial learning rate of 7 &#xd7; 10&#x207b;&#x2074;, a batch size of 8, and the model achieved optimal performance after 45 epochs of training.</p>
</list-item>
</list>
</p>
<p>For the benchmark machine learning algorithms, they were implemented by MATLAB 2022a software, and their parameters were set strictly according to the guidelines provided by the authors. Each experiment was run five times, and the average accuracy score along with the standard deviation was reported to ensure the robustness and comparability of the results The specific parameters are set as follows.<list list-type="simple">
<list-item>
<p>&#x2022; For the PFA algorithm, the local sample-spectrum for each biological data type was captured using the <italic>Algorithm_1</italic> function from the <italic>PFA package.</italic> Next, the global sample-spectrum was captured using the <italic>Algorithm_4</italic> function, with the hyperparameter lambda set to 1.</p>
</list-item>
<list-item>
<p>&#x2022; For the SNF algorithm, an affinity matrix for each omics dataset was calculated using the <italic>dist2</italic> and <italic>affinityMatrix</italic> functions from the <italic>SNFtool</italic> package. The number of neighbors was set to 1/10 of the total number of samples, and the sigma parameter was set to 0.5. These affinity matrices were then integrated using the SNF method with the same number of neighbors and 30 iterations for the multi-omics data. Spectral clustering was performed on the integrated matrix with default parameters.</p>
</list-item>
<list-item>
<p>&#x2022; For the ANF algorithm, an affinity matrix for each omics dataset was calculated using the <italic>affinity_matrix</italic> function from the <italic>ANFtool</italic> package, with the number of neighbors set to 1/10 of the total number of samples. The matrices were integrated using the ANF method with the same number of neighbors for the multi-omics data.</p>
</list-item>
<list-item>
<p>&#x2022; For the MVCMO algorithm, an affinity matrix for each omics dataset was calculated using the <italic>knnAffinity</italic> function from the <italic>MVCMO</italic> package, with the number of neighbors set to 5. A fused low-dimensional matrix was then generated using the <italic>adaptedweight</italic> function, with beta set to 1.</p>
</list-item>
</list>
</p>
</sec>
<sec id="s3-2-4">
<title>3.2.4 Settings of cluster number</title>
<p>For the labeled cancer multi-omics data, the number of clusters corresponds to the number of cancer subtypes in the data itself. The number of clusters for the three datasets, KIPAN, ROSMAP, and LGG, is set to 3, 2, and 2, respectively. For the unlabeled data, we follow the commonly accepted number of cancer subtypes as reported in the majority of studies, such as in references (<xref ref-type="bibr" rid="B34">Wang et al., 2014</xref>; <xref ref-type="bibr" rid="B20">Ma and Zhang, 2017</xref>; <xref ref-type="bibr" rid="B43">Yu et al., 2019</xref>). The number of clusters for the five datasets, GBM, BIC, KRCCC, LSCC, and COAD, is set to 3, 5, 3, 4, and 3, respectively.</p>
</sec>
</sec>
<sec id="s3-3">
<title>3.3 Results on labeled multi-omics datasets</title>
<p>
<xref ref-type="table" rid="T1">Table 1</xref> presents the clustering accuracy of the SMMSN algorithm and competing methods on several labeled cancer multi-omics datasets. In the KIPAN dataset, SMMSN achieved a clustering accuracy of 85.34%, outperforming all other competing methods, especially the two other deep learning models. In comparison, SMMSN&#x2019;s accuracy was about 3 percentage points higher than the suboptimal method, SNF. This demonstrates SMMSN&#x2019;s superior ability to distinguish between different types of kidney cancer. In the ROSMAP dataset, SMMSN also exhibited high accuracy, reaching 68.83%, outperforming both classical machine learning models and deep learning models. In the LGG dataset, SMMSN achieved a clustering accuracy of 65.80%. Although DCC (68.39%) performed slightly better than SMMSN, SMMSN still outperformed most of the other methods and maintained stable performance across multiple datasets.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>The clustering accuracy (%) of SMMSN and competing methods on several real and labeled cancer multi-omics datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Datasets</th>
<th colspan="4" align="center">Methods</th>
</tr>
<tr>
<th align="center">SMMSN</th>
<th align="center">PFA</th>
<th align="center">SNF</th>
<th align="center">ANF</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">KIPAN</td>
<td align="center">
<bold>85.34 &#xb1; 3.41</bold>
</td>
<td align="center">75.81 &#xb1; 3.52</td>
<td align="center">82.27 &#xb1; 0.00</td>
<td align="center">81.18 &#xb1; 0.00</td>
</tr>
<tr>
<td align="center">ROSMAP</td>
<td align="center">
<bold>68.83 &#xb1; 0.71</bold>
</td>
<td align="center">61.22 &#xb1; 2.98</td>
<td align="center">66.32 &#xb1; 0.00</td>
<td align="center">62.64 &#xb1; 0.00</td>
</tr>
<tr>
<td align="center">LGG</td>
<td align="center">65.80 &#xb1; 0.40</td>
<td align="center">60.48 &#xb1; 3.85</td>
<td align="center">62.96 &#xb1; 0.00</td>
<td align="center">63.41 &#xb1; 0.00</td>
</tr>
</tbody>
</table>
<table>
<thead valign="top">
<tr>
<th align="center">Continued</th>
<th align="center">MVSCO</th>
<th align="center">Former</th>
<th align="center">DCC</th>
<th align="left"/>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">KIPAN</td>
<td align="center">79.45 &#xb1; 1.55</td>
<td align="center">79.86 &#xb1; 0.76</td>
<td align="center">78.66 &#xb1; 0.07</td>
<td align="left"/>
</tr>
<tr>
<td align="center">ROSMAP</td>
<td align="center">65.93 &#xb1; 2.44</td>
<td align="center">65.01 &#xb1; 4.48</td>
<td align="center">64.10 &#xb1; 3.80</td>
<td align="left"/>
</tr>
<tr>
<td align="center">LGG</td>
<td align="center">62.32 &#xb1; 2.67</td>
<td align="center">64.00 &#xb1; 0.40</td>
<td align="center">
<bold>68.39 &#xb1; 2.55</bold>
</td>
<td align="left"/>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Here Subtype-Former and Subtype-DCC, methods are referred to as Former and DCC, respectively. The best results have been highlighted in bold.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The results indicate that SMMSN consistently outperformed traditional methods such as PFA, SNF, ANF, and MVSCO on all datasets, with particularly noticeable improvements in the KIPAN and ROSMAP datasets. This suggests that SMMSN, by leveraging deep learning&#x2019;s representation capabilities, better captures the complex nonlinear relationships in multi-omics data and effectively integrates various omics types to improve clustering performance, which is more challenging for traditional algorithms. Compared to other deep learning models, DCC and Former, SMMSN showed significant advantages in the KIPAN and ROSMAP datasets. Although DCC performed slightly better in the LGG dataset, SMMSN demonstrated greater robustness across multiple datasets, with lower standard deviations, indicating more stable performance. SMMSN&#x2019;s high clustering accuracy highlights its unique advantage in integrating multi-omics data and effectively capturing complementary information between different omics types for cancer subtype classification tasks.</p>
</sec>
<sec id="s3-4">
<title>3.4 Results on unlabeled multi-omics datasets</title>
<sec id="s3-4-1">
<title>3.4.1 Survival analysis on unlabeled multi-omics datasets</title>
<p>
<xref ref-type="table" rid="T2">Table 2</xref> shows the <italic>p</italic>-values from survival analysis between SMMSN and competing methods across five datasets. This comparison evaluates the statistical significance of survival differences among cancer subtypes identified by each algorithm. Across all five cancer types, SMMSN consistently yielded the lowest <italic>p</italic>-values compared to other algorithms. <xref ref-type="fig" rid="F4">Figure 4</xref> presents Kaplan-Meier survival curves generated by SMMSN for each cancer type, depicting the survival trends of respective subtypes. Each curve in <xref ref-type="fig" rid="F4">Figure 4</xref> illustrates the survival times of distinct cancer subtypes, with sample counts annotated for clarity. These results demonstrate SMMSN&#x2019;s ability to discern significantly distinct cancer subtypes across various cancer types.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>The <italic>p</italic>-values from survival analysis between SMMSN and competing methods across five datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Cancer<break/>Types</th>
<th colspan="7" align="center">Methods</th>
</tr>
<tr>
<th align="center">SMMSN</th>
<th align="center">PFA</th>
<th align="center">SNF</th>
<th align="center">ANF</th>
<th align="center">MVCMO</th>
<th align="center">Former</th>
<th align="center">DCC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">GBM</td>
<td align="center">
<bold>3.39E-5</bold>
</td>
<td align="center">2.15E-4</td>
<td align="center">4.24E-5</td>
<td align="center">4.68E-4</td>
<td align="center">2.14E-3</td>
<td align="center">7.51E-5</td>
<td align="center">2.62E-4</td>
</tr>
<tr>
<td align="center">BIC</td>
<td align="center">
<bold>7.05E-5</bold>
</td>
<td align="center">2.85E-4</td>
<td align="center">7.63E-4</td>
<td align="center">2.65E-4</td>
<td align="center">2.98E-4</td>
<td align="center">1.25E-4</td>
<td align="center">4.63E-4</td>
</tr>
<tr>
<td align="center">KRCCC</td>
<td align="center">
<bold>6.02E-3</bold>
</td>
<td align="center">6.89E-2</td>
<td align="center">3.04E-2</td>
<td align="center">5.17E-2</td>
<td align="center">2.14E-2</td>
<td align="center">1.65E-2</td>
<td align="center">2.67E-2</td>
</tr>
<tr>
<td align="center">LSCC</td>
<td align="center">
<bold>1.21E-3</bold>
</td>
<td align="center">2.04E-2</td>
<td align="center">1.23E-2</td>
<td align="center">9.05E-3</td>
<td align="center">8.97E-3</td>
<td align="center">3.54E-3</td>
<td align="center">2.53E-2</td>
</tr>
<tr>
<td align="center">COAD</td>
<td align="center">
<bold>5.21E-4</bold>
</td>
<td align="center">7.25E-2</td>
<td align="center">3.17E-3</td>
<td align="center">8.78E-3</td>
<td align="center">7.94E-3</td>
<td align="center">2.35E-3</td>
<td align="center">1.58E-3</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Here Subtype-Former and Subtype-DCC, methods are referred to as Former and DCC, respectively. The best results have been highlighted in bold.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Kaplan-Meier survival curves of discovered subtypes by SMMSN on five datasets. <bold>(A)</bold> GBM <bold>(B)</bold> BIC <bold>(C)</bold> KRCCC <bold>(D)</bold> LSCC <bold>(E)</bold> COAD.</p>
</caption>
<graphic xlink:href="fgene-15-1466825-g004.tif"/>
</fig>
<p>To further validate the effectiveness of each module in SMMSN, we conducted ablation studies. The SMMSN algorithm mainly consists of three key components: the Feature Representation Fusion (FRF) module based on SAE, the Structural Information Fusion (SIF) module based on GCN, and the Dual Self-supervised (DSS) module. The results of the ablation study are shown in <xref ref-type="table" rid="T3">Table 3</xref>. It is important to note that when we use only the FRF module or the SIF module, only a single self-supervised learning operation is required, which is denoted as SS module in <xref ref-type="table" rid="T3">Table 3</xref>. In other words, when both the FRF and SIF modules are used simultaneously in SMMSN, we apply the dual self-supervised module for model learning. From the ablation results shown in <xref ref-type="table" rid="T3">Table 3</xref>, we can observe different performance outcomes for three different module combinations across five unlabeled multi-omics datasets (GBM, BIC, KRCCC, LSCC, COAD). The analysis can be broken down as follows.<list list-type="simple">
<list-item>
<p>&#x2022; When only the GCN module and single self-supervised module are used, the results are relatively poor across all datasets, particularly on the KRCCC and LSCC datasets, with p-values of 9.60E-2 and 2.26E-2, respectively. This suggests that while the GCN module can capture structural features, its performance is limited without the feature representation fusion from the SAE module.</p>
</list-item>
<list-item>
<p>&#x2022; When only the SAE module and single self-supervised module are used, the results are significantly better than the combination with GCN alone, especially on the GBM, KRCCC, and COAD datasets. For example, the error for the KRCCC dataset decreases from 9.60E-2 to 1.79E-2, and for COAD, it reduces from 5.15E-3 to 1.02E-3. This indicates that the feature representation fusion from the SAE module is more effective in capturing the fused characteristics of multi-omics data than structural features alone.</p>
</list-item>
<list-item>
<p>&#x2022; When SAE, GCN, and the dual self-supervised module are all used together, the errors across all datasets reach their lowest values. For instance, the <italic>p</italic>-value on the KRCCC dataset is further reduced from 1.79E-2 to 6.02E-3, and on LSCC from 9.97E-3 to 1.21E-3, demonstrating that the dual self-supervised module leverages the strengths of both SAE and GCN, greatly enhancing the model&#x2019;s performance.</p>
</list-item>
</list>
</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Ablation results (<italic>p</italic>-values) of SMMSN on five unlabeled multi-omics datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th colspan="4" align="center">Components</th>
<th colspan="5" align="center">Cancer types</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">SAE</td>
<td align="center">GCN</td>
<td align="center">SS</td>
<td align="center">DSS</td>
<td align="center">GBM</td>
<td align="center">BIC</td>
<td align="center">KRCCC</td>
<td align="center">LSCC</td>
<td align="center">COAD</td>
</tr>
<tr>
<td align="center">--</td>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">--</td>
<td align="center">5.77E-4</td>
<td align="center">6.18E-4</td>
<td align="center">9.60E-2</td>
<td align="center">2.26E-2</td>
<td align="center">5.15E-3</td>
</tr>
<tr>
<td align="center">&#x221a;</td>
<td align="center">--</td>
<td align="center">&#x221a;</td>
<td align="center">--</td>
<td align="center">2.16E-4</td>
<td align="center">2.58E-3</td>
<td align="center">1.79E-2</td>
<td align="center">9.97E-3</td>
<td align="center">1.02E-3</td>
</tr>
<tr>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">--</td>
<td align="center">&#x221a;</td>
<td align="center">3.39E-5</td>
<td align="center">7.05E-5</td>
<td align="center">6.02E-3</td>
<td align="center">1.21E-3</td>
<td align="center">5.21E-4</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-4-2">
<title>3.4.2 GBM case analysis</title>
<p>GBM stands as the most prevalent and deadly primary brain tumor in adults, categorized within the glioma group. Numerous studies have extensively explored GBM at the molecular level, identifying distinct cancer subtypes with corresponding clinical implications. For instance, <xref ref-type="bibr" rid="B33">Verhaak et al. (2010)</xref> classified GBM based on mRNA expression into four subtypes: Mesenchymal, Classical, Neural, and Proneural. Another study (<xref ref-type="bibr" rid="B22">Noushmehr et al., 2010</xref>) differentiated GBM into G-CIMP and non-G-CIMP subtypes based on CpG Island Methylator Phenotype (CIMP).</p>
<p>Using GBM data, we analyzed the distribution of clustering results obtained by SMMSN across the subtypes identified in the aforementioned studies, summarized in <xref ref-type="table" rid="T4">Table 4</xref>. It is worth noting that the cancer subtypes in references (<xref ref-type="bibr" rid="B33">Verhaak et al., 2010</xref>) and (<xref ref-type="bibr" rid="B22">Noushmehr et al., 2010</xref>) are classifications derived from different research methods and standards, but they are not considered gold standards for cancer subtypes. Instead, they serve as reference classifications used to help understand and validate the biological differences between the three subtypes identified by the SMMSN algorithm. <xref ref-type="table" rid="T4">Table 4</xref> highlights that a majority of patients in subtype 1 align with the Proneural subtype. Subtype 2 shows a closer association with Classical and Proneural subtypes. Subtype 3 predominantly corresponds to the Mesenchymal subtype. It shows that the three subtypes identified have certain differences. Notably, all patients in subtypes 2 and 3 belong to the non-G-CIMP category, while a portion of patients in subtype 1 are classified under G-CIMP. This indicates The difference between the identified subtype 1 and subtype 2&#x2013;3 (subtype 2 and subtype 3) was obvious, and this conclusion was also verified in <xref ref-type="fig" rid="F4">Figure 4A</xref> that subtype 1 had a longer survival time than subtype 2&#x2013;3.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>The distribution of subtypes identified by SMMSN in relation to those defined in <xref ref-type="bibr" rid="B33">Verhaak et al. (2010)</xref> and <xref ref-type="bibr" rid="B22">Noushmehr et al. (2010)</xref>.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">SMMSN subtypes</th>
<th colspan="3" align="center">Subtypes in <xref ref-type="bibr" rid="B33">Verhaak et al. (2010)</xref>
</th>
<th colspan="3" align="center">Subtypes in <xref ref-type="bibr" rid="B22">Noushmehr et al. (2010)</xref>
</th>
</tr>
<tr>
<th align="center">Mesenchymal</th>
<th align="center">Classical</th>
<th align="center">Neural</th>
<th align="center">Proneural</th>
<th align="center">G-CIMP</th>
<th align="center">Non-G-CIMP</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Subtype 1</td>
<td align="center">6</td>
<td align="center">9</td>
<td align="center">4</td>
<td align="center">26</td>
<td align="center">19</td>
<td align="center">26</td>
</tr>
<tr>
<td align="center">Subtype 2</td>
<td align="center">17</td>
<td align="center">35</td>
<td align="center">16</td>
<td align="center">26</td>
<td align="center">0</td>
<td align="center">94</td>
</tr>
<tr>
<td align="center">Subtype 3</td>
<td align="center">42</td>
<td align="center">14</td>
<td align="center">14</td>
<td align="center">4</td>
<td align="center">0</td>
<td align="center">74</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The values shown in the table represent the count of patients in each subtype identified by SMMSN, with some association and difference with the classification established by <xref ref-type="bibr" rid="B33">Verhaak et al. (2010)</xref> and <xref ref-type="bibr" rid="B22">Noushmehr et al. (2010)</xref>.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Subsequently, we further compared long-term survival subtype 1 with short-term survival subtype 2&#x2013;3 and looked for their differences in gene mutations. <xref ref-type="fig" rid="F5">Figure 5</xref> show the difference in Copy Number Variation (CNV) abundance between long-lived subtype 1 and short-lived subtype 2&#x2013;3. In this figure, each point represents a gene, and its axis is the number of patients carrying a variant of that gene in the two survival differential subtypes. The most abundant mutated genes in subtypes 2&#x2013;3 were EGFR, SEC61G and RP11-745C15.2. EGFR mutations and amplifications are very common in GBM, especially the EGFRvIII variant, which drives rapid tumor cell proliferation, increased invasiveness, and resistance to treatment. EGFR overexpression is closely associated with the progression of more malignant subtypes, which generally indicate poorer survival outcomes (<xref ref-type="bibr" rid="B16">Hu et al., 2022</xref>). As a key component of the SEC61 translocation complex in the endoplasmic reticulum, SEC61G is involved in regulating protein transport and processing. Abnormalities in SEC61G may affect proteins involved in cell proliferation and stress responses, thus promoting tumor growth and progression (<xref ref-type="bibr" rid="B44">Zeng et al., 2023</xref>). RP11-745C15.2 represents a class of long non-coding RNAs (lncRNAs), whose role in cancer is becoming increasingly recognized. RP11-745C15.2 may regulate key oncogenes like EGFR or its downstream signaling pathways, enhancing malignant cell behavior. In GBM, several lncRNAs, including RP11 family members, are believed to be involved in tumor progression by regulating gene expression, influencing cell growth, and contributing to the aggressive nature of the tumor (<xref ref-type="bibr" rid="B45">Zhang et al., 2020</xref>). To sum up, EGFR mutations drive malignant cell proliferation, while SEC61G and RP11-745C15.2, through their roles in protein transport and gene regulation, further promote tumor cell growth, survival, and invasiveness. Their combined action leads to greater gene variation in subtypes 2 and 3, resulting in higher aggressiveness and worse prognosis. This genetic association suggests that these gene alterations are key drivers of tumor progression in the short-term survival subtypes, helping to distinguish subtype 1 (long-term survival) from subtypes 2 and 3 (short-lived).</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Differences in Copy Number Variation (CNV) abundance of the identified GBM subtypes. Each point represents a gene, and the horizontal and vertical axes show the number of patients carrying a variant of that gene in long-term survival subtype 1 and short-term survival subtype 2&#x2013;3, respectively.</p>
</caption>
<graphic xlink:href="fgene-15-1466825-g005.tif"/>
</fig>
<p>Further analysis of the cancer subtypes identified by SMMSN involved accessing clinical data for all GBM patients from the cBio Cancer Genomics Portal database. <xref ref-type="fig" rid="F6">Figure 6</xref> illustrates boxplots depicting the distribution of survival time and age among these subtypes, demonstrating discernible differences. In <xref ref-type="fig" rid="F6">Figure 6A</xref>, subtype 1 exhibits significantly longer survival compared to subtype 2 and subtype 3, supported by <italic>p</italic>-values from two-sided Welch&#x2019;s t-tests: 1.45E-4 and 5.69E-3, respectively. <xref ref-type="fig" rid="F6">Figure 6B</xref> reveals that patients in subtype 1 are younger than those in subtype 2 and subtype 3, with corresponding <italic>t</italic>-test <italic>p</italic>-values of 2.37E-7 and 5.21E-5, respectively. Moreover, we conducted an Analysis of Variance (ANOVA) test across the three subtypes, confirming significant differences in both survival time (<italic>p</italic> &#x3d; 8.24E-7) and age distribution (<italic>p</italic> &#x3d; 2.48E-9). Similarly, the Kruskal&#x2013;Wallis test also indicated statistically significant distinctions in age (<italic>p</italic> &#x3d; 1.97E-7) and survival time (<italic>p</italic> &#x3d; 1.84E-04) among the subtypes. These consistent findings underscore the biological relevance and statistical significance of the identified subtypes with respect to both age demographics and survival outcomes.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Boxplots used to visualize the distribution of survival time and age among patients classified into the three identified cancer subtypes. <bold>(A)</bold> Displays the variation in survival times across these subtypes, highlighting significant differences. <bold>(B)</bold> Displays the age distributions of patients in each subtype are compared, revealing notable variations among the groups.</p>
</caption>
<graphic xlink:href="fgene-15-1466825-g006.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="F7">Figure 7</xref> presents Kaplan-Meier survival curves depicting the response of patients to the drug Temozolomide (TMZ). Patients are stratified into two groups: those treated with TMZ and those not treated with TMZ. The <italic>p</italic>-values for subtype 1, subtype 2, and subtype 3 are 0.65, 4.12E-5, and 5.42E-2, respectively. These results indicate that TMZ treatment has minimal impact on the survival outcomes of patients in subtype 1, while it significantly affects the survival of patients in subtypes 2 and 3.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Here are the Kaplan-Meier survival curves depicting the response to Temozolomide (TMZ) for the identified cancer subtypes by SMMSN: <bold>(A)</bold> Kaplan-Meier survival curve for Sub-type 1 in response to TMZ. <bold>(B)</bold> Kaplan-Meier survival curve for Subtype 2 in response to TMZ. <bold>(C)</bold> Kaplan-Meier survival curve for Subtype 3 in response to TMZ.</p>
</caption>
<graphic xlink:href="fgene-15-1466825-g007.tif"/>
</fig>
<p>Differential gene expression and GO enrichment analyses were conducted on GBM data to assess differences among the three subtypes identified by SMMSN. Initially, significant differentially expressed genes across the subtypes were identified using the ANOVA method. <xref ref-type="fig" rid="F8">Figure 8</xref> displays a heatmap illustrating the top 1,000 differentially expressed genes in mRNA expression data, with panels A, B, and C representing subtype 1, subtype 2, and subtype 3, respectively. The heatmap reveals distinct clusters among the differentially expressed genes, indicating subtype-specific expression patterns.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>The heatmap showcases the top 1,000 genes whose mRNA expression varies significantly across the three subtypes identified by SMMSN in GBM. Subtype 1, Subtype 2, and Subtype 3 are represented by different colors respectively, highlighting distinct clusters of gene expression patterns among the subtypes.</p>
</caption>
<graphic xlink:href="fgene-15-1466825-g008.tif"/>
</fig>
<p>Further analysis involved functional enrichment of these differentially expressed genes. <xref ref-type="fig" rid="F9">Figure 9</xref> presents the results of GO enrichment analysis categorizing the genes into four distinct groups (X1, X2, X3, and X4). Each group is associated with specific GO biological processes, as indicated by the number of enriched genes listed below. Notably, genes related to the regulation of mRNA metabolism/chromosome organization were downregulated in subtype 3 and upregulated in subtype 2. The genes related to the regulation of immune effector process/leukocyte-mediated immunity/lymphocyte-mediated immunity and hetero-cell bonding were downregulated in subtype 2. Genes associated with DNA recombination, nuclear transport/export, and RNA splicing were downregulated in subtype 3. In conclusion, GBM subtypes identified by SMMSN have obvious differences in clinical indicators and molecular levels.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Functional enrichment analysis was performed on the differentially expressed genes identified in <xref ref-type="fig" rid="F8">Figure 8</xref>.</p>
</caption>
<graphic xlink:href="fgene-15-1466825-g009.tif"/>
</fig>
</sec>
<sec id="s3-4-3">
<title>3.4.3 BIC case analysis</title>
<p>BIC refers to a malignancy in which cancer cells have penetrated the basement membrane of breast ducts or lobular acinus and invaded the stroma. Similar to the GBM case analysis procedure described above, we first made a comparison with previous BIC subtype study. PAM50 is a molecular subtype of BIC based on quantitative detection of the expression levels of 50 functional genes in breast tumor tissues, including Luminal A, Luminal B, HER2-enriched and Basal-like subtypes (<xref ref-type="bibr" rid="B23">Parker et al., 2009</xref>). <xref ref-type="table" rid="T5">Table 5</xref> depicts the comparison results of the distribution of BIC subtypes identified by SMMSN in PAM50 subtypes.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>The distribution of subtypes identified by SMMSN in PAM50 subtypes.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">SMMSN subtypes</th>
<th colspan="4" align="center">Subtypes in <xref ref-type="bibr" rid="B23">Parker et al. (2009)</xref>
</th>
</tr>
<tr>
<th align="center">Luminal A</th>
<th align="center">Luminal B</th>
<th align="center">Basal-like</th>
<th align="center">HER2-enriched</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Subtype 1</td>
<td align="center">11</td>
<td align="center">2</td>
<td align="center">6</td>
<td align="center">4</td>
</tr>
<tr>
<td align="center">Subtype 2</td>
<td align="center">14</td>
<td align="center">3</td>
<td align="center">0</td>
<td align="center">0</td>
</tr>
<tr>
<td align="center">Subtype 3</td>
<td align="center">13</td>
<td align="center">0</td>
<td align="center">1</td>
<td align="center">4</td>
</tr>
<tr>
<td align="center">Subtype 4</td>
<td align="center">13</td>
<td align="center">7</td>
<td align="center">0</td>
<td align="center">3</td>
</tr>
<tr>
<td align="center">Subtype 5</td>
<td align="center">2</td>
<td align="center">0</td>
<td align="center">16</td>
<td align="center">0</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>From <xref ref-type="table" rid="T5">Table 5</xref>, it can be observed that the five cancer subtypes identified by the SMMSN algorithm show different distribution patterns in the PAM50 subtypes (Luminal A, Luminal B, Basal-like, HER2-enriched) from the study in (<xref ref-type="bibr" rid="B23">Parker et al., 2009</xref>). SMMSN subtypes 1, 2, 3, and 4 are mainly concentrated in Luminal A, with subtype 2 almost entirely composed of Luminal A patients, indicating a high level of consistency between these subtypes and the Luminal A subtype. SMMSN subtype 5, on the other hand, is predominantly composed of Basal-like patients, suggesting a strong correspondence with the Basal-like subtype. In contrast, Luminal B and HER2-enriched patients are more dispersed across multiple SMMSN subtypes, especially in subtypes 1 and 4, revealing a certain degree of discrepancy between the subtypes identified by SMMSN and the PAM50 subtypes. These observations reflect a strong alignment between SMMSN subtypes and PAM50 in certain subtypes, while in others, cross-subtype distribution patterns are apparent.</p>
<p>To further validate the biological differences among the cancer subtypes 1, 2, 3, 4, and 5 identified by SMMSN, especially the first four subtypes, differential expression analysis of BIC expression data was performed using the Kruskal&#x2013;Wallis test. Hierarchical clustering method was used to group the top 1,000 differential genes of BIC, and GO: BP analysis was performed based on the grouping results, and their results are presented in <xref ref-type="fig" rid="F10">Figures 10</xref>, <xref ref-type="fig" rid="F11">11</xref>. <xref ref-type="fig" rid="F10">Figure 10</xref> presents the heatmap of the top 1,000 genes whose mRNA expression varies significantly across the five subtypes identified by SMMSN in BIC. It can be observed that there exists distinct differences in gene expression among the five subtypes. <xref ref-type="fig" rid="F11">Figure 11</xref> shows the functional enrichment analysis on the differentially expressed genes identified in <xref ref-type="fig" rid="F10">Figure 10</xref>. It can be found that genes related to Wnt signaling pathway and epidermal development are upregulated in subtype 5 and downregulated in subtype 2 and 4. Genes related to extracellular matrix/structural organization, metabolic processes, tissue remodeling and bone development were downregulated in subtypes 4 and 5. Genes related to prostate/breast development and organic/carboxylic acid catabolic processes were downregulated in subtype 5. The differential expression analysis reveals significant gene expression differences among the five cancer subtypes identified by SMMSN. Functional enrichment analysis shows distinct up- and downregulation patterns across the subtypes, highlighting their biological divergence.</p>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>The heatmap showcases the top 1,000 genes whose mRNA expression varies significantly across the three subtypes identified by SMMSN in BIC.</p>
</caption>
<graphic xlink:href="fgene-15-1466825-g010.tif"/>
</fig>
<fig id="F11" position="float">
<label>FIGURE 11</label>
<caption>
<p>Functional enrichment analysis was performed on the differentially expressed genes identified in <xref ref-type="fig" rid="F10">Figure 10</xref>.</p>
</caption>
<graphic xlink:href="fgene-15-1466825-g011.tif"/>
</fig>
</sec>
</sec>
</sec>
<sec sec-type="conclusion" id="s4">
<title>4 Conclusion</title>
<p>Over the past decades, numerous models integrating multi-view biological data, utilizing technologies have been developed and applied to various bioinformatics challenges. These studies have provided valuable insights into understanding the etiology and progression of cancer. Effective mining of cancer subtypes based on biological characteristics from multi-omics data is crucial in bioinformatics research.</p>
<p>In this paper, we introduce a novel method for predicting cancer subtypes called Self-supervised Multi-fusion Strategy Network (SMMSN). SMMSN leverages Stacked Autoencoder (SAE) and Graph Convolutional Network (GCN) modules to learn high-level feature representations and structural representations from each omics data type, respectively. These representations are then integrated to capture comprehensive information across different omics data using two fusion methods: error reconstruction and adaptive weighting network. A dual self-supervised module is employed to jointly train SAE and GCN in an end-to-end manner. Upon convergence, the SMMSN model yields clustering results. We validate the efficacy of SMMSN using 8 real-world cancer datasets, including both labeled and unlabeled multi-omics data, demonstrating its superior performance compared to existing integration methods. Specifically, on GBM data and BIC data, extensive studies confirm that the cancer subtypes predicted by SMMSN exhibit significant and biologically meaningful differences. This underscores the capability of SMMSN to effectively integrate multi-omics data and enhance the understanding of cancer heterogeneity and subtype classification.</p>
<p>Our future research directions could focus on enhancing the interpretability and robustness of the SMMSN model, exploring its application across additional cancer types and expanding its utility in personalized medicine through integration with clinical data.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec sec-type="author-contributions" id="s6">
<title>Author contributions</title>
<p>JL: Data curation, Funding acquisition, Methodology, Writing&#x2013;original draft, Writing&#x2013;review and editing. XX: Data curation, Software, Writing&#x2013;original draft, Investigation. PW: Validation, Writing&#x2013;review and editing, Visualization. QS: Investigation, Validation, Writing&#x2013;review and editing. JY: Formal Analysis, Writing&#x2013;review and editing. SG: Conceptualization, Project administration, Writing&#x2013;review and editing, Investigation.</p>
</sec>
<sec sec-type="funding-information" id="s7">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This research was supported by the National Natural Science Foundation of China (Grant Nos 61906198 and 32100998), the Natural Science Foundation of Jiangsu Province (Grant No. BK20190622), Xuzhou Special Fund for Promoting Science and Technology Innovation-Key R&#x26;D Program (Social Development) (Grant No. KC23237), Wenling Science and Technology Project (Grant Nos 2021S00033 and 2020S0180030).</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Akbani</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Ng</surname>
<given-names>K. S.</given-names>
</name>
<name>
<surname>Werner</surname>
<given-names>H. M.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Mills</surname>
<given-names>G. B.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Abstract 4262: a pan-cancer proteomic analysis of the Cancer Genome Atlas (TCGA) project</article-title>. <source>Cancer Res.</source> <volume>74</volume>, <fpage>4262</fpage>. <pub-id pub-id-type="doi">10.1158/1538-7445.am2014-4262</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Baird</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Roychoudhuri</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>GS-TCGA: gene set-based analysis of the cancer genome atlas</article-title>. <source>J. Comput. Biol.</source> <volume>31</volume> (<issue>3</issue>), <fpage>229</fpage>&#x2013;<lpage>240</lpage>. <pub-id pub-id-type="doi">10.1089/cmb.2023.0278</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bennett</surname>
<given-names>D. A.</given-names>
</name>
<name>
<surname>Schneider</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Arvanitakis</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wilson</surname>
<given-names>R. S.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Overview and findings from the religious orders study</article-title>. <source>Curr. Alzheimer Res.</source> <volume>9</volume> (<issue>6</issue>), <fpage>628</fpage>&#x2013;<lpage>645</lpage>. <pub-id pub-id-type="doi">10.2174/156720512801322573</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bruna</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zaremba</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Szlam</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>LeCun</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Spectral networks and locally connected networks on graphs</article-title>. <source>arXiv Prepr. arXiv:1312.6203</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1312.6203</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Stochastic training of graph convolutional networks with variance reduction</article-title>. <source>arXiv Prepr. arXiv:1710.10568</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1710.10568</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Goodison</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Deep-learning approach to identifying cancer subtypes using high-dimensional genomic data</article-title>. <source>Bioinformatics</source> <volume>35</volume>, <fpage>1476</fpage>&#x2013;<lpage>1483</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz769</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Bioinformatic analysis reveals lysosome-related biomarkers and molecular subtypes in preeclampsia: novel insights into the pathogenesis of preeclampsia</article-title>. <source>Front. Genet.</source> <volume>14</volume>, <fpage>1228110</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2023.1228110</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Dai</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Kozareva</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Smola</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>PMLR. Learning steady-states of iterative algorithms over graphs</article-title>,&#x201d; in <conf-name>In International Conference on Machine Learning</conf-name>, <fpage>1106</fpage>&#x2013;<lpage>1114</lpage>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Defferrard</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bresson</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Vandergheynst</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Convolutional neural networks on graphs with fast localized spectral filtering</article-title>. <source>Proc. 30th Int. Conf. Neural Inf. Process. Syst.</source> <volume>29</volume>, <fpage>3844</fpage>&#x2013;<lpage>3852</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1606.09375</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>De Jager</surname>
<given-names>P. L.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>McCabe</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Vardarajan</surname>
<given-names>B. N.</given-names>
</name>
<name>
<surname>Felsky</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>A multi-omic atlas of the human frontal cortex for aging and Alzheimer&#x2019;s disease research</article-title>. <source>Sci. Data</source> <volume>5</volume> (<issue>1</issue>), <fpage>180142</fpage>&#x2013;<lpage>180213</lpage>. <pub-id pub-id-type="doi">10.1038/sdata.2018.142</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dunnett</surname>
<given-names>C. W.</given-names>
</name>
<name>
<surname>Sobel</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>1954</year>). <article-title>A bivariate generalization of Student&#x27;s t-distribution, with tables for certain special cases</article-title>. <source>Biometrika</source> <volume>41</volume>, <fpage>153</fpage>&#x2013;<lpage>169</lpage>. <pub-id pub-id-type="doi">10.2307/2333013</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Grover</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Leskovec</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Node2vec: scalable feature learning for networks</article-title>,&#x201d; in <conf-name>Proceedings of the 20th ACM International Conference on Knowledge Discovery and Data Mining</conf-name>, <fpage>855</fpage>&#x2013;<lpage>864</lpage>. <pub-id pub-id-type="doi">10.1145/2939672.2939754</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hamilton</surname>
<given-names>W. L.</given-names>
</name>
<name>
<surname>Ying</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Leskovec</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Inductive representation learning on large graphs</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>30</volume>, <fpage>1024</fpage>&#x2013;<lpage>1034</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1706.02216</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hodes</surname>
<given-names>R. J.</given-names>
</name>
<name>
<surname>Buckholtz</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Accelerating medicines partnership: Alzheimer&#x2019;s disease (AMP-AD) knowledge portal aids Alzheimer&#x2019;s drug discovery through open data sharing</article-title>. <source>Expert Opin. Ther. Targets</source> <volume>20</volume> (<issue>4</issue>), <fpage>389</fpage>&#x2013;<lpage>391</lpage>. <pub-id pub-id-type="doi">10.1517/14728222.2016.1135132</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hosmer</surname>
<given-names>D. W.</given-names>
</name>
<name>
<surname>Lemeshow</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>May</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>Applied survival analysis: regression modeling of time to event data</article-title>. <source>J. Stat. Plan. Inference</source> <volume>91</volume>, <fpage>173</fpage>&#x2013;<lpage>175</lpage>. <pub-id pub-id-type="doi">10.1016/s0378-3758(00)00130-0</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Leche</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Kiyatkin</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Stayrook</surname>
<given-names>S. E.</given-names>
</name>
<name>
<surname>Ferguson</surname>
<given-names>K. M.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Glioblastoma mutations alter EGFR dimer structure to prevent ligand bias</article-title>. <source>Nature</source> <volume>602</volume> (<issue>7897</issue>), <fpage>518</fpage>&#x2013;<lpage>522</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-021-04393-3</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jin</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Bernards</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Rational combinations of targeted cancer therapies: background, advances and challenges</article-title>. <source>Nat. Rev. Drug Discov.</source> <volume>22</volume> (<issue>3</issue>), <fpage>213</fpage>&#x2013;<lpage>234</lpage>. <pub-id pub-id-type="doi">10.1038/s41573-022-00615-z</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Livesey</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Eshibona</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Bendou</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Assessment of the progression of kidney renal clear cell carcinoma using transcriptional profiles revealed new cancer subtypes with variable prognosis</article-title>. <source>Front. Genet.</source> <volume>14</volume>, <fpage>1291043</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2023.1291043</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Luxburg</surname>
<given-names>U.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>A tutorial on spectral clustering</article-title>. <source>Statistics Comput.</source> <volume>17</volume>, <fpage>395</fpage>&#x2013;<lpage>416</lpage>. <pub-id pub-id-type="doi">10.1007/s11222-007-9033-z</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Integrate multi-omic data using affinity network fusion (ANF) for cancer patient clustering</article-title>,&#x201d; in <conf-name>2017 IEEE International Conference on Bioinformatics and Biomedicine (BIBM)</conf-name>, <fpage>398</fpage>&#x2013;<lpage>403</lpage>. <pub-id pub-id-type="doi">10.1109/bibm.2017.8217682</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mo</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Seshan</surname>
<given-names>V. E.</given-names>
</name>
<name>
<surname>Olshen</surname>
<given-names>A. B.</given-names>
</name>
<name>
<surname>Schultz</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Sander</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Pattern discovery and cancer gene identification in integrated cancer genomic data</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A.</source> <volume>110</volume>, <fpage>4245</fpage>&#x2013;<lpage>4250</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1208949110</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Noushmehr</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Weisenberger</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Diefes</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Phillips</surname>
<given-names>H. S.</given-names>
</name>
<name>
<surname>Pujara</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Berman</surname>
<given-names>B. P.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>Identification of a CpG island methylator phenotype that defines a distinct subgroup of glioma</article-title>. <source>Cancer Cell</source> <volume>17</volume>, <fpage>510</fpage>&#x2013;<lpage>522</lpage>. <pub-id pub-id-type="doi">10.1016/j.ccr.2010.03.017</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Parker</surname>
<given-names>J. S.</given-names>
</name>
<name>
<surname>Mullins</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Cheang</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Leung</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Voduc</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Vickery</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>Supervised risk predictor of breast cancer based on intrinsic subtypes</article-title>. <source>J. Clin. Oncol.</source> <volume>27</volume> (<issue>8</issue>), <fpage>1160</fpage>&#x2013;<lpage>1167</lpage>. <pub-id pub-id-type="doi">10.1200/JCO.2008.18.1370</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Perozzi</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Al-Rfou</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Skiena</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>Deepwalk: online learning of social representations</article-title>,&#x201d; in <conf-name>Proceedings of the 20th ACM International Conference on Knowledge Discovery and Data Mining</conf-name>, <fpage>701</fpage>&#x2013;<lpage>710</lpage>. <pub-id pub-id-type="doi">10.1145/2623330.2623732</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rappoport</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Shamir</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Multi-omic and multi-view clustering algorithms: review and cancer benchmark</article-title>. <source>Nucleic Acids Res.</source> <volume>46</volume>, <fpage>10546</fpage>&#x2013;<lpage>10562</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gky889</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Olshen</surname>
<given-names>A. B.</given-names>
</name>
<name>
<surname>Ladanyi</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Integrative clustering of multiple genomic data types using a joint latent variable model with application to breast and lung cancer subtype analysis</article-title>. <source>Bioinformatics</source> <volume>25</volume>, <fpage>2906</fpage>&#x2013;<lpage>2912</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp543</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Pattern fusion analysis by adaptive alignment of multiple heterogeneous omics data</article-title>. <source>Bioinformatics</source> <volume>33</volume>, <fpage>2706</fpage>&#x2013;<lpage>2714</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btx176</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sosinsky</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ambrose</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cross</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Turnbull</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Henderson</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Insights for precision oncology from the integration of genomic and clinical data of 13,880 tumors from the 100,000 Genomes Cancer Programme</article-title>. <source>Nat. Med.</source> <volume>30</volume> (<issue>1</issue>), <fpage>279</fpage>&#x2013;<lpage>289</lpage>. <pub-id pub-id-type="doi">10.1038/s41591-023-02682-0</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Tang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Qu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Mei</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Line: large-scale information network embedding</article-title>,&#x201d; in <conf-name>Proceedings of the 24th International Conference on World Wide Web</conf-name>, <fpage>1067</fpage>&#x2013;<lpage>1077</lpage>. <pub-id pub-id-type="doi">10.1145/2736277.2741093</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Tao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Adversarial graph embedding for ensemble clustering</article-title>,&#x201d; in <conf-name>Proceedings of the Twenty-Eighth International Joint Conference on Artificial Intelligence</conf-name>, <fpage>3562</fpage>&#x2013;<lpage>3568</lpage>. <pub-id pub-id-type="doi">10.24963/ijcai.2019/494</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Thomas</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Kipf</surname>
<given-names>M. W.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Semi-supervised classification with graph convolutional networks</article-title>,&#x201d; in <conf-name>Proceedings of International Conference on Learning Representations</conf-name>, <fpage>1</fpage>&#x2013;<lpage>14</lpage>.</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Veli&#x10d;kovi&#x107;</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Cucurull</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Casanova</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Romero</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lio</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Graph attention networks</article-title>. <source>arXiv Prepr. arXiv:1710.10903</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1710.10903</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Verhaak</surname>
<given-names>R. G. W.</given-names>
</name>
<name>
<surname>Hoadley</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Purdom</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wilkerson</surname>
<given-names>M. D.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>Integrated genomic analysis identifies clinically relevant subtypes of glioblastoma characterized by abnormalities in PDGFRA, IDH1, EGFR, and NF1</article-title>. <source>Cancer Cell</source> <volume>17</volume>, <fpage>98</fpage>&#x2013;<lpage>110</lpage>. <pub-id pub-id-type="doi">10.1016/j.ccr.2009.12.020</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Mezlini</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Demir</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Fiume</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Tu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Brudno</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Similarity network fusion for aggregating data types on a genomic scale</article-title>. <source>Nat. Methods</source> <volume>11</volume>, <fpage>333</fpage>&#x2013;<lpage>337</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.2810</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Accelerating the understanding of cancer biology through the lens of genomics</article-title>. <source>Cell</source> <volume>186</volume> (<issue>8</issue>), <fpage>1755</fpage>&#x2013;<lpage>1771</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2023.02.015</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Tao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Yun</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Partial multi-view clustering via consistent GAN</article-title>,&#x201d; in <conf-name>Proceedings of the 2018 IEEE International Conference on Data Mining (ICDM)</conf-name>, <fpage>1290</fpage>&#x2013;<lpage>1295</lpage>.</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Shao</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>MOGONET integrates multi-omics data using graph convolutional networks allowing patient classification and biomarker identification</article-title>. <source>Nat. Commun.</source> <volume>12</volume> (<issue>1</issue>), <fpage>3445</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-021-23774-w</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Way</surname>
<given-names>G. P.</given-names>
</name>
<name>
<surname>Greene</surname>
<given-names>C. S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Extracting a biologically relevant latent space from cancer transcriptomes with variational autoencoders</article-title>. <source>Pac Symp. Biocomput</source> <volume>23</volume>, <fpage>80</fpage>&#x2013;<lpage>91</lpage>. <pub-id pub-id-type="doi">10.1142/9789813235533_0008</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>M. Q.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Fast dimension reduction and integrative clustering of multi-omics data using low-rank approximation: application to cancer molecular classification</article-title>. <source>BMC Genomics</source> <volume>16</volume>, <fpage>1022</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-015-2223-8</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Dawood</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A hierarchical integration deep flexible neural forest framework for cancer subtype classification by integrating multi-omics data</article-title>. <source>BMC Bioinforma.</source> <volume>20</volume>, <fpage>527</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-019-3116-7</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Metabolic reprogramming and epigenetic modifications in cancer: from the impacts and mechanisms to the treatment potential</article-title>. <source>Exp. and Mol. Med.</source> <volume>55</volume> (<issue>7</issue>), <fpage>1357</fpage>&#x2013;<lpage>1370</lpage>. <pub-id pub-id-type="doi">10.1038/s12276-023-01020-1</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Sheng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Subtype-former: a deep learning approach for cancer subtype discovery with multi-omics data</article-title>. <source>arxiv Prepr. arxiv:2207.14639</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2207.14639</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L. H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Simultaneous clustering of multiview biomedical data using manifold optimization</article-title>. <source>Bioinformatics</source> <volume>35</volume>, <fpage>4029</fpage>&#x2013;<lpage>4037</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz217</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zeng</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhan</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>SEC61G assists EGFR-amplified glioblastoma to evade immune elimination</article-title>. <source>Proc. Natl. Acad. Sci.</source> <volume>120</volume> (<issue>32</issue>), <fpage>e2303400120</fpage>. <pub-id pub-id-type="doi">10.1073/pnas.2303400120</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Mou</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Shang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Long noncoding RNA RP11&#x2010;626G11. 3 promotes the progression of glioma through miR&#x2010;375&#x2010;SP1 axis</article-title>. <source>Mol. Carcinog.</source> <volume>59</volume> (<issue>5</issue>), <fpage>492</fpage>&#x2013;<lpage>502</lpage>. <pub-id pub-id-type="doi">10.1002/mc.23173</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Lyu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Xiong</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Subtype-DCC: decoupled contrastive clustering method for cancer subtype identification based on multi-omics data</article-title>. <source>Briefings Bioinforma.</source> <volume>24</volume> (<issue>2</issue>), <fpage>bbad025</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbad025</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>