<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">869906</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2022.869906</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Dynamic Meta-data Network Sparse PCA for Cancer Subtype Biomarker Screening</article-title>
<alt-title alt-title-type="left-running-head">Miao et al.</alt-title>
<alt-title alt-title-type="right-running-head">DM-ESPCA for Cancer Biomarkers Screening</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Miao</surname>
<given-names>Rui</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1441450/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Dong</surname>
<given-names>Xin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Xiao-Ying</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lo</surname>
<given-names>Sio-Long</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Mei</surname>
<given-names>Xin-Yue</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Dang</surname>
<given-names>Qi</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Cai</surname>
<given-names>Jie</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Shao</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/483402/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yang</surname>
<given-names>Kuo</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1551623/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xie</surname>
<given-names>Sheng-Li</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1730734/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Liang</surname>
<given-names>Yong</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1666077/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Institute of Systems Engineering</institution>, <institution>Macau University of Science and Technology</institution>, <institution>Avenida Wai Long</institution>, <addr-line>Taipa</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Computer Engineering Technical College</institution>, <institution>Guangdong Polytechnic of Science and Technology</institution>, <addr-line>Zhuhai</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>MOE Key Laboratory of Bioinformatics</institution>, <institution>TCM-X Center/Bioinformatics Division</institution>, <institution>BNRIST/Department of Automation</institution>, <institution>Tsinghua University</institution>, <addr-line>Beijing</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Guangdong-HongKong-Macao Joint Laboratory for Smart Discrete Manufacturing</institution>, <addr-line>Guangzhou</addr-line>, <country>China</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Peng Cheng Laboratory</institution>, <addr-line>Shenzhen</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/806390/overview">Pietro Zoppoli</ext-link>, University of Naples Federico II, Italy</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1419511/overview">Wenwen Min</ext-link>, Yunnan University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/153656/overview">Guoxian Yu</ext-link>, Shandong University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Yong Liang, <email>yongliangresearch@gmail.com</email>
</corresp>
<fn fn-type="other">
<p>This article was submitted to Computational Genomics, a section of the journal Frontiers in Genetics</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>09</day>
<month>05</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>869906</elocation-id>
<history>
<date date-type="received">
<day>05</day>
<month>02</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>31</day>
<month>03</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Miao, Dong, Liu, Lo, Mei, Dang, Cai, Li, Yang, Xie and Liang.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Miao, Dong, Liu, Lo, Mei, Dang, Cai, Li, Yang, Xie and Liang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Previous research shows that each type of cancer can be divided into multiple subtypes, which is one of the key reasons that make cancer difficult to cure. Under these circumstances, finding a new target gene of cancer subtypes has great significance on developing new anti-cancer drugs and personalized treatment. Due to the fact that gene expression data sets of cancer are usually high-dimensional and with high noise and have multiple potential subtypes&#x2019; information, many sparse principal component analysis (sparse PCA) methods have been used to identify cancer subtype biomarkers and subtype clusters. However, the existing sparse PCA methods have not used the known cancer subtype information as prior knowledge, and their results are greatly affected by the quality of the samples. Therefore, we propose the Dynamic Metadata Edge-group Sparse PCA (DM-ESPCA) model, which combines the idea of meta-learning to solve the problem of sample quality and uses the known cancer subtype information as prior knowledge to capture some gene modules with better biological interpretations. The experiment results on the three biological data sets showed that the DM-ESPCA model can find potential target gene probes with richer biological information to the cancer subtypes. Moreover, the results of clustering and machine learning classification models based on the target genes screened by the DM-ESPCA model can be improved by up to 22&#x2013;23% of accuracies compared with the existing sparse PCA methods. We also proved that the result of the DM-ESPCA model is better than those of the four classic supervised machine learning models in the task of classification of cancer subtypes.</p>
</abstract>
<kwd-group>
<kwd>Cancer subtype</kwd>
<kwd>biomarkers</kwd>
<kwd>sparse PCA</kwd>
<kwd>DM-ESPCA model</kwd>
<kwd>meta-data</kwd>
<kwd>dynamic network</kwd>
</kwd-group>
<contract-sponsor id="cn001">Macau University of Science and Technology Foundation<named-content content-type="fundref-id">10.13039/501100011322</named-content>
</contract-sponsor>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>As the most difficult-to-cure malignant disease in the world, how to defeat cancer has received extensive attention from researchers (<xref ref-type="bibr" rid="B36">Siegel et al., 2016</xref>; <xref ref-type="bibr" rid="B37">Siegel et al., 2019</xref>). The latest research shows that each type of cancer can derive many subtypes, which may be one of the reasons why personalized cancer treatment is needed (<xref ref-type="bibr" rid="B27">Nguyen et al., 2008</xref>; <xref ref-type="bibr" rid="B3">Cancello et al., 2010</xref>; <xref ref-type="bibr" rid="B16">Houssami et al., 2012</xref>; <xref ref-type="bibr" rid="B39">Symmans et al., 2017</xref>; and <xref ref-type="bibr" rid="B44">Waks and Winer, 2019</xref>). For example, the ceritinib capsule is a targeted drug for lung cancer (the target gene is ALK) (<xref ref-type="bibr" rid="B5">Cooper et al., 2015</xref>; <xref ref-type="bibr" rid="B29">Raedler, 2015</xref>). However, existing studies have shown that it only has a good effect on a small number of lung cancer patients. The reason for this problem is that only 35&#x2013;36% of lung cancer patients are caused by ALK gene mutations, which means that the ceritinib capsule is only effective for one subtype of lung cancer (<xref ref-type="bibr" rid="B9">Deeks, 2016</xref>). Therefore, the identification and recognition of potential target genes corresponding to cancer subtypes have become an important task in cancer research (<xref ref-type="bibr" rid="B1">Banerji et al., 2012</xref>; <xref ref-type="bibr" rid="B2">Calon et al., 2015</xref>; and <xref ref-type="bibr" rid="B8">De Cecco et al., 2015</xref>).</p>
<p>With the rapid development of the high-throughput sequencing technology, there are a lot of biological data that have been collected from many large-scale projects, which provides a basis to establish machine learning models for biomarker screening. At present, there are two types of machine learning models for screening target genes of potential cancer subtypes. One is the supervised classification models. Gene expression data sets of cancer are usually high-dimensional and with high noise and small sample sizes, which easily lead to overfitting of supervised machine learning models (<xref ref-type="bibr" rid="B13">Gao et al., 2019</xref>; <xref ref-type="bibr" rid="B18">Lee et al., 2020</xref>). Moreover, the other problem with the supervised models is that the gene probes screened by these models may not have good biological interpretation, and different models may screen out very different gene probes in the same data set (<xref ref-type="bibr" rid="B46">Xie et al., 2019</xref>; <xref ref-type="bibr" rid="B47">Yang et al., 2019</xref>). The other type is the unsupervised biomarker extraction models. The principle of these models is to perform cancer subtype clustering and target gene screening based on potential patterns of samples. Among them, the sparse principal component analysis (sparse PCA) methods are widely used methods of unsupervised biomarker extraction, which can capture the linear relationship of variables to best explain the latent patterns of cancer subtypes. Moreover, the potential target genes screened by the sparse PCA methods may tend to have good biological interpretability (<xref ref-type="bibr" rid="B34">Shen et al., 2009</xref>; <xref ref-type="bibr" rid="B33">Shen et al., 2012</xref>; and <xref ref-type="bibr" rid="B24">Min et al., 2018</xref>).</p>
<p>Currently, researchers have proposed some sparse PCA and joint latent variable methods for identifying driver genes of cancer or biomarkers of cancer subtypes. For example, in 2009, <xref ref-type="bibr" rid="B34">Shen et al. (2009)</xref> proposed a cancer subtype clustering model (iCluster) based on joint latent variable of data. In 2011, SAN et al. (<xref ref-type="bibr" rid="B26">Navarro Silvera et al., 2011</xref>) used PCA and logistic regression to analyze the risk factors of esophageal cancer and gastric cancer. <xref ref-type="bibr" rid="B32">Shen et al. (2013)</xref> further extended the iCluster model with LASSO, elastic net, and fusion LASSO methods to allow feature selection in an integrated clustering environment. The overall goal of these models is to obtain joint clustering of samples and identify cluster-related features across data sets. In 2015, <xref ref-type="bibr" rid="B38">Sill et al. (2015)</xref> proposed a sparse PCA method (S4VDPCA) with stable selection ability to process the medulloblastoma brain gene expression data set and revealed that the genes determined by the first two sparse PC loadings significantly participated in the marrow and several key pathways between the molecular subgroups of blastoma. In 2018, <xref ref-type="bibr" rid="B24">Min et al. (2018)</xref> proposed an edge group sparse PCA model (ESPCA) which effectively enhanced the potential gene selection ability of sparse PCA. Existing research shows that structured sparse models similar to ESPCA can effectively improve the biological interpretability and feature selection capabilities of the models (<xref ref-type="bibr" rid="B23">Min et al., 2016</xref>; <xref ref-type="bibr" rid="B25">Min et al., 2019</xref>; <xref ref-type="bibr" rid="B43">Vinga, 2021</xref>).</p>
<p>However, the existing sparse PCA methods still have three main issues. First, all these methods are reference-free methods, which means that they do not consider the known subtype classification information of the cancer data set (<xref ref-type="bibr" rid="B30">Reis-Filho and Pusztai, 2011</xref>; <xref ref-type="bibr" rid="B7">Dai et al., 2015</xref>). The existing research works have shown that reference-free sparse PCA methods may discard some potential biomarkers in the process of sparseness (<xref ref-type="bibr" rid="B17">Kim et al., 2019</xref>). The second one is that the samples of the biological data contain a lot of noise (<xref ref-type="bibr" rid="B40">Teng, 2003</xref>; <xref ref-type="bibr" rid="B22">Linck and Battey, 2019</xref>), which will affect the final results of the model and eventually lead researchers to find the wrong potential target gene. The third issue is that most of the existing sparse PCA methods use the greedy optimization principle to select target gene probes, which will make the model fall quickly into a local optimum.</p>
<p>In order to solve the three problems mentioned mentioned above, this article proposes the DM-ESPCA model, which uses the dynamic gene network, meta-learning approach, and random sampling algorithm based on the greedy principle (<xref ref-type="fig" rid="F1">Figure 1</xref>). The purpose of the dynamic gene network is to enhance the feature selection ability of the model to screen out potential target genes that are more relevant to the cancer subtype. The meta-learning approach is an efficient machine learning framework, which uses a small number of high-quality samples to adjust the parameters of the machine learning model to reduce the errors caused by the noise data. We also proposed a random sampling algorithm based on the greedy principle to obtain a better solution in the process of sparseness.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Flow chart of the DM-ESPCA model. <bold>(A)</bold> The DM-ESPCA model requires input gene expression and pathway data. <bold>(B)</bold> The DM-ESPCA model selects meta-data by clustering all samples. <bold>(C)</bold> Workflow of the DM-ESPCA model to screen targeted genes. The DM-ESPCA model will generate a dynamic gene network for each subtype. <bold>(D)</bold> Finally, this model will output the screened genes.</p>
</caption>
<graphic xlink:href="fgene-13-869906-g001.tif"/>
</fig>
<p>The steps of the DM-ESPCA model are as follows: 1) filter meta-data for each subtype in the cancer data set; 2) based on meta-data, use known subtype classification information as prior knowledge to calculate the correlation degree of each gene probe corresponding to each subtype; 3) use the quantitative value of correlation as a parameter to generate a unique biological network for each subtype; and 4) build the DM-ESPCA model using the dynamic gene network to screen biomarkers for each subtype.</p>
<p>This article conducted experiments on three data sets, and the results showed that the DM-ESPCA model is better than the existing sparse PCA methods. The heat maps and bio-enrichment analyses show that the potential target genes screened by the DM-ESPCA model have higher correlations and richer biological information with the corresponding cancer subtypes. The results of re-clustering and the accuracies of machine learning classification models based on the potential target genes screened by the DM-ESPCA model can be improved by up to 23 and 22%, respectively.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>Materials and Methods</title>
<sec id="s2-1">
<title>Data Sets</title>
<p>In this experiment, we used three cancer data sets to test the performance of the DM-ESPCA model, including two breast cancer data sets and one gastric cancer data set. All these data sets were assayed with the Human Genome U133 Plus 2.0 microarray (HG-U133_Plus_2). This gene chip contains 54,675 probes (<xref ref-type="bibr" rid="B4">Carlson et al., 2016</xref>). The following is a detailed introduction to the data sets (<xref ref-type="table" rid="T1">Table 1</xref>):</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Details of the three data sets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="center">BCI</th>
<th align="center">BCII</th>
<th align="center">GC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Number of samples</td>
<td align="center">155</td>
<td align="center">178</td>
<td align="center">70</td>
</tr>
<tr>
<td align="left">Number of genes</td>
<td align="center">54,675</td>
<td align="center">54,675</td>
<td align="center">54,675</td>
</tr>
<tr>
<td align="left">Number of subtypes</td>
<td align="center">4</td>
<td align="center">4</td>
<td align="center">5</td>
</tr>
<tr>
<td align="left">ID</td>
<td align="center">E-GEOD-45827</td>
<td align="center">E-GEOD-65194</td>
<td align="center">E-GEOD-35809</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>First, we used a breast cancer subtype data set, numbered E-GEOD-45827 (BCI, <ext-link ext-link-type="uri" xlink:href="https://www.ebi.ac.uk/arrayexpress/experiments/E-GEOD-45827/">https://www.ebi.ac.uk/arrayexpress/experiments/E-GEOD-45827/</ext-link>). Since breast cancer is a kind of malignant cancer, its incidence rate ranks first among female malignant cancers all year round and is still increasing year by year (<xref ref-type="bibr" rid="B10">DeSantis et al., 2014</xref>; <xref ref-type="bibr" rid="B11">Fan et al., 2014</xref>). Therefore, the analysis of breast cancer data sets is greatly significant. Meanwhile, breast cancer has a clear subtype division, which is mainly divided into four subtypes, including Basal, Her2, Luminal A, and Luminal B (<xref ref-type="bibr" rid="B41">Tran and Bedard, 2011</xref>). The BCI data set we used in this experiment contains 155 samples (Supplementary Fig.1.A).</p>
<p>Next, we used another breast cancer data set, numbered E-GEOD-65194 (BCII, <ext-link ext-link-type="uri" xlink:href="https://www.ebi.ac.uk/arrayexpress/experiments/E-GEOD-65194/">https://www.ebi.ac.uk/arrayexpress/experiments/E-GEOD-65194/</ext-link>). The purpose of using the BCII data set is to verify whether our proposed model can correctly classify the subtypes and whether it has sufficient stability in the same cancer but different batches of data collection. Here, the BCII data set also has four subtypes, including TNBC, Her2, Luminal A, and Luminal B. Based on the existing studies, TNBC and Basal can easily be regarded as the same subtype (<xref ref-type="bibr" rid="B45">Wiese et al., 2013</xref>). We obtained BCII with 178 samples (Supplementary Fig.1.B).</p>
<p>Finally, we conducted an experiment using a gastric cancer data set, numbered E-GEOD-35809 (GC, <ext-link ext-link-type="uri" xlink:href="https://www.ebi.ac.uk/arrayexpress/experiments/E-GEOD-35809/">https://www.ebi.ac.uk/arrayexpress/experiments/E-GEOD-35809/</ext-link>). Gastric cancer is also a common malignant cancer (<xref ref-type="bibr" rid="B6">Crew and Neugut, 2006</xref>). Its incidence rate remains high in the global incidence statistics of malignant cancers (<xref ref-type="bibr" rid="B14">Hartgrink et al., 2009</xref>). In addition, the existing studies have found that gastric cancer also has multiple subtypes. The data set used in this experiment includes three subtypes: proliferative, invasive, and metabolic (<xref ref-type="sec" rid="s10">Supplementary Figure 1C</xref>) (<xref ref-type="bibr" rid="B19">Lei et al., 2013</xref>; <xref ref-type="bibr" rid="B49">Zeng et al., 2018</xref>). The purpose of using gastric cancer data is to test whether the DM-ESPCA model can be applied to different cancer subtypes&#x2019; research.</p>
<p>In this study, we used a mixed model of GC-RMA to preprocess all these three data sets to reduce the negative impact of the batch. Specifically, we discarded all the probes with a log2 intensity of less than 4.</p>
</sec>
<sec id="s2-2">
<title>Gene Pathway Data Sets</title>
<p>The basic network data set used by the DM-ESPCA model is obtained from the following database: Pathway Commons database (<ext-link ext-link-type="uri" xlink:href="http://www.pathwaycommons.org/">http://www.pathwaycommons.org/</ext-link>).</p>
<p>Totally, the BCI and BCII data sets retained the same 29,873 gene probes, and the corresponding relationship network retained 1,239,154 edges. The GC data set retained 28,838 gene probes, and 1,181,312 edges were retained in the corresponding relationship network.</p>
</sec>
<sec id="s2-3">
<title>Methods</title>
<p>In this section, we first introduced the general sparse PCA framework (SPCA). Then, we introduced the ESPCA model. Finally, we proposed the DM-ESPCA model which includes meta-data selection, the dynamic gene network, and the random sampling algorithm based on the greedy principle.</p>
<sec id="s2-3-1">
<title>SPCA</title>
<p>Suppose there is a gene matrix <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> containing <inline-formula id="inf2">
<mml:math id="m2">
<mml:mi>m</mml:mi>
</mml:math>
</inline-formula> genes and <inline-formula id="inf3">
<mml:math id="m3">
<mml:mi>n</mml:mi>
</mml:math>
</inline-formula> samples. Using the <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> norm for sparseness, we can get the following expression matrix (<xref ref-type="bibr" rid="B48">Yuan and Zhang, 2013</xref>):<disp-formula id="e1">
<mml:math id="m5">
<mml:mrow>
<mml:munder>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>z</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2225;</mml:mo>
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2225;</mml:mo>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munder>
<mml:msup>
<mml:mi>u</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mi>X</mml:mi>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>s</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>.</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mo>&#x2225;</mml:mo>
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>&#x2225;</mml:mo>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <inline-formula id="inf5">
<mml:math id="m6">
<mml:mi>u</mml:mi>
</mml:math>
</inline-formula> is the <inline-formula id="inf6">
<mml:math id="m7">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> vector to represent the first principal component (PC) loading and s represents the number of genes retained by the model, and <inline-formula id="inf7">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf8">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represent the <inline-formula id="inf9">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf10">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> norms, respectively. Researchers usually use the SVD framework to solve this problem (<xref ref-type="bibr" rid="B21">Lin et al., 2016</xref>). Therefore, the formula can also be written as<disp-formula id="e2">
<mml:math id="m12">
<mml:mrow>
<mml:munder>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>z</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mi>u</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mi>v</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munder>
<mml:msup>
<mml:mi>u</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mi>X</mml:mi>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>s</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>.</mml:mo>
<mml:mo>&#x2225;</mml:mo>
<mml:mi>u</mml:mi>
<mml:msub>
<mml:mo>&#x2225;</mml:mo>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <inline-formula id="inf11">
<mml:math id="m13">
<mml:mi>v</mml:mi>
</mml:math>
</inline-formula> is <inline-formula id="inf12">
<mml:math id="m14">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> PC. The problem is solved using the following strategies:<disp-formula id="e3">
<mml:math id="m15">
<mml:mrow>
<mml:mi mathvariant="normal">u&#x2190;</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="normal">u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="italic">where</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mi mathvariant="normal">&#x3d;P</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">z,s</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="italic">and</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">z&#x3d;Xv</mml:mi>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
<disp-formula id="e4">
<mml:math id="m16">
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>&#x2190;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>v</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mi>v</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mi>w</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>v</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>where <inline-formula id="inf13">
<mml:math id="m17">
<mml:mrow>
<mml:mi mathvariant="script">P</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents sparse projection. In the vector <inline-formula id="inf14">
<mml:math id="m18">
<mml:mi>u</mml:mi>
</mml:math>
</inline-formula>, its <inline-formula id="inf15">
<mml:math id="m19">
<mml:mi>k</mml:mi>
</mml:math>
</inline-formula>-th element has the following defined:<disp-formula id="e5">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">z,s</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi mathvariant="normal">k</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">&#x3d;</mml:mi>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">z</mml:mi>
<mml:mi mathvariant="normal">k</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">,&#xa0;if&#xa0;k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="normal">supp</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">z,s</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi mathvariant="italic">0,otherwise</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>where <inline-formula id="inf16">
<mml:math id="m21">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the set of indexes of the largest <inline-formula id="inf17">
<mml:math id="m22">
<mml:mi>s</mml:mi>
</mml:math>
</inline-formula> absolute element of <inline-formula id="inf18">
<mml:math id="m23">
<mml:mi>z</mml:mi>
</mml:math>
</inline-formula>.</p>
</sec>
<sec id="s2-3-2">
<title>ESPCA</title>
<p>In 2018, Min et al. proposed the edge group sparse PCA (ESPCA), which uses known genome structures as prior knowledge (<xref ref-type="bibr" rid="B24">Min et al., 2018</xref>). The ESPCA model is transformed from a traditional point sparse to a group sparse which effectively improves the feature screening ability of sparse PCA. Suppose <inline-formula id="inf19">
<mml:math id="m24">
<mml:mi mathvariant="script">G</mml:mi>
</mml:math>
</inline-formula> is a group structure, in the gene interaction network, the two linked genes can be considered as a group. Obviously, such edge groups are overlapping. We denoted <inline-formula id="inf20">
<mml:math id="m25">
<mml:mrow>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> as an edge set with all edges from a given gene interaction network. Here, the ESPCA model is as follows:<disp-formula id="e6">
<mml:math id="m26">
<mml:mrow>
<mml:mo>&#x2225;</mml:mo>
<mml:msub>
<mml:mi>u&#x2225;</mml:mi>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:munder>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>z</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2200;</mml:mo>
<mml:msup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>u</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2286;</mml:mo>
<mml:mi>V</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:munder>
<mml:mrow>
<mml:mo>&#x7c;</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
<mml:mo>&#x7c;</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>where <inline-formula id="inf21">
<mml:math id="m27">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is a subset of <inline-formula id="inf22">
<mml:math id="m28">
<mml:mi mathvariant="script">G</mml:mi>
</mml:math>
</inline-formula> , <inline-formula id="inf23">
<mml:math id="m29">
<mml:mrow>
<mml:mi mathvariant="normal">V</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi mathvariant="normal">&#x2032;</mml:mi>
</mml:msup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is a vertex (gene) set induced from the edge set <inline-formula id="inf24">
<mml:math id="m30">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf25">
<mml:math id="m31">
<mml:mrow>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mrow>
<mml:mo>&#x7c;</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
<mml:mo>&#x7c;</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the number of elements of <inline-formula id="inf26">
<mml:math id="m32">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf27">
<mml:math id="m33">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>u</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the set of indexes of the non-zero elements of <inline-formula id="inf28">
<mml:math id="m34">
<mml:mi>u</mml:mi>
</mml:math>
</inline-formula> (<xref ref-type="bibr" rid="B24">Min et al., 2018</xref>). Based on <xref ref-type="disp-formula" rid="e6">formula 6</xref>, this sparse model can be expressed as the following formula:<disp-formula id="e7">
<mml:math id="m35">
<mml:mrow>
<mml:munder>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>z</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mi>u</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mi>v</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munder>
<mml:msup>
<mml:mi>u</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mi>X</mml:mi>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>s</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>.</mml:mo>
<mml:mo>&#x2225;</mml:mo>
<mml:mi>u</mml:mi>
<mml:msub>
<mml:mo>&#x2225;</mml:mo>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>where <inline-formula id="inf29">
<mml:math id="m36">
<mml:mi>s</mml:mi>
</mml:math>
</inline-formula> is the amount of edges. The model is solved based on a greedy algorithm.</p>
</sec>
<sec id="s2-3-3">
<title>DM-ESPCA</title>
<p>On the basis of SPCA and ESPCA models, we propose the DM-ESPCA model. Compared to existing models, the DM-ESPCA model has three main improvements. First, the DM-ESPCA model generates independent dynamic network weights for each PC based on known cancer subtype classification information and integrates the weights into the sparse PCA framework which enhances the model&#x2019;s cancer subtype target selection capabilities. Second, in the process of generating the dynamic network weights of the DM-ESPCA model, the DM-ESPCA model improves the sample quality and noise of the data set by selecting a subset of meta-data. It ensures the accuracy and reliability of the dynamic network weights. Third, the DM-ESPCA model improves the traditional greedy algorithm and proposes a random sampling algorithm based on the greedy principle, which improves the local optimal solution of the model. Next, we introduce the details of meta-data selection, the dynamic network, and the random sampling algorithm based on the greedy principle modules in the order of model construction (<xref ref-type="fig" rid="F2">Figure 2</xref>).</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Algorithm of the DM-ESPCA model.</p>
</caption>
<graphic xlink:href="fgene-13-869906-g002.tif"/>
</fig>
<sec id="s2-3-3-1">
<title>Meta-data Selection</title>
<p>The cancer subtype data sets are inevitably noisy, which will mislead the results of machine learning models (since the cancer subtype data sets are inevitably noisy and mislead the results of machine learning models). To solve this problem, the establishment of the dynamic network is based on meta-data (high-quality samples) after preprocessing, not all samples. Here, we adopt the idea of meta-learning to initialize model parameters with high-quality samples as much as possible and guide the operation of the entire model. It should be stated that the idea of meta-learning here means that the model uses a batch of high-quality sample data sets to guide the training of the model based on all samples (<xref ref-type="bibr" rid="B35">Shu et al., 2019</xref>). It does not refer to the multi-task meta-learning training mode similar to the MAML model (<xref ref-type="bibr" rid="B12">Finn et al., 2017</xref>). The following content is the steps for selecting meta-data from the cancer subtype data set:</p>
<p>First, we use all gene probes to cluster the subtype data sets which adopt the K-means algorithm.</p>
<p>According to the known clustering information, we select h samples closest to the cluster center point in each cancer subtype.</p>
<p>We repeat clustering multiple times, and the final result is that the samples are stably selected each time.</p>
</sec>
<sec id="s2-3-3-2">
<title>Dynamic Meta-data Network</title>
<p>Existing sparse PCA methods are all reference-free methods. Even in the ESPCA model, its used weights of the biological networks for the principal components are the same. In this article, we pre-calculate the correlation weights of each gene probe and each cancer subtype based on general biological knowledge and meta-data. These weights are used to establish a dynamic biological network for each cancer subtype, thereby enhancing the model&#x2019;s gene screening ability.</p>
<p>Here, we presented the DM-ESPCA model as formulas 8 and 12. First, we assume that <inline-formula id="inf30">
<mml:math id="m37">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>h</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mi>m</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, and the weight <inline-formula id="inf31">
<mml:math id="m38">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>h</mml:mi>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> of <inline-formula id="inf32">
<mml:math id="m39">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>h</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is defined as <xref ref-type="disp-formula" rid="e8">formula 8</xref>:<disp-formula id="e8">
<mml:math id="m40">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>h</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:msubsup>
<mml:mi>u</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mi>u</mml:mi>
<mml:mi>j</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>where <inline-formula id="inf33">
<mml:math id="m41">
<mml:mrow>
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf34">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the left and right gene probes of <inline-formula id="inf35">
<mml:math id="m43">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>h</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, respectively.</p>
<p>Then, we adopted <xref ref-type="disp-formula" rid="e9">formula 9</xref> to pre-calculate the correlation weight <inline-formula id="inf36">
<mml:math id="m44">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> of <inline-formula id="inf37">
<mml:math id="m45">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>th&#xa0;subtype&#xa0;and&#xa0;</mml:mtext>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th gene probe in the dynamic network of the DM-ESPCA model<disp-formula id="e9">
<mml:math id="m46">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>where <inline-formula id="inf38">
<mml:math id="m47">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mtext>i</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf39">
<mml:math id="m48">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mtext>i</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the average value and the standard deviation of the <inline-formula id="inf40">
<mml:math id="m49">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula>-th gene probe in the <inline-formula id="inf41">
<mml:math id="m50">
<mml:mi>p</mml:mi>
</mml:math>
</inline-formula>-th subtype with meta-data samples, respectively, and <inline-formula id="inf42">
<mml:math id="m51">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mtext>i</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the number of samples of the <inline-formula id="inf43">
<mml:math id="m52">
<mml:mi>p</mml:mi>
</mml:math>
</inline-formula>-th subtype in the meta-data. <inline-formula id="inf44">
<mml:math id="m53">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>i</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf45">
<mml:math id="m54">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>i</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> indicate the average value and the standard deviation of the samples with the <inline-formula id="inf46">
<mml:math id="m55">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula>-th gene probe not in the <inline-formula id="inf47">
<mml:math id="m56">
<mml:mi>p</mml:mi>
</mml:math>
</inline-formula>-th subtype, respectively, and <inline-formula id="inf48">
<mml:math id="m57">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>i</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the number of the samples not in the <inline-formula id="inf49">
<mml:math id="m58">
<mml:mi>p</mml:mi>
</mml:math>
</inline-formula>-th subtype.</p>
<p>Therefore, the weight of the <inline-formula id="inf50">
<mml:math id="m59">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula>-th gene probe in <inline-formula id="inf51">
<mml:math id="m60">
<mml:mi>p</mml:mi>
</mml:math>
</inline-formula>-th subtypes in the dynamic gene network can be expressed as<disp-formula id="e10">
<mml:math id="m61">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mi>u</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mi>u</mml:mi>
<mml:mi>j</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>
</p>
<p>Here, the dynamic network of the <inline-formula id="inf52">
<mml:math id="m62">
<mml:mi>p</mml:mi>
</mml:math>
</inline-formula>-th subtype can be represented as <inline-formula id="inf53">
<mml:math id="m63">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mrow>
<mml:mtext>ph</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mtext>h</mml:mtext>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi>m</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>. According to <xref ref-type="disp-formula" rid="e10">formula (10)</xref>, we can construct a completely different gene network for each cancer subtype. Our purpose of constructing the dynamic network is to hope that the DM-ESPCA model screens the gene probes which are most relevant to the corresponding cancer subtype. Then, we can use the following dynamic meta-data (DM) network as the sparse penalty:<disp-formula id="e11">
<mml:math id="m64">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">&#x2225;u&#x2225;</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">DM</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi mathvariant="italic">&#x3d;</mml:mi>
<mml:munder>
<mml:mrow>
<mml:mi mathvariant="italic">minimize</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2200;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="italic">G</mml:mi>
<mml:mi mathvariant="italic">p</mml:mi>
<mml:mi mathvariant="normal">&#x2032;</mml:mi>
</mml:msubsup>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">G</mml:mi>
<mml:mi mathvariant="italic">p</mml:mi>
</mml:msub>
<mml:mi mathvariant="italic">,support</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi mathvariant="italic">u</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2286;</mml:mo>
<mml:mi mathvariant="italic">V</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="italic">G</mml:mi>
<mml:mi mathvariant="italic">p</mml:mi>
<mml:mi mathvariant="normal">&#x2032;</mml:mi>
</mml:msubsup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:munder>
<mml:mrow>
<mml:mo>&#x7c;</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="italic">G</mml:mi>
<mml:mi mathvariant="italic">p</mml:mi>
<mml:mi mathvariant="normal">&#x2032;</mml:mi>
</mml:msubsup>
</mml:mrow>
<mml:mo>&#x7c;</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>where <inline-formula id="inf54">
<mml:math id="m65">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">&#x2032;</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is a subset of <inline-formula id="inf55">
<mml:math id="m66">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf56">
<mml:math id="m67">
<mml:mrow>
<mml:mi mathvariant="normal">V</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">&#x2032;</mml:mi>
</mml:msubsup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is a vertex (gene) set induced from the edge set <inline-formula id="inf57">
<mml:math id="m68">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf58">
<mml:math id="m69">
<mml:mrow>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mrow>
<mml:mo>&#x7c;</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
<mml:mo>&#x7c;</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the number of elements of <inline-formula id="inf59">
<mml:math id="m70">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf60">
<mml:math id="m71">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>u</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the set of indexes of nonzero elements of <inline-formula id="inf61">
<mml:math id="m72">
<mml:mi>u</mml:mi>
</mml:math>
</inline-formula>.</p>
<p>Finally, the sparse model of this article can be represented as<disp-formula id="e12">
<mml:math id="m73">
<mml:mrow>
<mml:munder>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>z</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mi>u</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mi>v</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munder>
<mml:msup>
<mml:mi>u</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mi>X</mml:mi>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>s</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>.</mml:mo>
<mml:mo>&#x2225;</mml:mo>
<mml:mi>u</mml:mi>
<mml:msub>
<mml:mo>&#x2225;</mml:mo>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
<label>(12)</label>
</disp-formula>where <inline-formula id="inf62">
<mml:math id="m74">
<mml:mi>u</mml:mi>
</mml:math>
</inline-formula> is the first PC loading, <inline-formula id="inf63">
<mml:math id="m75">
<mml:mi>v</mml:mi>
</mml:math>
</inline-formula> is the first PC, and <inline-formula id="inf64">
<mml:math id="m76">
<mml:mi>k</mml:mi>
</mml:math>
</inline-formula> is the parameter to control the number of edges selected for each cancer subtype.</p>
</sec>
<sec id="s2-3-3-3">
<title>Random Sampling Algorithm Based on the Greedy Principle</title>
<p>To solve sparse PCA methods, the key issue is how to solve a projection problem with fixed <inline-formula id="inf65">
<mml:math id="m77">
<mml:mi>v</mml:mi>
</mml:math>
</inline-formula> and <inline-formula id="inf66">
<mml:math id="m78">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>X</mml:mi>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. This is a typical NP-hard problem (<xref ref-type="bibr" rid="B24">Min et al., 2018</xref>). Many of the traditional sparse PCA methods use <inline-formula id="inf67">
<mml:math id="m79">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mi mathvariant="normal">0</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the greedy principle to screen the gene probes with the largest weights. However, the greedy principle will mislead a local optimal solution. Here, we proposed a random sampling algorithm based on the greedy principle to find a better solution of the DM-ESPCA model. We adopted the idea of a simulated annealing algorithm and add randomization to the traditional greedy algorithm. Existing research shows that introducing randomization parameters into the model can improve the local optimal solution problem of the greedy algorithm (<xref ref-type="bibr" rid="B42">Van Laarhoven and Aarts, 1987</xref>; <xref ref-type="bibr" rid="B31">Rutenbar, 1989</xref>). In addition, due to the difficulty of convergence caused by randomization parameters, we also designed an independent parameter to reduce the randomization rate during the model cycle and finally reduce the randomization rate to 0 to ensure that the model can converge. Note that we cannot guarantee that the algorithm converges to the optimal solution due to the non-convexity of this problem. Thus, we repeated our algorithm with a number of different random initial solutions.</p>
<p>In algorithm 1, <inline-formula id="inf68">
<mml:math id="m80">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mi mathvariant="normal">G</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">z,k</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the sparse projection; <inline-formula id="inf69">
<mml:math id="m81">
<mml:mrow>
<mml:mi mathvariant="normal">&#xa0;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">z,k</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">i&#x3d;1,&#x2026;,m</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> meets<disp-formula id="e13">
<mml:math id="m82">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">P</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">G</mml:mi>
<mml:mi mathvariant="italic">p</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="italic">z,k</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi mathvariant="italic">i</mml:mi>
</mml:msub>
<mml:mi mathvariant="italic">&#x3d;</mml:mi>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">z</mml:mi>
<mml:mi mathvariant="italic">i</mml:mi>
</mml:msub>
<mml:mi mathvariant="italic">,</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="italic">if</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mi mathvariant="italic">G</mml:mi>
<mml:mi mathvariant="italic">p</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi mathvariant="italic">i</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2229;</mml:mo>
<mml:mi mathvariant="italic">sample</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="italic">I,k</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2260;</mml:mo>
<mml:mo>&#x2205;</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi mathvariant="italic">0,otherwise</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(13)</label>
</disp-formula>where <inline-formula id="inf70">
<mml:math id="m83">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the edge set of the gene network corresponding to the cancer subtype <inline-formula id="inf71">
<mml:math id="m84">
<mml:mi>p</mml:mi>
</mml:math>
</inline-formula> and <inline-formula id="inf72">
<mml:math id="m85">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:msubsup>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. If gene <inline-formula id="inf73">
<mml:math id="m86">
<mml:mi mathvariant="normal">i</mml:mi>
</mml:math>
</inline-formula> is selected, <inline-formula id="inf74">
<mml:math id="m87">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mi mathvariant="normal">G</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">z,k</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">&#x3d;</mml:mi>
<mml:msub>
<mml:mi mathvariant="normal">z</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>; otherwise, <inline-formula id="inf75">
<mml:math id="m88">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">P</mml:mi>
<mml:mi mathvariant="script">G</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf76">
<mml:math id="m89">
<mml:mi>k</mml:mi>
</mml:math>
</inline-formula> represents the number of edges expected to be retained. <inline-formula id="inf77">
<mml:math id="m90">
<mml:mi>&#x3c9;</mml:mi>
</mml:math>
</inline-formula> is a parameter that controls the random ratio. For example, if we set the parameter <inline-formula id="inf78">
<mml:math id="m91">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>100</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf79">
<mml:math id="m92">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, then the algorithm will keep 120 edges with the largest weight in each cycle and randomly select 100 of them as the result.</p>
<p>Finally, we use <xref ref-type="disp-formula" rid="e14">formulas 14</xref>, <xref ref-type="disp-formula" rid="e15">15</xref> to update vectors <inline-formula id="inf80">
<mml:math id="m93">
<mml:mi>u</mml:mi>
</mml:math>
</inline-formula> and <inline-formula id="inf81">
<mml:math id="m94">
<mml:mi>v</mml:mi>
</mml:math>
</inline-formula> until the algorithm convergence:<disp-formula id="equ1">
<mml:math id="m95">
<mml:mrow>
<mml:mi mathvariant="italic">u&#x3d;Xv</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ2">
<mml:math id="m96">
<mml:mrow>
<mml:mi mathvariant="italic">where</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>v</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mi mathvariant="italic">&#x3d;</mml:mi>
<mml:msup>
<mml:mi mathvariant="italic">X</mml:mi>
<mml:mi mathvariant="italic">T</mml:mi>
</mml:msup>
<mml:mi mathvariant="italic">u</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="e14">
<mml:math id="m97">
<mml:mrow>
<mml:mi mathvariant="italic">u&#x2190;</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:mi mathvariant="italic">,</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="italic">where</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mi mathvariant="italic">&#x3d;</mml:mi>
<mml:msub>
<mml:mi mathvariant="italic">P</mml:mi>
<mml:mi mathvariant="italic">G</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="italic">z,k</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="italic">and</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="italic">z&#x3d;Xv</mml:mi>
</mml:mrow>
</mml:math>
<label>(14)</label>
</disp-formula>
<disp-formula id="e15">
<mml:math id="m98">
<mml:mrow>
<mml:mi mathvariant="italic">v</mml:mi>
<mml:mi mathvariant="italic">&#x2190;</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="italic">v</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="italic">v</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:mi mathvariant="italic">,where</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="italic">v</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mi mathvariant="italic">&#x3d;</mml:mi>
<mml:msup>
<mml:mi mathvariant="italic">X</mml:mi>
<mml:mi mathvariant="italic">T</mml:mi>
</mml:msup>
<mml:mi mathvariant="italic">u</mml:mi>
</mml:mrow>
</mml:math>
<label>(15)</label>
</disp-formula>
</p>
<p>
<statement content-type="algorithm" id="alg1">
<label>Algorithm 1</label>
<p>Random sampling algorithm based on the greedy principle sparse projection for the dynamic network<disp-formula id="equ3">
<mml:math id="m99">
<mml:mrow>
<mml:mi mathvariant="normal">Require:&#xa0;X</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">m&#xd7;n</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mi mathvariant="normal">,&#x3bd;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">n&#xd7;1</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mi mathvariant="normal">,parameter</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">k,&#x3c9;,&#x3c1;,</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ4">
<mml:math id="m100">
<mml:mrow>
<mml:mi mathvariant="normal">edge</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">set</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">&#x3d;</mml:mi>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">1</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">2</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">,L</mml:mi>
<mml:msub>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ5">
<mml:math id="m101">
<mml:mrow>
<mml:mi mathvariant="normal">1:Z&#x3d;X&#x3bd;</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ6">
<mml:math id="m102">
<mml:mrow>
<mml:mi mathvariant="normal">2:for</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">any</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">weight</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">of</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">edge</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">in&#xa0;</mml:mi>
<mml:msub>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">&#xa0;do</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ7">
<mml:math id="m103">
<mml:mrow>
<mml:mi mathvariant="normal">3:</mml:mi>
<mml:msubsup>
<mml:mi mathvariant="normal">w</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mi mathvariant="normal">&#x3d;</mml:mi>
<mml:msqrt>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">Pi</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mi mathvariant="normal">&#x396;</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">2</mml:mi>
</mml:msubsup>
<mml:mi mathvariant="normal">&#x2b;</mml:mi>
<mml:msub>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">Pj</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mi mathvariant="normal">&#x396;</mml:mi>
<mml:mi mathvariant="normal">j</mml:mi>
<mml:mi mathvariant="normal">2</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:msqrt>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">&#x23;Generate</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">dynamic</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">network</mml:mi>
<mml:mi mathvariant="normal">.</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ8">
<mml:math id="m104">
<mml:mrow>
<mml:mi mathvariant="normal">4:update&#xa0;</mml:mi>
<mml:msubsup>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">Pn</mml:mi>
</mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mi mathvariant="normal">&#x3d;&#xa0;</mml:mi>
<mml:msubsup>
<mml:mi mathvariant="normal">w</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ9">
<mml:math id="m105">
<mml:mrow>
<mml:mi mathvariant="normal">5:end</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">for</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ10">
<mml:math id="m106">
<mml:mrow>
<mml:mi mathvariant="normal">6:Let</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">nor</mml:mi>
<mml:msubsup>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">DM</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mi mathvariant="normal">&#x3d;</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">1</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
<mml:mi mathvariant="normal">,&#x2026;,</mml:mi>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ11">
<mml:math id="m107">
<mml:mrow>
<mml:mi mathvariant="normal">7:I&#x3d;supp</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">nor</mml:mi>
<mml:msubsup>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">DM</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mi mathvariant="normal">,&#xa0;</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">1&#x2b;&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mi mathvariant="normal">&#xd7;k</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mi mathvariant="normal">&#x23;Extract&#xa0;</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">1&#x2b;&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mi mathvariant="normal">&#xd7;k</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">edges</mml:mi>
<mml:mi mathvariant="normal">.</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ12">
<mml:math id="m108">
<mml:mrow>
<mml:mi mathvariant="normal">8:</mml:mi>
<mml:msub>
<mml:mi mathvariant="normal">J</mml:mi>
<mml:mi mathvariant="normal">k</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">&#x3d;sample</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">I,k</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">&#x23;Randomly</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">select</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">k</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">edges</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">from</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">I</mml:mi>
<mml:mi mathvariant="normal">.</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ13">
<mml:math id="m109">
<mml:mrow>
<mml:mi mathvariant="normal">9:if&#xa0;</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">&#x3c9;&#x3e;0</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">then&#xa0;</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">&#x3c9;&#x3d;</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">&#x3c9;-&#x3c1;</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">&#x23;Reduce</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">random</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">rate</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ14">
<mml:math id="m110">
<mml:mrow>
<mml:mi mathvariant="normal">10:</mml:mi>
<mml:msub>
<mml:mi mathvariant="normal">V</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">&#xa0;</mml:mi>
<mml:msubsup>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:msub>
<mml:mi mathvariant="normal">&#x3d;V</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ15">
<mml:math id="m111">
<mml:mrow>
<mml:mi mathvariant="normal">11:for</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">any</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">gene</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">in</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mi mathvariant="normal">V</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">&#xa0;</mml:mi>
<mml:msubsup>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:msub>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">do</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ16">
<mml:math id="m112">
<mml:mrow>
<mml:mi mathvariant="normal">12:&#xa0;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="normal">u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
<mml:mi mathvariant="normal">&#x3d;</mml:mi>
<mml:msub>
<mml:mi mathvariant="normal">z</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ17">
<mml:math id="m113">
<mml:mrow>
<mml:mi mathvariant="normal">13:end&#xa0;for</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ18">
<mml:math id="m114">
<mml:mrow>
<mml:mi mathvariant="normal">14:u&#x3d;</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="normal">u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="normal">u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ19">
<mml:math id="m115">
<mml:mrow>
<mml:mn>15</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi mathvariant="bold-italic">return</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>u</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">P</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi mathvariant="script">P</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>In order to ensure the convergence of the algorithm, when the model completes the edge sparse projection, we use the parameter <inline-formula id="inf82">
<mml:math id="m116">
<mml:mrow>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>&#x3c1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to reduce the randomness of the model, that is, <inline-formula id="inf83">
<mml:math id="m117">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#x3e;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>&#x3c1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Furthermore, the DM-ESPCA model can be applied to generate multiple PCs and their PC loadings. Specifically, given the current PCs, we adopted Min&#x2019;s model to compute the next PC and its loading (<xref ref-type="bibr" rid="B24">Min et al., 2018</xref>).</p>
</statement>
</p>
</sec>
</sec>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<p>The experiments are divided into two steps. First, we use three sparse PCA methods including DM-ESPCA, ESPCA, and SPCA models to perform unsupervised sparse PCA on the cancer data sets. This step will allow each model to screen the subset of the potential target genes for each cancer subtype. We adopted three indicators including heat map, the cluster results, and <italic>p</italic>-value to evaluate the gene subset screen by each model. We also conducted a bio-enrichment analysis (<xref ref-type="bibr" rid="B50">Zhou et al., 2019</xref>) to count the key biological pathways corresponding to these gene subsets, such as the GO biological process (GO-BP), KEGG, and so forth, to determine whether these gene subsets are related to the cancer subtypes.</p>
<p>In order to further compare these gene subsets screened by the three sparse PCA methods, we used all samples based on the gene subsets to build four machine learning classification models, such as the K-Nearest Neighbor (KNN) model, the Support Vector Machines (SVM), the Logistic Regression, and the Random Forest model (<xref ref-type="bibr" rid="B15">Hearst et al., 1998</xref>; <xref ref-type="bibr" rid="B20">Liaw and Wiener, 2002</xref>; <xref ref-type="bibr" rid="B28">Peterson, 2009</xref>). In addition, we also built four machine learning models based on all genes, which was performed to compare whether the DM-ESPCA model is better than the classic supervised learning model in classification tasks. In sections 3.1&#x2013;3.3, we only illustrate the results of the KNN model, and the results of other models are in the supplementary materials. Four classic statistical indicators, including precision, recall, F1-score, and accuracy, are used to evaluate the classification results. All machine learning experiments use the 5-fold cross-validation approach, and the final results are the averages of five runs. (The detail of indicators is in the <xref ref-type="sec" rid="s10">Supplementary Materials</xref>.)</p>
<sec id="s3-1">
<title>Application to the BCI Data Set</title>
<p>In <xref ref-type="fig" rid="F3">Figure 3A</xref> of the heat map analysis, we can find that the DM-ESPCA model can clearly distinguish the four breast cancer subtypes with clear boundaries. However, the gene probes screened by the ESPCA and SPCA models could not distinguish these four subtypes well (<xref ref-type="sec" rid="s10">Supplementary Figure S2, S3</xref>). <xref ref-type="table" rid="T2">Table 2</xref> summarizes these clustering results, where the clustering accuracy of the DM-ESPCA model reached 82.3%, which is 14.61% higher than the results of the ESPCA model and 21.6% higher than that of the results of the SPCA model (<xref ref-type="sec" rid="s10">Supplementary Table S1</xref>). These results showed that the DM-ESPCA model had a relatively strong distinguishing ability for the four subtypes of breast cancer, especially in Luminal B subtypes. In addition, according to the <italic>p</italic>-values shown in <xref ref-type="fig" rid="F7">Figure 7A</xref> and <xref ref-type="sec" rid="s10">Supplementary Figure S10</xref>, the performance of the DM-ESPCA model was significantly better than that of the ESPCA and SPCA models in the correlation of Luminal A subtype. Moreover, the average <italic>p</italic>-values of select genes in all subtypes are very low, which means that the results of our proposed model are highly related to breast cancer (<xref ref-type="sec" rid="s10">Supplementary Table S2</xref>).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Heat maps of the DM-ESPCA model. <bold>(A)</bold> Result of the BCI data set. <bold>(B)</bold> Result of the BCII data set. <bold>(C)</bold> Result of the GC data set. The row is the gene probs; different color blocks of rows indicate genes selected by different PC loadings. The column is the samples. The color of each block in the heat maps is the expression value of the genes.</p>
</caption>
<graphic xlink:href="fgene-13-869906-g003.tif"/>
</fig>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Clustering results obtained by the three sparse PCA methods.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="center">DM-ESPCA (%)</th>
<th align="center">ESPCA (%)</th>
<th align="center">SPCA (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">BCI</td>
<td align="char" char=".">
<bold>82.30</bold>
</td>
<td align="char" char=".">67.69</td>
<td align="char" char=".">60.70</td>
</tr>
<tr>
<td align="left">BCII</td>
<td align="char" char=".">
<bold>82.35</bold>
</td>
<td align="char" char=".">75.16</td>
<td align="char" char=".">59.87</td>
</tr>
<tr>
<td align="left">GC</td>
<td align="char" char=".">
<bold>82.86</bold>
</td>
<td align="char" char=".">77.14</td>
<td align="char" char=".">78.57</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In order to further verify the gene screening ability of the DM-ESPCA model, we conducted a bio-enrichment analysis. It can be seen from <xref ref-type="table" rid="T3">Table 3</xref> that the DM-ESPCA model can find genes related to breast cancer in all four subtypes, but the ESPCA and SPCA models can only be found in three subtypes. From <xref ref-type="fig" rid="F4">Figures 4</xref>, <xref ref-type="fig" rid="F5">5</xref>, we can see that the DM-ESPCA model can find 1,286 biological pathways in the GO-BP and KEGG data sets. These results are much better than that of the ESPCA and SPCA models.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Number of PCs that can find gene probes related to the target cancer for each model.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="center">DM-ESPCA</th>
<th align="center">ESPCA</th>
<th align="center">SPCA</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">BCI</td>
<td align="center">4</td>
<td align="center">3</td>
<td align="center">3</td>
</tr>
<tr>
<td align="left">BCII</td>
<td align="center">4</td>
<td align="center">2</td>
<td align="center">3</td>
</tr>
<tr>
<td align="left">GC</td>
<td align="center">3</td>
<td align="center">0</td>
<td align="center">0</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Pathway numbers with screened genes of GO, KEGG, and Reactome in the bio-enrichment analysis; <bold>(A)</bold> number of pathways in the BCI data set; <bold>(B)</bold> number of pathways in the BCII data set; <bold>(C)</bold> number of pathways in the GC data set. The blue bar is the DM-ESPCA model, the orange bar is the ESPCA model, and the gray one is the SPCA model.</p>
</caption>
<graphic xlink:href="fgene-13-869906-g004.tif"/>
</fig>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Results of the DisGeNET dataset and PPI pathways of the Basal subtype in the BCI dataset; <bold>(A)</bold> relationship between the diseases and gene selected by the DM-ESPCA model of the Basal subtype in the BCI dataset.The blue bar shows the z-score of each gene.Data collected from the DisGeNET dataset. <bold>(B)</bold> KeyPPI pathways of part of the gene selected by the DM-ESPCA data set.</p>
</caption>
<graphic xlink:href="fgene-13-869906-g005.tif"/>
</fig>
<p>Among the results of the enrichment analysis, the basal subtype results of the DM-ESPCA model are particularly encouraging. First, in the PPI networks, it found multiple key target protein sites. Among them, ESR1, NRIP1, FOXA1, RARA, and GATA3 are highly correlated with the gene pathway R-HSA-9018519 of estrogen-dependent gene expression (<xref ref-type="fig" rid="F6">Figure 6B</xref>). The secretion of estrogen is one of the important causes of breast cancer. We also found that the z-scores of the aforementioned gene probes are generally high (<xref ref-type="sec" rid="s10">Supplementary Figure S8</xref>). Next, in the DisGeNET set, the potential target gene probes screened by the DM-ESPCA model are related to 32 known breast cancer disease signatures (<xref ref-type="fig" rid="F6">Figure 6A</xref>). Among them, the gene probes ARHGAP1, ESR1, FBP1, GATA3, FOXA1, PDCD6IP, AR, FASN, RARA, and TMED7 are directly related to the basal-like breast carcinoma and HER2-negative breast cancer with the data set numbers C3642347 and C4733095 (<xref ref-type="sec" rid="s10">Supplementary Table S3</xref>). Finally, the enrichment analysis results of the PaGenBase data set show that the gene set found by the DM-ESPCA model is highly correlated with breast cells (<xref ref-type="sec" rid="s10">Supplementary Table S4</xref>). In general, the results of the gene enrichment analysis clearly prove that DM-ESPCA has a strong ability to select target genes of breast cancer subtypes.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Functional pathways collected from the BCI data set Luminal A subtype; <bold>(A)</bold> results of GO-BP in the DMESPCA model; <bold>(B)</bold> results of GO-BP in the ESPCA model; and <bold>(C)</bold> results of GO-BP in the SPCA model.</p>
</caption>
<graphic xlink:href="fgene-13-869906-g006.tif"/>
</fig>
<p>The gene subset selected by the DM-ESPCA model also achieved the best classification results; the accuracy reached 97%, the precision reached 98%, the recall reached 97%, and the F1-score reached 97% (<xref ref-type="fig" rid="F7">Figure 7B</xref>, <xref ref-type="sec" rid="s10">Supplementary Table S5</xref>). Simultaneously, the classification accuracy based on the gene subset selected by the ESPCA model and its precision, recall, and F1-score only reached 77, 79, 76, and 76%, respectively. The classification accuracy based on the gene subset selected by the SPCA model and its precision, recall, and F1-score only reached 75, 74, 71, and 74%, respectively. It is worth noting that even if we use all genes to build four supervised machine learning models, the best result of precision, recall, and F1-score only reached 85, 86, 85, and 85% (Logistic Regression model), which is much lower than the result of the DM-ESPCA model.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Boxplots and classification comprehensive indicators of the BCI data set; <bold>(A)</bold> <italic>p</italic>-values of selected genes in all subtypes. <bold>(B)</bold> Results of KNN in three sparse PCA methods and the use of all genes.</p>
</caption>
<graphic xlink:href="fgene-13-869906-g007.tif"/>
</fig>
<p>In summary, these results demonstrated that the DM-ESPCA model can identify more biologically relevant gene sets than the ESPCA and SPCA models. In classification tasks, the DM-ESPCA model is better than ESPCA, SPCA, and classic supervised learning models. From the perspective of model construction, it is expected that the DM-ESPCA model can obtain better results than ESPCA and SPCA in heat map, cluster analysis, correlation analysis, enrichment analysis, and classification experiments. Because the dynamic network takes known cancer subtype classification information as prior knowledge, this enables the DM-ESPCA model to select cancer targets that are more relevant to the corresponding cancer subtype. The screening of meta-data further alleviates the problem of sample quality in the data, and the random sampling algorithm based on the greedy principle improves the local optimal solution problem of the traditional greedy algorithm. In addition, we believe that the dimensional challenges and overfitting problems of the data prevent the machine learning model (use all gene probes) from achieving a better performance, which is the same point of view as existing research works.</p>
</sec>
<sec id="s3-2">
<title>Application to the BCII Data Set</title>
<p>In order to further verify the stability of the DM-ESPCA model in the same type but different batches of cancer subtype data sets, we also used the BCII data set to conduct the experiments, which showed similar results compared with the BCI data set. According to <xref ref-type="fig" rid="F3">Figure 3B</xref>, the DM-ESPCA model could distinguish four breast cancer subtypes well, and the boundary corresponding to each subtype was very clear. In contrast, the heat map results of the ESPCA and SPCA models were worse in the BCII data set, and they were difficult to judge the boundary of the subtype (<xref ref-type="sec" rid="s10">Supplementary Figure S4, S5</xref>). In <xref ref-type="table" rid="T2">Table 2</xref>, the cluster accuracy of the DM-ESPCA model reached 82.3%; however, the cluster accuracies of the ESPCA and SPCA models only reached 75.1 and 59.8%, respectively. Similar to the results in the BCI data set, the Lumina B subtype was difficult to distinguish; the DM-ESPCA model could relatively accurately divide all samples into four subtypes, including the Lumina B subtype. Neither the ESPCA model nor the SPCA model could cluster Lumina B subtypes well (<xref ref-type="sec" rid="s10">Supplementary Table S6</xref>). Besides, in Supplementary Fig.11, the DM-ESPCA model outperformed the ESPCA and the SPCA models in <italic>p</italic>-values, especially the correlation of a comprehensive Luminal A subtype (<xref ref-type="sec" rid="s10">Supplementary Table S7</xref>). These meant that the genetic points screened by the DM-ESPCA model had a higher correlation with cancer subtypes, which was more conducive to the analysis by biological researchers.</p>
<p>An enrichment analysis showed that the DM-ESPCA model selected gene probes containing the largest number of biological pathways (<xref ref-type="fig" rid="F4">Figure 4B</xref>). In addition, the DM-ESPCA model can find gene probes known to be related to breast cancer diseases in the DisGeNET set among all four principal components (<xref ref-type="table" rid="T3">Table 3</xref>). In comparison, the ESPCA model can only find genes related to breast cancer in two principal components, while the SPCA model can find genes related to breast cancer in three principal components. Especially in the Luminal B subtype, the DM-ESPCA model can find 13 gene probes related to eight breast cancer disease entries which show a very high correlation with breast cancer (<xref ref-type="sec" rid="s10">Supplementary Figure S9</xref>).</p>
<p>Finally, based on Supplementary Fig.12, the optimal classification results were obtained by the KNN method based on the gene subset selected by the DM-ESPCA model. Its accuracy, precision, recall, and F1-score reached 90, 90, 89, and 88%, respectively. In comparison, these four classification indicators of the model based on the gene subset selected by the ESPCA model could only reach 86, 86, 80, and 80%, respectively, while these four classification indicators of the model based on the gene subset selected by the SPCA model could only reach 82, 82, 80, and 80%, respectively (<xref ref-type="sec" rid="s10">Supplementary Table S8</xref>). The best results of precision, recall, and F1-score for the supervised machine learning model which used all genes only reached 85, 87, 85, and 85% (Logistic Regression model), which is lower than the result of the DM-ESPCA model, 5, 3, 4, and 3%.</p>
<p>Based on the results of the BCII data set, we can see that in the same cancer subtype, but in different data batches, the performance of the DM-ESPCA model was very stable.</p>
</sec>
<sec id="s3-3">
<title>Application to the GC DataSet</title>
<p>To verify the applicability of the DM-ESPCA model in different cancer data, we used a gastric cancer data set for experimentation. Based on the result of the heat map (<xref ref-type="fig" rid="F3">Figure 3C</xref>, <xref ref-type="sec" rid="s10">Supplementary Figure S6, S7</xref>), the DM-ESPCA model performed well, especially in subtypes Invasive and Metabolic. In <xref ref-type="table" rid="T1">Table 1</xref>, the clustering accuracy of the DM-ESPCA model reached 84.23%. Compared with the ESPCA and SPCA models, the clustering accuracy of the DM-ESPCA model increased by 9 and 6%, respectively (<xref ref-type="sec" rid="s10">Supplementary Table S9</xref>). Meanwhile, based on Supplementary Fig.13, the <italic>p</italic>-values of the DM-ESPCA model have had significant improvements compared with other models (<xref ref-type="sec" rid="s10">Supplementary Table S10</xref>).</p>
<p>In addition, it can be seen from <xref ref-type="fig" rid="F3">Figure 3C</xref>, the DM-ESPCA model has more number of GO, KEGG, and Reactome pathways than the comparison methods in bio-enrichment analysis. In particular, the DM-ESPCA model is the only one that can find genetic probes related to all subtypes of gastric cancer. However, neither ESPCA nor SPCA can find genes related to gastric cancer in the three subtypes (<xref ref-type="table" rid="T3">Table 3</xref>).</p>
<p>Based on Supplementary Fig.14, the optimal classification results were obtained by the KNN method based on the gene subset selected by the DM-ESPCA model. Its accuracy, precision, recall, and F1-score reached 95, 96, 95, and 95%, respectively. In comparison, these four classification indicators of the model based on the gene subset selected by the ESPCA model could only reach 76, 72, 77, and 73%, respectively. While these four classification indicators of the model based on the gene subset selected by the SPCA model could only reach 86, 86, 86, and 86%, respectively (<xref ref-type="sec" rid="s10">Supplementary Table S11</xref>). The best results of precision, recall, and F1-score for the supervised machine learning model which used all genes only reached 90, 93, 90, and 91%, respectively (Logistic Regression model), which is also lower than the result of the DM-ESPCA model. In summary, whether in the same cancer data sets with different batches or in different cancer data sets, the DM-ESPCA model performed better than the existing sparse PCA methods. Therefore, we believe that the DM-ESPCA model could reliably and stably screen the gene probes corresponding to the cancer subtypes.</p>
</sec>
<sec id="s3-4">
<title>Ablation Experiment</title>
<p>In order to further verify the influence of three main modules of DM-ESPCA, which include the random sampling algorithm based on the greedy principle, the dynamic network, and the meta-data selection module on model performance, we performed ablation experiments based on the BCI data set (<xref ref-type="table" rid="T4">Table 4</xref>, <xref ref-type="sec" rid="s10">Supplementary Table S12</xref>). As shown in <xref ref-type="table" rid="T4">Table 4</xref>, non-<inline-formula id="inf84">
<mml:math id="m118">
<mml:mi>&#x3c9;</mml:mi>
</mml:math>
</inline-formula> refers to the experimental results with the random sampling algorithm based on the greedy principle module removed (use the greedy algorithm instead). Non-DM refers to the experimental results with dynamic network modules removed. Non-Meta refers to experimental results with meta-data selection modules removed. We use the results of clustering, accuracy, precision, recall, and F1-score as evaluation metrics. The classification experiments use the KNN method as the classifier because the KNN method performs the best on the three real data sets. The experimental results show that the three main modules proposed in this article all have a significant impact on the results. Among them, the removal of the meta-data selection module has the greatest impact on the results. After removing the meta-data, the clustering accuracy of the model dropped to 65.38% and the result of classification accuracy dropped to 79%. The experimental results mean that there are indeed sample quality issues and data noise in the data set and that it can be improved by incorporating the meta-data selection module. The dynamic network also has a great influence on the model. After removing the dynamic network module, the clustering accuracy of the DM-ESPCA model can only reach 66.15%, which shows that dynamic networks can improve model performance. In addition, the experimental results show that the random sampling algorithm based on the greedy principle can effectively improve the results of the model and alleviate the local optimal solution problem of the greedy algorithm.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Result of the ablation experiment.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="center">Clustering (%)</th>
<th align="center">Accuracy (%)</th>
<th align="center">Recall (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">DM-ESPCA</td>
<td align="char" char=".">82.30</td>
<td align="center">97</td>
<td align="center">97</td>
</tr>
<tr>
<td align="left">Non-<inline-formula id="inf85">
<mml:math id="m119">
<mml:mi>&#x3c9;</mml:mi>
</mml:math>
</inline-formula>
</td>
<td align="char" char=".">80.07</td>
<td align="center">82</td>
<td align="center">82</td>
</tr>
<tr>
<td align="left">Non-DM</td>
<td align="char" char=".">66.15</td>
<td align="center">87</td>
<td align="center">87</td>
</tr>
<tr>
<td align="left">Non-Meta</td>
<td align="char" char=".">65.38</td>
<td align="center">79</td>
<td align="center">78</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<p>Since the beginning of the 21st century, with the development of the gene sequencing technology, researchers have discovered that the same cancer can be divided into different subtypes, which also explains that the same drug is only effective for some cancer patients but not for other patients. Therefore, how to find target genes corresponding to cancer subtypes has gradually become an important task of cancer research.</p>
<p>The traditional screening models for potential targets of cancer subtypes have three main problems. The first problem is that no known subtype classification information can be used. In this study, we have shown that if researchers can integrate the known subtype classification information as prior knowledge to carry out cancer subtype screening models and establish a dynamic gene network, then the screening ability of potential cancer subtype targets of the model can be greatly enhanced. The second is that the experiment&#x2019;s sample quality is uneven, and low-quality samples will affect the final results of analyses. In this article, we used the idea of meta-learning to screen high-quality samples. The third point is that most of the existing models adopt the greedy principle, which will make the model quickly fall into a local optimum. We designed a new random sampling algorithm to improve the model, which may find better target genes.</p>
<p>Based on the aforementioned ideas, this article proposes the DM-ESPCA model, which is based on meta-learning, the dynamic gene network, and sparse PCA to screen the corresponding potential target gene probes for each cancer subtype. The bio-enrichment analysis shows that the DM-ESPCA model can directly find gene probes related to the corresponding cancer subtype. Moreover, all indicators indicate that the DM-ESPCA model can reveal more modules related to biology. Even in the task of classification of cancer subtypes, the DM-ESPCA model is superior to the existing supervised learning model. In summary, we believe that the DM-ESPCA model is a good extension of the PCA-based methods. This model can provide an effective tool for researchers to find target genes corresponding to cancer subtypes.</p>
<p>Although the experiment has achieved good results, the DM-ESPCA model can still be extended. We have proved that the idea of meta-learning reduces the errors caused by the noise data. However, the results of the gastric cancer data set are not very satisfactory. The reason may mean that there is still noise in the meta-data. We would consider using more powerful statistical methods to filter the meta-data. In addition, the random sampling algorithm based on the greedy principle proposed in this article can also be further improved. There are many optimization principles for NP-hard problems that can be considered. This may further improve the feature selection ability of the proposed model. In addition, it is worth noting that there are many multi-omics cancer subtype target screening models. Compared with single omics, multi-omics data can provide different views of the same batch of samples, which may lead to new and interesting biological discoveries. In theory, the DM-ESPCA model can be extended to a multi-omics model. However, how to solve the multi-omics joint sparse PCA problem still needs to be further discussed.</p>
</sec>
</body>
<back>
<sec id="s5">
<title>Data Availability Statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s10">Supplementary Material</xref>, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s6">
<title>Author Contributions</title>
<p>RM, XD, and YL contributed to conception and design of the study. X-YL organized the database. RM and XD performed the statistical analysis. RM, XD, S-LL, X-YM, and S-LX wrote the first draft of the manuscript. QD, JC, SL, and KY wrote sections of the manuscript. All authors contributed to manuscript revision, read, and approved the submitted version.</p>
</sec>
<sec id="s7">
<title>Funding</title>
<p>The authors wish to thank editors and reviewers. This work was supported in part by the Macau Science and Technology Development Funds Grant No.0158/2019/A3 and 0056/2020/ AFJ from the Macau Special Administrative Region of the People&#x2019;s Republic of China and the Key Project for University of Educational Commission of Guangdong Province of China Funds (Natural, Grant No. 2019GZDXM005)</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations or those of the publisher, the editors, and the reviewers. Any product that may be evaluated in this article or claim that may be made by its manufacturer is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s10">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2022.869906/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2022.869906/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table2.XLSX" id="SM1" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table3.XLSX" id="SM2" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table9.XLSX" id="SM3" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table6.XLSX" id="SM4" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table11.XLSX" id="SM5" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table4.XLSX" id="SM6" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table1.XLSX" id="SM7" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table10.XLSX" id="SM8" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table12.XLSX" id="SM9" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table5.XLSX" id="SM10" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table7.XLSX" id="SM11" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table8.XLSX" id="SM12" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="DataSheet1.DOCX" id="SM13" mimetype="application/DOCX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Banerji</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Cibulskis</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Rangel-Escareno</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Brown</surname>
<given-names>K. K.</given-names>
</name>
<name>
<surname>Carter</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Frederick</surname>
<given-names>A. M.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Sequence Analysis of Mutations and Translocations across Breast Cancer Subtypes</article-title>. <source>Nature</source> <volume>486</volume>, <fpage>405</fpage>&#x2013;<lpage>409</lpage>. <pub-id pub-id-type="doi">10.1038/nature11154</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Calon</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lonardo</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Berenguer-Llergo</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Espinet</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Hernando-Momblona</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Iglesias</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Stromal Gene Expression Defines Poor-Prognosis Subtypes in Colorectal Cancer</article-title>. <source>Nat. Genet.</source> <volume>47</volume>, <fpage>320</fpage>&#x2013;<lpage>329</lpage>. <pub-id pub-id-type="doi">10.1038/ng.3225</pub-id> </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cancello</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Maisonneuve</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Rotmensz</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Viale</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Mastropasqua</surname>
<given-names>M. G.</given-names>
</name>
<name>
<surname>Pruneri</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>Prognosis and Adjuvant Treatment Effects in Selected Breast Cancer Subtypes of Very Young Women</article-title>. <source>Ann. Oncol.</source> <volume>21</volume>, <fpage>1974</fpage>&#x2013;<lpage>1981</lpage>. <pub-id pub-id-type="doi">10.1093/annonc/mdq072</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Carlson</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Falcon</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Pages</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>hgu133plus2. Db: Affymetrix Human Genome U133 Plus 2.0 Array Annotation Data (Chip Hgu133plus2)</article-title>. <source>R. Package Version</source> <volume>3</volume>. </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cooper</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Chim</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Durand</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Ceritinib</article-title>. <source>Ann. Pharmacother.</source> <volume>49</volume>, <fpage>107</fpage>&#x2013;<lpage>112</lpage>. <pub-id pub-id-type="doi">10.1177/1060028014553619</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Crew</surname>
<given-names>K. D.</given-names>
</name>
<name>
<surname>Neugut</surname>
<given-names>A. I.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Epidemiology of Gastric Cancer</article-title>. <source>Wjg</source> <volume>12</volume>, <fpage>354</fpage>. <pub-id pub-id-type="doi">10.3748/wjg.v12.i3.354</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dai</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Bai</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhan</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Breast Cancer Intrinsic Subtype Classification, Clinical Use and Future Trends</article-title>. <source>Am. J. Cancer Res.</source> <volume>5</volume>, <fpage>2929</fpage>&#x2013;<lpage>2943</lpage>. <pub-id pub-id-type="doi">10.1534/g3.114.014894</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>De Cecco</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Nicolau</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Giannoccaro</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Daidone</surname>
<given-names>M. G.</given-names>
</name>
<name>
<surname>Bossi</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Locati</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Head and Neck Cancer Subtypes with Biological and Clinical Relevance: Meta-Analysis of Gene-Expression Data</article-title>. <source>Oncotarget</source> <volume>6</volume>, <fpage>9627</fpage>&#x2013;<lpage>9642</lpage>. <pub-id pub-id-type="doi">10.18632/oncotarget.3301</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Deeks</surname>
<given-names>E. D.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Ceritinib: a Review in ALK-Positive Advanced NSCLC</article-title>. <source>Targ Oncol.</source> <volume>11</volume>, <fpage>693</fpage>&#x2013;<lpage>700</lpage>. <pub-id pub-id-type="doi">10.1007/s11523-016-0460-7</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>DeSantis</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Bryan</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Jemal</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Breast Cancer Statistics, 2013</article-title>. <source>CA A Cancer J. Clinicians</source> <volume>64</volume>, <fpage>52</fpage>&#x2013;<lpage>62</lpage>. <pub-id pub-id-type="doi">10.3322/caac.21203</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fan</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Strasser-Weippl</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.-J.</given-names>
</name>
<name>
<surname>St Louis</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Finkelstein</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>K.-D.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Breast Cancer in China</article-title>. <source>Lancet Oncol.</source> <volume>15</volume>, <fpage>e279</fpage>&#x2013;<lpage>e289</lpage>. <pub-id pub-id-type="doi">10.1016/s1470-2045(13)70567-9</pub-id> </citation>
</ref>
<ref id="B12">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Finn</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Abbeel</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Levine</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Model-agnostic Meta-Learning for Fast Adaptation of Deep Networks</article-title>,&#x201d; in <conf-name>International conference on machine learning</conf-name> (<publisher-name>PMLR</publisher-name>), <fpage>1126</fpage>&#x2013;<lpage>1135</lpage>. </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fessler</surname>
<given-names>E.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>DeepCC: a Novel Deep Learning-Based Framework for Cancer Molecular Subtype Classification</article-title>. <source>Oncogenesis</source> <volume>8</volume>, <fpage>44</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1038/s41389-019-0157-8</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hartgrink</surname>
<given-names>H. H.</given-names>
</name>
<name>
<surname>Jansen</surname>
<given-names>E. P.</given-names>
</name>
<name>
<surname>van Grieken</surname>
<given-names>N. C.</given-names>
</name>
<name>
<surname>van de Velde</surname>
<given-names>C. J.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Gastric Cancer</article-title>. <source>The Lancet</source> <volume>374</volume>, <fpage>477</fpage>&#x2013;<lpage>490</lpage>. <pub-id pub-id-type="doi">10.1016/s0140-6736(09)60617-6</pub-id> </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hearst</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Dumais</surname>
<given-names>S. T.</given-names>
</name>
<name>
<surname>Osuna</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Platt</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Scholkopf</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>1998</year>). <article-title>Support Vector Machines</article-title>. <source>IEEE Intell. Syst. Their Appl.</source> <volume>13</volume>, <fpage>18</fpage>&#x2013;<lpage>28</lpage>. <pub-id pub-id-type="doi">10.1109/5254.708428</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Houssami</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Macaskill</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>von Minckwitz</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Marinovich</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>Mamounas</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Meta-analysis of the Association of Breast Cancer Subtype and Pathologic Complete Response to Neoadjuvant Chemotherapy</article-title>. <source>Eur. J. Cancer</source> <volume>48</volume>, <fpage>3342</fpage>&#x2013;<lpage>3354</lpage>. <pub-id pub-id-type="doi">10.1016/j.ejca.2012.05.023</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Bismeijer</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zwart</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wessels</surname>
<given-names>L. F. A.</given-names>
</name>
<name>
<surname>Vis</surname>
<given-names>D. J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Genomic Data Integration by WON-PARAFAC Identifies Interpretable Factors for Predicting Drug-Sensitivity <italic>In Vivo</italic>
</article-title>. <source>Nat. Commun.</source> <volume>10</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1038/s41467-019-13027-2</pub-id> </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lim</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Sung</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Cancer Subtype Classification and Modeling by Pathway Attention and Propagation</article-title>. <source>Bioinformatics</source> <volume>36</volume>, <fpage>3818</fpage>&#x2013;<lpage>3824</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa203</pub-id> </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lei</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>I. B.</given-names>
</name>
<name>
<surname>Das</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Zouridis</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Pattison</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Identification of Molecular Subtypes of Gastric Cancer with Different Responses to PI3-Kinase Inhibitors and 5-fluorouracil</article-title>. <source>Gastroenterology</source> <volume>145</volume>, <fpage>554</fpage>&#x2013;<lpage>565</lpage>. <pub-id pub-id-type="doi">10.1053/j.gastro.2013.05.010</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liaw</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wiener</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Classification and Regression by randomForest</article-title>. <source>R. News</source> <volume>2</volume>, <fpage>18</fpage>&#x2013;<lpage>22</lpage>. </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Duchi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Simultaneous Dimension Reduction and Adjustment for Confounding Variation</article-title>. <source>Proc. Natl. Acad. Sci. U.S.A.</source> <volume>113</volume>, <fpage>14662</fpage>&#x2013;<lpage>14667</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1617317113</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Linck</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Battey</surname>
<given-names>C. J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Minor Allele Frequency Thresholds Strongly Affect Population Structure Inference with Genomic Data Sets</article-title>. <source>Mol. Ecol. Resour.</source> <volume>19</volume>, <fpage>639</fpage>&#x2013;<lpage>647</lpage>. <pub-id pub-id-type="doi">10.1111/1755-0998.12995</pub-id> </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Min</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Network-regularized Sparse Logistic Regression Models for Clinical Risk Prediction and Biomarker Discovery</article-title>. <source>Ieee/acm Trans. Comput. Biol. Bioinform</source> <volume>15</volume>, <fpage>944</fpage>&#x2013;<lpage>953</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2016.2640303</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Min</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Edge-group Sparse PCA for Network-Guided High Dimensional Data Analysis</article-title>. <source>Bioinformatics</source> <volume>34</volume>, <fpage>3479</fpage>&#x2013;<lpage>3487</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty362</pub-id> </citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Min</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Group-Sparse SVD Models via $ L_1 $ L 1-and $ L_0 $ L 0-norm Penalties and Their Applications in Biological Data</article-title>. <source>IEEE Trans. Knowledge Data Eng.</source> <volume>33</volume>, <fpage>536</fpage>&#x2013;<lpage>550</lpage>. </citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Navarro Silvera</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Mayne</surname>
<given-names>S. T.</given-names>
</name>
<name>
<surname>Risch</surname>
<given-names>H. A.</given-names>
</name>
<name>
<surname>Gammon</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>Vaughan</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chow</surname>
<given-names>W.-H.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Principal Component Analysis of Dietary and Lifestyle Patterns in Relation to Risk of Subtypes of Esophageal and Gastric Cancer</article-title>. <source>Ann. Epidemiol.</source> <volume>21</volume>, <fpage>543</fpage>&#x2013;<lpage>550</lpage>. <pub-id pub-id-type="doi">10.1016/j.annepidem.2010.11.019</pub-id> </citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nguyen</surname>
<given-names>P. L.</given-names>
</name>
<name>
<surname>Taghian</surname>
<given-names>A. G.</given-names>
</name>
<name>
<surname>Katz</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Niemierko</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Abi Raad</surname>
<given-names>R. F.</given-names>
</name>
<name>
<surname>Boon</surname>
<given-names>W. L.</given-names>
</name>
<etal/>
</person-group> (<year>2008</year>). <article-title>Breast Cancer Subtype Approximated by Estrogen Receptor, Progesterone Receptor, and HER-2 Is Associated with Local and Distant Recurrence after Breast-Conserving Therapy</article-title>. <source>Jco</source> <volume>26</volume>, <fpage>2373</fpage>&#x2013;<lpage>2378</lpage>. <pub-id pub-id-type="doi">10.1200/jco.2007.14.4287</pub-id> </citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peterson</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>K-nearest Neighbor</article-title>. <source>Scholarpedia</source> <volume>4</volume>, <fpage>1883</fpage>. <pub-id pub-id-type="doi">10.4249/scholarpedia.1883</pub-id> </citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Raedler</surname>
<given-names>L. A.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Zykadia (Ceritinib) Approved for Patients with Crizotinib-Resistant ALK-Positive Non&#x2013;small-cell Lung Cancer</article-title>. <source>Am. Health Drug benefits</source> <volume>8</volume>, <fpage>163</fpage>. </citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reis-Filho</surname>
<given-names>J. S.</given-names>
</name>
<name>
<surname>Pusztai</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Gene Expression Profiling in Breast Cancer: Classification, Prognostication, and Prediction</article-title>. <source>The Lancet</source> <volume>378</volume>, <fpage>1812</fpage>&#x2013;<lpage>1823</lpage>. <pub-id pub-id-type="doi">10.1016/s0140-6736(11)61539-0</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rutenbar</surname>
<given-names>R. A.</given-names>
</name>
</person-group> (<year>1989</year>). <article-title>Simulated Annealing Algorithms: An Overview</article-title>. <source>IEEE Circuits Devices Mag.</source> <volume>5</volume>, <fpage>19</fpage>&#x2013;<lpage>26</lpage>. <pub-id pub-id-type="doi">10.1109/101.17235</pub-id> </citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Mo</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Sparse Integrative Clustering of Multiple Omics Data Sets</article-title>. <source>Ann. Appl. Stat.</source> <volume>7</volume>, <fpage>269</fpage>&#x2013;<lpage>294</lpage>. <pub-id pub-id-type="doi">10.1214/12-AOAS578</pub-id> </citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Mo</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Schultz</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Seshan</surname>
<given-names>V. E.</given-names>
</name>
<name>
<surname>Olshen</surname>
<given-names>A. B.</given-names>
</name>
<name>
<surname>Huse</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Integrative Subtype Discovery in Glioblastoma Using iCluster</article-title>. <source>PloS one</source> <volume>7</volume>, <fpage>e35236</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0035236</pub-id> </citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Olshen</surname>
<given-names>A. B.</given-names>
</name>
<name>
<surname>Ladanyi</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Integrative Clustering of Multiple Genomic Data Types Using a Joint Latent Variable Model with Application to Breast and Lung Cancer Subtype Analysis</article-title>. <source>Bioinformatics</source> <volume>25</volume>, <fpage>2906</fpage>&#x2013;<lpage>2912</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp543</pub-id> </citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Yi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Meta-weight-net: Learning an Explicit Mapping for Sample Weighting</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>32</volume>. </citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Siegel</surname>
<given-names>R. L.</given-names>
</name>
<name>
<surname>Miller</surname>
<given-names>K. D.</given-names>
</name>
<name>
<surname>Jemal</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Cancer Statistics, 2016</article-title>. <source>CA: a Cancer J. clinicians</source> <volume>66</volume>, <fpage>7</fpage>&#x2013;<lpage>30</lpage>. <pub-id pub-id-type="doi">10.3322/caac.21332</pub-id> </citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Siegel</surname>
<given-names>R. L.</given-names>
</name>
<name>
<surname>Miller</surname>
<given-names>K. D.</given-names>
</name>
<name>
<surname>Jemal</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Cancer Statistics, 2019</article-title>. <source>CA A. Cancer J. Clin.</source> <volume>69</volume>, <fpage>7</fpage>&#x2013;<lpage>34</lpage>. <pub-id pub-id-type="doi">10.3322/caac.21551</pub-id> </citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sill</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Saadati</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Benner</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Applying Stability Selection to Consistently Estimate Sparse Principal Components in High-Dimensional Molecular Data</article-title>. <source>Bioinformatics</source> <volume>31</volume>, <fpage>2683</fpage>&#x2013;<lpage>2690</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btv197</pub-id> </citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Symmans</surname>
<given-names>W. F.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Gould</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Long-term Prognostic Risk after Neoadjuvant Chemotherapy Associated with Residual Cancer burden and Breast Cancer Subtype</article-title>. <source>Jco</source> <volume>35</volume>, <fpage>1049</fpage>&#x2013;<lpage>1060</lpage>. <pub-id pub-id-type="doi">10.1200/jco.2015.63.1010</pub-id> </citation>
</ref>
<ref id="B40">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Teng</surname>
<given-names>C.-M.</given-names>
</name>
</person-group> (<year>2003</year>). &#x201c;<article-title>Applying Noise Handling Techniques to Genomic Data: A Case Study</article-title>,&#x201d; in <conf-name>Third IEEE International Conference on Data Mining</conf-name> (<publisher-name>IEEE</publisher-name>), <fpage>743</fpage>&#x2013;<lpage>746</lpage>. </citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tran</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Bedard</surname>
<given-names>P. L.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Luminal-B Breast Cancer and Novel Therapeutic Targets</article-title>. <source>Breast Cancer Res.</source> <volume>13</volume>, <fpage>221</fpage>. <pub-id pub-id-type="doi">10.1186/bcr2904</pub-id> </citation>
</ref>
<ref id="B42">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Van Laarhoven</surname>
<given-names>P. J. M.</given-names>
</name>
<name>
<surname>Aarts</surname>
<given-names>E. H. L.</given-names>
</name>
</person-group> (<year>1987</year>). <source>Simulated Annealing, Simulated Annealing: Theory and Applications</source>. <publisher-loc>Germany</publisher-loc>: <publisher-name>Springer</publisher-name>, <fpage>7</fpage>&#x2013;<lpage>15</lpage>. <pub-id pub-id-type="doi">10.1007/978-94-015-7744-1_2</pub-id> </citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vinga</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Structured Sparsity Regularization for Analyzing High-Dimensional Omics Data</article-title>. <source>Brief. Bioinform.</source> <volume>22</volume>, <fpage>77</fpage>&#x2013;<lpage>87</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbaa122</pub-id> </citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Waks</surname>
<given-names>A. G.</given-names>
</name>
<name>
<surname>Winer</surname>
<given-names>E. P.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Breast Cancer Treatment</article-title>. <source>Jama</source> <volume>321</volume>, <fpage>288</fpage>&#x2013;<lpage>300</lpage>. <pub-id pub-id-type="doi">10.1001/jama.2018.19323</pub-id> </citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wiese</surname>
<given-names>D. A.</given-names>
</name>
<name>
<surname>Thaiwong</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yuzbasiyan-Gurkan</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Kiupel</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Feline Mammary Basal-like Adenocarcinomas: a Potential Model for Human Triple-Negative Breast Cancer (TNBC) with Basal-like Subtype</article-title>. <source>BMC cancer</source> <volume>13</volume>, <fpage>403</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2407-13-403</pub-id> </citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xie</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Bai</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Machine Learning-Based Analysis of MR Multiparametric Radiomics for the Subtype Classification of Breast Cancer</article-title>. <source>Front. Oncol.</source> <volume>9</volume>, <fpage>505</fpage>. <pub-id pub-id-type="doi">10.3389/fonc.2019.00505</pub-id> </citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>Z. Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X. Y.</given-names>
</name>
<name>
<surname>Shu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>Y. Q.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Z. B.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Multi-view Based Integrative Analysis of Gene Expression Data for Identifying Biomarkers</article-title>. <source>Sci. Rep.</source> <volume>9</volume>, <fpage>13504</fpage>&#x2013;<lpage>13515</lpage>. <pub-id pub-id-type="doi">10.1038/s41598-019-49967-4</pub-id> </citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yuan</surname>
<given-names>X.-T.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Truncated Power Method for Sparse Eigenvalue Problems</article-title>. <source>J. Machine Learn. Res.</source> <volume>14</volume>, <fpage>899</fpage>&#x2013;<lpage>925</lpage>. </citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zeng</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Rao</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Genome-wide Analyses on Single Disease Samples for Potential Biomarkers and Biological Features of Molecular Subtypes: a Case Study in Gastric Cancer</article-title>. <source>Int. J. Biol. Sci.</source> <volume>14</volume>, <fpage>833</fpage>&#x2013;<lpage>842</lpage>. <pub-id pub-id-type="doi">10.7150/ijbs.24816</pub-id> </citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Pache</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Khodabakhshi</surname>
<given-names>A. H.</given-names>
</name>
<name>
<surname>Tanaseichuk</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Metascape Provides a Biologist-Oriented Resource for the Analysis of Systems-Level Datasets</article-title>. <source>Nat. Commun.</source> <volume>10</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1038/s41467-019-09234-6</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>