<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="methods-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1407072</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2024.1407072</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Specific feature recognition on group specific networks (SFR-GSN): a biomarker identification model for cancer stages</article-title>
<alt-title alt-title-type="left-running-head">Chen et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2024.1407072">10.3389/fgene.2024.1407072</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Chen</surname>
<given-names>Bolin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/833694/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Yuxin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Jinlei</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1937974/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Han</surname>
<given-names>Yourui</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2704222/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Benhammouda</surname>
<given-names>Hamza</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Bian</surname>
<given-names>Jun</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Kang</surname>
<given-names>Ruiming</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Shang</surname>
<given-names>Xuequn</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/898708/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Computer Science</institution>, <institution>Northwestern Polytechnical University</institution>, <addr-line>Xi&#x2019;an</addr-line>, <addr-line>Shaanxi</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Key Laboratory of Big Data Storage and Management</institution>, <institution>Northwestern Polytechnical University</institution>, <institution>Ministry of Industry and Information Technology</institution>, <addr-line>Xi&#x2019;an</addr-line>, <addr-line>Shaanxi</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of General Surgery</institution>, <institution>Xi&#x2019;an Children&#x2019;s Hosptial</institution>, <institution>Xi&#x2019;an Jiaotong University Affiliated Children&#x2019;s Hosptial</institution>, <addr-line>Xi&#x2019;an</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Rewise (Hangzhou) Information Technology Co., Ltd</institution>, <addr-line>Hangzhou</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/744584/overview">Xuefeng Cui</ext-link>, Shandong University, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/674022/overview">Pu-Feng Du</ext-link>, Tianjin University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/703922/overview">Jiazhou Chen</ext-link>, South China University of Technology, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/338783/overview">Hongdong Li</ext-link>, Central South University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Bolin Chen, <email>blchen@nwpu.edu.cn</email>; Jun Bian, <email>blandbird@126.com</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>23</day>
<month>05</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1407072</elocation-id>
<history>
<date date-type="received">
<day>26</day>
<month>03</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>01</day>
<month>05</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Chen, Wang, Zhang, Han, Benhammouda, Bian, Kang and Shang.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Chen, Wang, Zhang, Han, Benhammouda, Bian, Kang and Shang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Background and Objective</title>
<p>Accurate identification of cancer stages is challenging due to the complexity and heterogeneity of the disease. Current clinical diagnosis methods primarily rely on phenotypic observations, which may not capture early molecular-level changes accurately.</p>
</sec>
<sec>
<title>Methods</title>
<p>In this study, a novel biomarker recognition method was proposed tailored for cancer stages by considering the change of gene expression relationships. Utilizing the sample-specific information and protein-protein interaction networks, the group specific networks were constructed to address the limited specificity of potential biomarkers. Then, a specific feature recognition method was proposed based on these group specific networks, which employed the random forest algorithm for initial screening followed by a recursive feature elimination process to identify the optimal biomarker subset. During exploring optimal results, a strategy termed the Cost-Benefit Ratio, was devised to facilitate the identification of stage-specific biomarkers.</p>
</sec>
<sec>
<title>Results</title>
<p>Comparative experiments were conducted on lung adenocarcinoma and breast cancer datasets to validate the method&#x2019;s efficacy and generalizability. The results showed that the identified biomarkers were highly stage-specific, and the F1 scores for predicting cancer stages were significantly improved. For the lung adenocarcinoma dataset, the F1 score reached 97.68%, and for the breast cancer dataset, it achieved 96.87%. These results significantly surpassed those of three conventional methods in terms of F1 scores. Moreover, from the perspective of biological functions, the biomarkers were proved playing an important role in cancer stage-evolution.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>The proposed method demonstrated its effectiveness in identifying stage-related biomarkers. By using these biomarkers as features, accurate prediction of cancer stages was achieved. Furthermore, the method exhibited potential for biomarker identification in subtype analyses, offering novel perspectives for cancer prognosis.</p>
</sec>
</abstract>
<kwd-group>
<kwd>biomarker</kwd>
<kwd>cancer stages</kwd>
<kwd>group specific network</kwd>
<kwd>multi classification tasks</kwd>
<kwd>edge feature</kwd>
</kwd-group>
<contract-num rid="cn001">2021YFA1000402</contract-num>
<contract-num rid="cn002">61972320</contract-num>
<contract-num rid="cn003">22YXYJ0057</contract-num>
<contract-sponsor id="cn001">National Key Research and Development Program of China<named-content content-type="fundref-id">10.13039/501100012166</named-content>
</contract-sponsor>
<contract-sponsor id="cn002">National Natural Science Foundation of China<named-content content-type="fundref-id">10.13039/501100001809</named-content>
</contract-sponsor>
<contract-sponsor id="cn003">Xi&#x2019;an Municipal Bureau of Science and Technology<named-content content-type="fundref-id">10.13039/501100008387</named-content>
</contract-sponsor>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Genomics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Cancer is a disease characterized by uncontrolled cell proliferation, posing a serious threat to human health. According to the World Health Organization, in 2020 alone, nearly 10 million people (about one-sixth of all deaths worldwide) died from cancer (<xref ref-type="bibr" rid="B20">Sung et al., 2021</xref>). Understanding cancer begins with an important dimension: its stages, which could describe the size and extent of tumor spread. Due to the high heterogeneity and complexity of cancer, it poses significant challenges for the identification of cancer stages (<xref ref-type="bibr" rid="B3">Burrell et al., 2013</xref>). Hence, investigating an intelligent model for the identification of stage-related biomarkers is very important. It helps in understanding the characteristics and changes during the development process of cancer. This research endeavor proves valuable in enhancing cancer treatment strategies and prognostic assessments.</p>
<p>As far as the biomarkers are concerned, encompass a range of molecules, cellular structures, or biological processes that can be objectively detected and quantified within or outside an organism (<xref ref-type="bibr" rid="B13">Moein et al., 2020</xref>). They play a crucial role in revealing an individual&#x2019;s health status, physiological functions, pathological conditions, and biological responses to treatment. This makes them integral players in the development of precision medicine and personalized treatment strategies (<xref ref-type="bibr" rid="B7">Holland, 2016</xref>). Specifically, stage-related biomarkers provide crucial information about tumor progression, metastasis, and treatment response (<xref ref-type="bibr" rid="B2">Amin et al., 2010</xref>; <xref ref-type="bibr" rid="B24">Van der Kloet et al., 2012</xref>). By analyzing the expression patterns and changes of stage-related biomarkers, healthcare professionals and researchers can gain a better understanding of the cancer&#x2019;s progression status, choose appropriate treatment strategies, and monitor treatment effectiveness.</p>
<p>However, molecular distinctions between different cancer stages are often subtle (<xref ref-type="bibr" rid="B27">Ye et al., 2020</xref>). For example, in early-stage cancer, molecular changes may be influenced by minor alterations in the activity of a few key genes or subtle modulation of signaling pathways. The boundaries between cancer stages, as defined clinically, are often indistinct at the molecular level. For instance, the molecular changes between stage I of a late-stage and stage II of an early-stage cancer could be very similar. Therefore, the identification of stage-related biomarkers at the molecular level has been a long-standing challenge.</p>
<p>Currently, two mainstream approaches primarily guide the identification of stage-related biomarkers. The first category is based on differential expression analysis. <xref ref-type="bibr" rid="B5">Deva Magendhra Rao et al. (2019)</xref> compared non-coding RNAs (lncRNAs) between invasive ductal carcinoma (IDC) breast cancer tissues and normal breast tissues. There were 375 differentially expressed lncRNAs identifying closely associated with the early-stage development of breast cancer. <xref ref-type="bibr" rid="B19">Shi et al. (2018)</xref> analyzed gene expression data from four stages of colorectal cancer, identifying stage-specific differentially expressed genes and exploring their shared biological functions. <xref ref-type="bibr" rid="B25">Wang et al. (2017)</xref> studied gene expression data in non-small cell lung cancer and found that differentially expressed genes at different stages significantly impacted biological functions and signaling pathways. However, these methods often overlook molecular interactions and typically validate their findings through functional or pathway enrichment analysis but few focus on the identification of stage-related biomarkers.</p>
<p>On the other hand, the second category, focuses on machine learning techeques. <xref ref-type="bibr" rid="B16">Patil and Bellary (2022)</xref> achieved good performance in stage identification of melanoma based on features from dermoscopic images and tumor thickness using machine learning. <xref ref-type="bibr" rid="B23">Ubaldi et al. (2021)</xref> performed a binary classification task to identify stage I and stage II non-small cell lung cancer using radiometric data and machine learning, achieving a high AUC value at 0.84. <xref ref-type="bibr" rid="B9">Jin et al. (2021)</xref> developed an interpretable machine learning model that could identify gene expression biomarkers for early-stage LUAD. However, these methods typically focus on building accurate prediction models similar to a &#x201c;black box&#x201d; with limited biological and clinical interpretability. Some researchers strive to construct interpretable machine learning models for identifying stage-related biomarkers, but this often leads to compromises in the predictive performance of the model to some extent for the samples are imbalanced, and there is minimal molecular-level difference between different stages. In summary, existent methods have weaker specificity in identifying stage-related molecular-level biomarkers.</p>
<p>In this paper, an efficient method was proposed to identify stage-related biomarkers through specific feature recognition on group specific networks (SFR-GSN), which could sensitively capture the differences between different stages and identify features that exhibit significant specificity between stages. Two mainly high-risk cancers, lung adenocarcinoma (LUAD) and breast carcinoma (BRCA), were used to evaluate the proposed method. Firstly, the clinical data, RNA-Seq data and protein-protein interactions (PPI) of LUAD and BRCA were first collected from public database. Then, based on the tumor samples and normal samples, the sample-specific networks (SSN) were constructed, which further intersected with PPI to construct the group-specific network (GSN). Through clinical data, GSNs were combined into one GSN corresponding to one cancer stage, which could address the weak specificity of existing biomarkers. Subsequently, a specific feature recognition (SFR) method based on these GSNs was proposed. SFR was designed in two-round, the first round was pre-screening by utilizing the random forest algorithm with Gini impurity quantifying the purity improvement. The second round was optimal subset screening of biomarkers by using the recursive feature elimination with cross-validation. Notably, during exploring the optimal results, the Cost-Benefit Ratio (CBR) was introduced as an important indicator for identifying the stage-related biomarkers. Eventually, comparative experiments among SFR-GSN and three state-of-the-art methods were conducted on LUAD and BRCA datasets to validate the effectiveness and generalization ability of the proposed method. The results showed that the identified biomarkers significantly improved F1 scores for predicting cancer stages. Also from the perspective of biological functions, the biomarkers were proved playing an important role in cancer stage-evolution.</p>
</sec>
<sec id="s2" sec-type="methods">
<title>2 Methods</title>
<sec id="s2-1">
<title>2.1 Data collection</title>
<p>In the study, we focused on two kinds of cancer, lung adenocarcinoma (LUAD) and breast cancer (BRCA). On one hand, LUAD and BRCA are both cancer types associated with high levels of severity. LUAD is one of the most common subtypes of lung cancer, while BRCA is one of the most prevalent cancers among women. These two cancer types significantly impact patients&#x2019; quality of life and survival rates. On the other hand, since LUAD and BRCA are two common types of cancer with relatively high incidence rates worldwide, as a result, these cancer types have ample sample data available. The richness of data helps improve the accuracy and reliability of the models. Therefore, studying and analyzing datasets related to LUAD and BRCA can enhance our understanding of the disease mechanisms, risk factors, and treatment strategies, providing valuable insights for cancer diagnosis and treatment.</p>
<p>We separately collected the clinical data and RNA-Seq data of LUAD and BRCA from Xena <xref ref-type="bibr" rid="B22">Tomczak et al. (2015)</xref>; <xref ref-type="bibr" rid="B26">Wang et al. (2022)</xref> and separated the RNA-Seq data into different pathological stages. Then, the counts per million (CPM) (<xref ref-type="bibr" rid="B10">Law et al., 2016</xref>) were applied to filter the low-expression genes, and genes with a value higher than 2 CPM in at least half of the samples were retained. Additionally, the protein-protein interactions were compiled from STRING (<xref ref-type="bibr" rid="B21">Szklarczyk et al., 2023</xref>). PPI was widely used in identifying biomolecules, including biomarkers, and driver genes in many studies. The RNA-Seq datasets used in the experiments is shown in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>The number of samples of LUAD and BRCA in experiments.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Cancer types</th>
<th align="center">Normal</th>
<th align="center">Stage I</th>
<th align="center">Stage II</th>
<th align="center">Stage III</th>
<th align="center">Stage IV</th>
<th align="center">Sum</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">LUAD</td>
<td align="center">59</td>
<td align="center">273</td>
<td align="center">122</td>
<td align="center">83</td>
<td align="center">26</td>
<td align="center">563</td>
</tr>
<tr>
<td align="center">BRCA</td>
<td align="center">114</td>
<td align="center">182</td>
<td align="center">621</td>
<td align="center">250</td>
<td align="center">20</td>
<td align="center">1,187</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-2">
<title>2.2 Construction of group specific networks</title>
<p>The group specific networks were constructed based on the two main kinds of networks: Sample-Specific Networks (SSN) and PPI networks. Proposed by <xref ref-type="bibr" rid="B11">Liu et al. (2016)</xref>, SSN could assist in identifying driver genes from the perspective of the personalized network. GSN, combined SSN, and the existing PPI could increase the robustness of the interactions. The flow of the construction of GSN was summarized in <xref ref-type="fig" rid="F1">Figure 1</xref>.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The flowchart of constructing GSN. The red dotted line section is the construction of SSN, while the blue dotted line section is the following part, using SSN and PPI to construct the GSN.</p>
</caption>
<graphic xlink:href="fgene-15-1407072-g001.tif"/>
</fig>
<p>SSN was initially constructed based on RNA-Seq data. For all normal samples, a reference network was constructed by calculating the pairwise gene-gene Pearson correlation coefficients (<italic>PCC</italic>, represented in the reference network as <italic>PCC</italic>
<sub>
<italic>n</italic>
</sub>). Meanwhile, for each disease sample, a perturbation network was generated by incorporating the normal sample set and reconstructing the network, resulting in <italic>PCC</italic>
<sub>
<italic>n</italic>&#x002B;1</sub>. Subsequently, the differential network was obtained by subtracting the perturbation network from the reference network, and the difference was derived as Formula <xref ref-type="disp-formula" rid="e1">(1)</xref>.<disp-formula id="e1">
<mml:math id="m1">
<mml:mo>&#x25b3;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>&#x003D;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mi>C</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x002B;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mi>C</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
<label>(1)</label>
</disp-formula>Edges with a statistical <italic>p</italic>-<italic>value</italic> &#x003c; 0.05 were considered significant and retained. In the constructed SSN, nodes represent genes, while the connections between nodes indicate significant differences in the correlation between the two genes in the disease sample compared to the normal sample set. This dissimilarity is quantified by &#x25b3;<italic>PCC</italic>.</p>
<p>Then, on the basis of the SSN, intersections were combined with the PPI. We retained the experimentally validated edges presented in PPI, with the edge weight calculated from SSN. Due to the samples could be divided into different pathological stages, the PPI-SSN for all samples was classified according to different stages (groups) of cancer. For instance, within one specific cancer group <italic>G</italic>
<sub>
<italic>i</italic>
</sub> responding to one GSN, consisting of <italic>N</italic> cancer samples, the <italic>N</italic> PPI-SSNs were integrated. Also, the edge weight was calculated by taking an average on the &#x25b3;<italic>PCC</italic> of same edge in <italic>N</italic> different samples. As for the edges not appearing in the samples, their &#x25b3;<italic>PCC</italic> was set to 0. Finally, the edge weight of GSN was derived as Formula <xref ref-type="disp-formula" rid="e2">(2)</xref>.<disp-formula id="e2">
<mml:math id="m2">
<mml:mi>w</mml:mi>
<mml:mo>&#x003D;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x003D;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x25b3;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mi>C</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(2)</label>
</disp-formula>
</p>
<p>Considering the generalization of GSN, a ten-fold cross-validation approach was employed during the experimental process. A GSN was constructed for each training fold, resulting in ten GSNs, and the edge weight from these ten GSNs was also averaged. Ultimately for every cancer group <italic>G</italic>
<sub>
<italic>i</italic>
</sub>, only one corresponding GSN was constructed, which is stage-specific.</p>
</sec>
<sec id="s2-3">
<title>2.3 Specific feature recognition</title>
<p>Based on the constructed GSN, we aimed to identify the most representative and minimal set of features as biomarkers. These features in the selected set contain a high degree of complementary information, resembling a minimal control network. Feature recognition consists of two main parts: pre-screening and optimal subset screening of biomarkers.</p>
<sec id="s2-3-1">
<title>2.3.1 Pre-screening of biomarkers</title>
<p>The edge set of each GSN corresponding to each group is sorted in descending order based on the edge weights and subjected to pre-screening to obtain the top 50 edges. Among the top 50 edges, the features at both ends of these selected edges are obtained, and their union forms the candidate feature set. Then the candidate feature set is further filtered using the feature importance calculation algorithm embedded in random forest (<xref ref-type="bibr" rid="B1">Acharjee et al., 2020</xref>), narrowing it down to a new candidate feature set, which containing only the top 50 features based on their importance rankings. During the feature pre-screening, the Gini impurity was introduced to quantify the purity improvement achieved through branching. The Gini impurity, presented as <italic>Gini</italic>, could be derived as Formula <xref ref-type="disp-formula" rid="e3">(3)</xref>.<disp-formula id="e3">
<mml:math id="m3">
<mml:mi>G</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x003D;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x003D;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msubsup>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:math>
<label>(3)</label>
</disp-formula>where <italic>p</italic>
<sub>
<italic>i</italic>
</sub> represents the relative frequency of the <italic>i</italic>-th class in the dataset, which is the probability of that class occurring in the dataset, and <italic>n</italic> is the total number of categories.</p>
<p>In random forest, the calculation of feature importance is based on the Gini impurity of each feature at each node in every tree. Specifically, for each feature, at each node of each tree, the algorithm splits the dataset into two subsets based on that feature. Then, the difference between the Gini impurity of the subsets after the split and the Gini impurity of the original node was calculated. Finally, by aggregating the feature importance scores from all nodes, the overall feature importance for each feature in the random forest was obtained. The built-in feature importance evaluation capability of the random forest makes it a powerful tool for understanding data and extracting key biomarkers in multi-class classification tasks. The whole pre-screening procession was described in <xref ref-type="statement" rid="Algorithm_1">Algorithm 1</xref>.</p>
<p>
<statement content-type="algorithm" id="Algorithm_1">
<label>Algorithm 1.</label>
<p>Pre-screening of biomarkers by random forest feature importance calculation.<list list-type="simple">
<list-item>
<p>
<bold>Require:</bold> Random forest model <italic>RF</italic>, training data set <italic>D</italic>;</p>
</list-item>
<list-item>
<p>
<bold>Ensure:</bold> A list of feature importances <italic>importance</italic>;</p>
</list-item>
<list-item>
<p>1: <bold>for</bold> each <italic>tree</italic> in <italic>RF</italic> <bold>do</bold>
</p>
</list-item>
<list-item>
<p>2: <bold>for</bold> each <italic>node</italic> in <italic>tree</italic> <bold>do</bold>
</p>
</list-item>
<list-item>
<p>3: <bold>for</bold> each <italic>feature f</italic> in <italic>node</italic> <bold>do</bold>
</p>
</list-item>
<list-item>
<p>4: Split the dataset at <italic>node</italic> into two subsets <italic>D</italic>
<sub>
<italic>left</italic>
</sub> and <italic>D</italic>
<sub>
<italic>right</italic>
</sub> based on feature <italic>f</italic>;</p>
</list-item>
<list-item>
<p>5: Calculate the Gini impurity of the original node, denoted as Gini;</p>
</list-item>
<list-item>
<p>6: Calculate the Gini impurity of <italic>D</italic>
<sub>
<italic>left</italic>
</sub>, denoted as <italic>Gini</italic>
<sub>
<italic>left</italic>
</sub>;</p>
</list-item>
<list-item>
<p>7: Calculate the Gini impurity of <italic>D</italic>
<sub>
<italic>right</italic>
</sub>, denoted as <italic>Gini</italic>
<sub>
<italic>right</italic>
</sub>;</p>
</list-item>
<list-item>
<p>8: Calculate the gain in impurity after splitting on feature <italic>f</italic>:</p>
</list-item>
<list-item>
<p>9: <inline-formula id="inf1">
<mml:math id="m4">
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
<mml:mi>G</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x003D;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">left</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>D</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>G</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">left</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x002B;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">right</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>D</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>G</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">right</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>;</p>
</list-item>
<list-item>
<p>10: Update the importance of <italic>feature f</italic> based on the impurity gain:</p>
</list-item>
<list-item>
<p>11: <italic>importance [f] &#x2190; importance [f] &#x002B; impurityGain</italic>;</p>
</list-item>
<list-item>
<p>12: <bold>end for</bold>
</p>
</list-item>
<list-item>
<p>13: <bold>end for</bold>
</p>
</list-item>
<list-item>
<p>14: <bold>end for</bold>
</p>
</list-item>
<list-item>
<p>15: Sort the features based on the values in <italic>importance</italic> using a suitable sorting algorithm.</p>
</list-item>
</list>
</p>
</statement>
</p>
</sec>
<sec id="s2-3-2">
<title>2.3.2 Optimal subset screening of biomarkers</title>
<p>After the pre-screening, the top 50 candidate feature sets were further filtered by Recursive Feature Elimination with Cross-Validation (RFECV). The RFECV algorithm finds the optimal feature subset by iteratively removing features, involving model training and cross-validation for each reduced feature set. In each iteration, the algorithm removes the least important feature (the one contributing the least to the model&#x2019;s performance improvement), retrains the model on the remaining feature set, and performs cross-validation. This process continues until a specific number of features is reached or further removal of features significantly degrades model performance.</p>
<p>Notably, to select the minimum number of features that achieve the best predictive performance, the Cost-Benefit Ratio (CBR) was introduced to assist in screening the optimal feature set (<xref ref-type="bibr" rid="B4">De Picker and Haarman, 2021</xref>). The CBR could be defined as Formula <xref ref-type="disp-formula" rid="e4">(4)</xref>.<disp-formula id="e4">
<mml:math id="m5">
<mml:mi>C</mml:mi>
<mml:mi>B</mml:mi>
<mml:mi>R</mml:mi>
<mml:mo>&#x003D;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>100</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>F</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>U</mml:mi>
<mml:mi>F</mml:mi>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(4)</label>
</disp-formula>in this formula, the symbols represent the following:<list list-type="simple">
<list-item>
<p>&#x2022; <italic>PR</italic>: Performance Gain, which refers to the improvement of the <italic>F</italic>1 score in the model.</p>
</list-item>
<list-item>
<p>&#x2022; <italic>INF</italic>: Increased Number of Features.</p>
</list-item>
<list-item>
<p>&#x2022; <italic>UFC</italic>: Unit Feature Cost.</p>
</list-item>
</list>
</p>
<p>Through CBR, we can quantitatively evaluate whether the performance improvement gained from adding specific features is worth the additional cost required. It is particularly important in situations where there is a need to balance decisions between performance improvement and cost.</p>
<p>During the model training, multiple thresholds (<italic>thresh</italic>) were set for the number of features and obtained their corresponding model performance evaluation metric, <italic>F</italic>1 score. Then, according to the CBR model, the optimal feature subset was screened in a recursive way. The optimal subset screening of biomarkers using RFECV was presented as <xref ref-type="statement" rid="Algorithm_2">Algorithm 2</xref>.</p>
<p>
<statement content-type="algorithm" id="Algorithm_2">
<label>Algorithm 2</label>
<p>Recursive Feature Elimination with Cross-Validation (RFECV) algorithm.<list list-type="simple">
<list-item>
<p>
<bold>Require:</bold> candidate feature set, threshold for the number of features <italic>thresh</italic>;</p>
</list-item>
<list-item>
<p>
<bold>Ensure:</bold> feature set <italic>S</italic>, model performance evaluation metric <italic>F</italic>1 score;</p>
</list-item>
<list-item>
<p>1: Initialize the feature set <italic>S</italic> and set it as the candidate feature set;</p>
</list-item>
<list-item>
<p>2: Define the model performance evaluation metric <italic>F</italic>1 score;</p>
</list-item>
<list-item>
<p>3: Define the threshold for the number of features <italic>thresh</italic>;</p>
</list-item>
<list-item>
<p>4: <bold>while</bold> <italic>S</italic> is not &#x2205; <bold>do</bold>
</p>
</list-item>
<list-item>
<p>5: Train the model using the feature set as the training set;</p>
</list-item>
<list-item>
<p>6: Introduce cross-validation to evaluate the model performance;</p>
</list-item>
<list-item>
<p>7: <bold>if</bold> the number of features &#x003D; &#x003D; <italic>thresh</italic> <bold>then</bold>
</p>
</list-item>
<list-item>
<p>8: Save the current feature set as <italic>S</italic>;</p>
</list-item>
<list-item>
<p>9: Save the current model performance metric as <italic>F</italic>1 score;</p>
</list-item>
<list-item>
<p>10: <italic>break</italic>;</p>
</list-item>
<list-item>
<p>11: <bold>end if</bold>
</p>
</list-item>
<list-item>
<p>12: Reove the least contributing features from <italic>S</italic>;</p>
</list-item>
<list-item>
<p>13: <bold>end while</bold>
</p>
</list-item>
</list>
</p>
</statement>
</p>
</sec>
</sec>
</sec>
<sec id="s3" sec-type="results">
<title>3 Results</title>
<p>The experimental results were obtained using ten-fold cross-validation to ensure reliability. In each round, nine folds of the datasets were treated as a train set and the other one fold acted as a test set. The train set was used to construct the GSN and select the feature. The test set was utilized to evaluate the model performance. In addition, specific feature experiments and comparative analyses were conducted to validate the effectiveness of the model. Moreover, the proposed method was expanded to identify cancer subtypes related biomarkers as well.</p>
<sec id="s3-1">
<title>3.1 Specific feature experiments</title>
<p>Specific feature experiments were conducted in the following two steps. Firstly, the important parameters were introduced including the CBR and number of features. Secondly, the stage-specific biomarkers in LUAD and BRCA datasets were identified. The effectiveness of the identified biomarker were performed through enrichment analysis.</p>
<sec id="s3-1-1">
<title>3.1.1 Setting of the important parameter</title>
<p>CBR was designed as a key parameter to assist in screening the optimal feature set, which is directly related to the number of features. The proposed methods were conducted on LUAD and BRCA datasets to determine a series of feature counts, and the F1 scores and CBRs were calculated through the experiments which was summarized in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>F1 score and CBR for multi-class classification in stages of LUAD and BRCA at different feature quantity thresholds.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center" colspan="3">LUAD</th>
<th align="center" colspan="3">BRCA</th>
</tr>
<tr>
<th align="center">Number of features</th>
<th align="center">F1 score (%)</th>
<th align="center">CBR</th>
<th align="center">Number of features</th>
<th align="center">F1 score(%)</th>
<th align="center">CBR</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">1</td>
<td align="center">48.7420</td>
<td align="center">-</td>
<td align="center">1</td>
<td align="center">71.9124</td>
<td align="center">-</td>
</tr>
<tr>
<td align="center">2</td>
<td align="center">86.2510</td>
<td align="center">37.5090</td>
<td align="center">2</td>
<td align="center">93.4443</td>
<td align="center">21.5318</td>
</tr>
<tr>
<td align="center">3</td>
<td align="center">91.2557</td>
<td align="center">5.0047</td>
<td align="center">3</td>
<td align="center">95.3618</td>
<td align="center">1.9174</td>
</tr>
<tr>
<td align="center">4</td>
<td align="center">92.5050</td>
<td align="center">1.2492</td>
<td align="center">4</td>
<td align="center">96.2117</td>
<td align="center">0.8499</td>
</tr>
<tr>
<td align="center">5</td>
<td align="center">93.3089</td>
<td align="center">0.8039</td>
<td align="center">5</td>
<td align="center">97.2808</td>
<td align="center">1.0691</td>
</tr>
<tr>
<td align="center">6</td>
<td align="center">96.1537</td>
<td align="center">2.8448</td>
<td align="center">6</td>
<td align="center">97.4655</td>
<td align="center">0.1846</td>
</tr>
<tr>
<td align="center">7</td>
<td align="center">96.8517</td>
<td align="center">0.6979</td>
<td align="center">7</td>
<td align="center">98.2629</td>
<td align="center">0.7974</td>
</tr>
<tr>
<td align="center">8</td>
<td align="center">96.9067</td>
<td align="center">0.0550</td>
<td align="center">8</td>
<td align="center">
<bold>99.1047</bold>
</td>
<td align="center">0.8417</td>
</tr>
<tr>
<td align="center">9</td>
<td align="center">97.4935</td>
<td align="center">0.5868</td>
<td align="center">9</td>
<td align="center">98.3727</td>
<td align="center">&#x2212;0.7319</td>
</tr>
<tr>
<td align="center">10</td>
<td align="center">
<bold>97.8804</bold>
</td>
<td align="center">0.3868</td>
<td align="center">10</td>
<td align="center">99.0001</td>
<td align="center">0.6273</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bold values represent the best results among the column.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>From the table, it is shown that in LUAD datasets, as the number of features increases, the F1 score generally improves, but the <italic>CBR</italic> shows non-monotonic variations. Therefore, to further illustrate the relationship between <italic>CBR</italic> and the number of features, their relationship in LUAD datasets was plotted in <xref ref-type="fig" rid="F2">Figure 2</xref>. In the figure, the <italic>CBR</italic> values were compared with 0.5, as this threshold is often used as a balancing point. When the <italic>CBR</italic> is greater than 0.5, it indicates a profitable decision, while a <italic>CBR</italic> lower than 0.5 suggests a cost-effective decision.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The relationship between <italic>CBR</italic> and number of features in LUAD datasets. The red dotted line represents <italic>CBR</italic> &#x003D; 0.5 for parameter setting. From the figure, the best number of features in LUAD datasets is 7.</p>
</caption>
<graphic xlink:href="fgene-15-1407072-g002.tif"/>
</fig>
<p>Therefore, the <italic>CBR</italic> metric was utilized to determine the optimal number of features.</p>
<p>Starting with a small number of features and gradually increasing, the point was identified where the first <italic>CBR</italic> value fell below 0.5.</p>
<p>The CBR indicates the overall benefit of adding a new feature to the model. Therefore, the feature count just before this point was identified as the optimal number of features.</p>
</sec>
<sec id="s3-1-2">
<title>3.1.2 Stage-specific biomarkers</title>
<p>Based on the parameter setting, features with CBR values greater than 0.5 were selected to maximize the F1 score. The obtained biomarkers were in the form of gene pairs or edges.</p>
<p>Compared with the node features, the edge biomarkers could better capture the interaction relationships between genes, aiding in understanding the structure and functionality of gene networks.</p>
<p>The edge features could reflect the interplay and coordinated regulation among genes, revealing more details about biological processes and disease development.</p>
<p>As for the LUAD dataset, seven features were eventually identified that meet this criterion, achieving an impressive F1 score at 96.8517% and a CBR at 0.6979. These features include: (ABI2, ARPC1B), (CDK12, POLR2I), (FRS2, FRS3), (PABPC4, ZC3H14), (SNAP29, TSNARE1), (SEC24C, TRAPPC6B), and (CUL4A, RPA1). Similarly, for the BRCA dataset, five features were selected that yielded a remarkable F1 score at 97.2808% and a CBR at 1.0691. These features are: (EXOSC3, SKIV2L2), (BYSL, UTP14C), (EXOSC8, UTP14C), (PPP3CB, WDR82), and (CD59, SEC24C). The Venn graph of the obtained biomarkers is shown in <xref ref-type="fig" rid="F3">Figure 3</xref>, which demonstrates the biomarkers were highly stage-specific.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Venn graph of the obtained stage-related biomarkers for LUAD and BRCA.</p>
</caption>
<graphic xlink:href="fgene-15-1407072-g003.tif"/>
</fig>
</sec>
</sec>
<sec id="s3-2">
<title>3.2 Enrichment analysis</title>
<p>Moreover, the Kyoto Encyclopedia of Genes and Genomes (KEGG) Pathway analysis and Gene Ontology (GO) enrichment analysis were performed to validate the effectiveness of identified biomarkers.</p>
<p>KEGG pathway enrichment analysis is a frequently employed method in bioinformatics to interpret gene expression or protein expression data (<xref ref-type="bibr" rid="B15">Ogata et al., 1999</xref>). After performing a significance test on 14 genes in the biomarkers of LUAD stages, a total of seven genes were found to be enriched in 10 pathways. Among them, the gene RPA1 was found to be involved in five pathway processes, as shown in <xref ref-type="fig" rid="F4">Figure 4</xref>. In the figure, the red dots represent genes, and the different colored curves represent different pathways. One end of the curve represents a gene, while the other end represents the hub of that pathway, and the size of the hub is proportional to the number of genes enriched in that pathway. As for stage-related biomarkers of BRCA, a total of three genes were found to be enriched in two pathways. Specifically, genes EXOSC8 and EXOSC3 were enriched in hsa03018: RNA degradation, while gene PPP3CB was enriched in hsa04370: VEGF signaling pathway. Due to the small number of genes, they were not visualized.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>KEGG pathway enrichment result of LUAD stage-related biomarker.</p>
</caption>
<graphic xlink:href="fgene-15-1407072-g004.tif"/>
</fig>
<p>GO (Gene Ontology) enrichment analysis was carried out to help understand the roles of a set of genes in biological processes <xref ref-type="bibr" rid="B6">Harris et al. (2004)</xref>. GO enrichment analysis was carried out on the 14 genes in the stage-related biomarkers of LUAD, and the results are shown in <xref ref-type="fig" rid="F5">Figure 5</xref> LUAD, which indicates these 14 genes are involved in a total of 240 biological processes. In the figure, the <italic>x</italic>-axis represents the enrichment score, which indicates the degree of influence of the target genes on the corresponding GO term, while <italic>y</italic>-axis represents the different GO terms. The different colors represent the three main categories of GO. Each category includes only the top 10 terms based on their enrichment score. Similarly, GO enrichment analysis was performed on the nine genes in the stage-related biomarkers of BRCA, and the results are shown in <xref ref-type="fig" rid="F5">Figure 5</xref> BRCA. These nine genes were found to participate in a total of 205 biological processes.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>GO enrichment results of stage biomarkers for LUAD and BRCA.</p>
</caption>
<graphic xlink:href="fgene-15-1407072-g005.tif"/>
</fig>
<p>The enrichment results demonstrate significant specificity of the features constructed using our proposed method across different stages within the two major cancer types, LUAD and BRCA. The evidence further validates the effectiveness of the proposed method.</p>
</sec>
<sec id="s3-3">
<title>3.3 Comparative experiments</title>
<p>Comparative experiments were conducted mainly in view of stage-related biomarker prediction. The proposed method was compared with the three conventional methods on biomarker identification: differentially expression genes (DEGs) <xref ref-type="bibr" rid="B12">Love et al. (2014)</xref>, WGCNA <xref ref-type="bibr" rid="B8">Horvath (2011)</xref> and RelifF <xref ref-type="bibr" rid="B18">Robnik-&#x160;ikonja and Kononenko (2003)</xref>. DEGs were mainly obtained using R package DESeq2 to conduct differential expression analysis, and the DEGs were treated as biomarkers. Based on differential expression data, WGCNA (Weighted Gene Co-expression Network Analysis) is a method used to construct co-expression networks from gene expression data, which is currently widely applied in the identification of biomarkers for complex diseases and drug targets. RelifF is a machine learning method on binary classification, which could identify the biomarkers.</p>
<p>Moreover, to ensure an equivalent comparison, the four methods were compared based on their best performance. Also, the features in all genes were performed as a control group. The F1 score was employed for evaluation since it is not influenced by the varying number of features across different methods. The results of the comparative experiments on LUAD datasets and BRCA datasets are shown in <xref ref-type="table" rid="T3">Table 3</xref>.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>The comparison of identification in stage-related biomarkers among SFR-GSN, three conventional methods, and all genes on LUAD and BRCA datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center"/>
<th align="center" colspan="2">LUAD</th>
<th align="center" colspan="2">BRCA</th>
</tr>
<tr>
<th align="center">Methods</th>
<th align="center">Number of features</th>
<th align="center">F1 score(%)</th>
<th align="center">Number of features</th>
<th align="center">F1 score(%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">All Genes</td>
<td align="center">1,3326</td>
<td align="center">38.90</td>
<td align="center">1,3168</td>
<td align="center">42.77</td>
</tr>
<tr>
<td align="center">DEGs</td>
<td align="center">225</td>
<td align="center">42.42</td>
<td align="center">318</td>
<td align="center">42.51</td>
</tr>
<tr>
<td align="center">WGCNA</td>
<td align="center">151</td>
<td align="center">40.35</td>
<td align="center">396</td>
<td align="center">43.89</td>
</tr>
<tr>
<td align="center">Relife</td>
<td align="center">100</td>
<td align="center">42.29</td>
<td align="center">100</td>
<td align="center">43.49</td>
</tr>
<tr>
<td align="center">SFR-GSN</td>
<td align="center">7</td>
<td align="center">
<bold>96.85</bold>
</td>
<td align="center">5</td>
<td align="center">
<bold>97.28</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bold values represent the best results among the column.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>From the table, it is shown that the proposed method significantly outperforms other methods in terms of F1 scores. Additionally, the proposed method provides fewer features than other methods, which indicates the proposed method could identify the biomarker more accurately.</p>
</sec>
<sec id="s3-4">
<title>3.4 SFR-GSN on cancer subtype-related biomarkers</title>
<p>Besides the evolutionary characteristics in different stages, cancer also exhibits various subtypes. As for LUAD, three types often occur in the evolution, which are Papillary Predominant (PP), Acinar Predominant (PI), and Trabecular (TRU). By studying subtype-related biomarkers, a better understanding of the differences in disease progression, treatment response, and prognosis among different subtypes could be obtained (<xref ref-type="bibr" rid="B17">Perou et al., 2000</xref>; <xref ref-type="bibr" rid="B14">Muller et al., 2022</xref>). Therefore, in order to enhance the generalization of our model, experiments on subtype-related data were conducted to identify the subtype-related biomarkers.</p>
<p>Firstly, the datasets were separated into the three subtypes and accordingly, three corresponding GSNs were constructed. Then, SFR was trained on the GSNs, features with CBR<inline-formula id="inf2">
<mml:math id="m6">
<mml:mo>&#x003e;</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:math>
</inline-formula> were obtained, and the F1 score and CBR were shown in <xref ref-type="table" rid="T4">Table 4</xref>. Finally, five features were identified as subtype-related biomarkers of LUAD, these are (HDAC6, SIRT2), (AKT2, RICTOR), (DHX33, PINX1), (SNAP29, TSNARE1) and (ASPSCR1, VCPIP1). Similarly, the BRCA datasets were divided into five groups due to the five subtypes of BRCA. Eventually, the results were shown in <xref ref-type="table" rid="T4">Table 4</xref>, where six features were screened as subtype-related biomarkers, these are (SRC, USP8), (IRAK4, TOLLIP), (SRC, TRAF6), (F8, SEC24C), (CDK12, SUPT5H) and (CDC40, SF3B2).</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>F1 score and CBR for multi-class classification in stages of LUAD and BRCA at different feature quantity thresholds.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center" colspan="3">LUAD</th>
<th align="center" colspan="3">BRCA</th>
</tr>
<tr>
<th align="center">Number of features</th>
<th align="center">F1 score(%)</th>
<th align="center">CBR</th>
<th align="center">Number of features</th>
<th align="center">F1 score(%)</th>
<th align="center">CBR</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">1</td>
<td align="center">73.2800</td>
<td align="center">-</td>
<td align="center">1</td>
<td align="center">51.2668</td>
<td align="center">-</td>
</tr>
<tr>
<td align="center">2</td>
<td align="center">91.3155</td>
<td align="center">18.0354</td>
<td align="center">2</td>
<td align="center">81.9457</td>
<td align="center">30.6788</td>
</tr>
<tr>
<td align="center">3</td>
<td align="center">95.9758</td>
<td align="center">4.1303</td>
<td align="center">3</td>
<td align="center">89.6415</td>
<td align="center">7.6958</td>
</tr>
<tr>
<td align="center">4</td>
<td align="center">96.8973</td>
<td align="center">0.5300</td>
<td align="center">4</td>
<td align="center">93.2176</td>
<td align="center">3.5760</td>
</tr>
<tr>
<td align="center">5</td>
<td align="center">97.3499</td>
<td align="center">0.9214</td>
<td align="center">5</td>
<td align="center">94.4989</td>
<td align="center">1.2813</td>
</tr>
<tr>
<td align="center">6</td>
<td align="center">97.7847</td>
<td align="center">0.4526</td>
<td align="center">6</td>
<td align="center">96.4546</td>
<td align="center">1.9557</td>
</tr>
<tr>
<td align="center">7</td>
<td align="center">97.7847</td>
<td align="center">0.4347</td>
<td align="center">7</td>
<td align="center">96.9095</td>
<td align="center">0.4549</td>
</tr>
<tr>
<td align="center">8</td>
<td align="center">97.7847</td>
<td align="center">0</td>
<td align="center">8</td>
<td align="center">97.3826</td>
<td align="center">0.4731</td>
</tr>
<tr>
<td align="center">9</td>
<td align="center">97.7847</td>
<td align="center">0</td>
<td align="center">9</td>
<td align="center">97.3919</td>
<td align="center">0.0093</td>
</tr>
<tr>
<td align="center">10</td>
<td align="center">
<bold>98.2410</bold>
</td>
<td align="center">0.4563</td>
<td align="center">10</td>
<td align="center">
<bold>97.7572</bold>
</td>
<td align="center">0.3652</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bold values represent the best results among the column.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Further, the enrichment analysis was conducted on the identified features. In the subtype-related biomarkers of LUAD, five genes were enriched in 10 pathway pathways, with the gene AKT2 was found in eight pathway pathways. In that of BRCA, eight genes were enriched in 24 pathways, with the gene TRAF6 being enriched in 21 pathways and the gene IRAK4 was found in 20 pathways. KEGG pathway enrichment results are shown in <xref ref-type="fig" rid="F6">Figure 6</xref>. Subsequently, the results of the GO enrichment analysis are shown in <xref ref-type="fig" rid="F7">Figure 7</xref>. The 10 genes in the LUAD subtypes are involved in 360 biological processes, while 11 genes in the BRCA subtypes are involved in 407 biological processes.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>KEGG pathway enrichment results of subtype biomarkers for LUAD and BRCA.</p>
</caption>
<graphic xlink:href="fgene-15-1407072-g006.tif"/>
</fig>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>GO enrichment results of subtype biomarkers for LUAD and BRCA.</p>
</caption>
<graphic xlink:href="fgene-15-1407072-g007.tif"/>
</fig>
<p>After providing the results of SFR-GSN on the identification, the proposed method was also compared with three conventional methods and all genes. The results are shown in <xref ref-type="table" rid="T5">Table 5</xref>. SFR-GSN gains the best performance and the least features, which suggests SFR-GSN exhibits superior capability in identifying subtype-related biomarkers.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>The comparison of identification in subtype-related biomarkers among SFR-GSN, three conventional methods, and all genes on LUAD and BRCA datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center"/>
<th align="center" colspan="2">LUAD</th>
<th align="center" colspan="2">BRCA</th>
</tr>
<tr>
<th align="center">Methods</th>
<th align="center">Number of features</th>
<th align="center">F1 score(%)</th>
<th align="center">Number of features</th>
<th align="center">F1 score(%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">All Genes</td>
<td align="center">1,3326</td>
<td align="center">72.73</td>
<td align="center">1,3168</td>
<td align="center">85.9674</td>
</tr>
<tr>
<td align="center">DEGs</td>
<td align="center">2,478</td>
<td align="center">82.38</td>
<td align="center">3,922</td>
<td align="center">87.77</td>
</tr>
<tr>
<td align="center">WGCNA</td>
<td align="center">426</td>
<td align="center">78.00</td>
<td align="center">632</td>
<td align="center">86.32</td>
</tr>
<tr>
<td align="center">Relife</td>
<td align="center">100</td>
<td align="center">81.16</td>
<td align="center">100</td>
<td align="center">83.45</td>
</tr>
<tr>
<td align="center">SFR-GSN</td>
<td align="center">5</td>
<td align="center">
<bold>96.89</bold>
</td>
<td align="center">6</td>
<td align="center">
<bold>96.45</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bold values represent the best results among the column.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec id="s4" sec-type="conclusion">
<title>4 Conclusion</title>
<p>In this work, a novel method called SFR-GSN has been proposed to identify the stage-related biomarkers, which gained remarkable results on LUAD and BRCA datasets. First, the clinical data, RNA-Seq data, and PPI were collected. Second, according to the pathological stage, the GSNs were constructed by combining the SSN and PPI. Third, based on GSNs, a two-round SFR was conducted, which firstly used random forest to pre-screen and later used RFECV to obtain the optimal feature sets. The CBR was introduced to assist in identifying stage-related biomarkers.</p>
<p>Finally, the results of the proposed method showed that the identified biomarkers were highly stage-specific and significantly improved the F1 scores for cancer stage prediction. For the lung adenocarcinoma dataset, the F1 score reached 97.68%, and for the breast cancer dataset, it achieved 96.87%. The results outperform the other conventional methods on both accuracy and F1 scores. Moreover, the enrichment analysis of biomarkers was conducted to validate the effectiveness of the proposed method in view of biological functions. The proposed method exhibits superior performance in identifying subtype-related biomarkers. The proposed method could be applied to other cancers to offer new insight into cancer treatment prognosis.</p>
</sec>
</body>
<back>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The RNA-Seq data presented in the study are deposited in the UCSC Xena repository, accession number TCGA Lung Adenocarcinoma (LUAD) and TCGA Breast Cancer (BRCA), the url is <ext-link ext-link-type="uri" xlink:href="https://tcga-xena-hub.s3.us-east-1.amazonaws.com/download/TCGA.BRCA.sampleMap%2FHiSeqV2.gz">https://tcga-xena-hub.s3.us-east-1.amazonaws.com/download/TCGA.BRCA.sampleMap%2FHiSeqV2.gz</ext-link>; the PPI data presented in the study are deposited in the STRING repository, accession number Homo sapiens, the url is <ext-link ext-link-type="uri" xlink:href="https://stringdb-downloads.org/download/protein.physical.links.v12.0/9606.protein.physical.links.v12.0.txt.gz">https://stringdb-downloads.org/download/protein.physical.links.v12.0/9606.protein.physical.links.v12.0.txt.gz</ext-link>.</p>
</sec>
<sec id="s6">
<title>Author contributions</title>
<p>BC: Data curation, Formal Analysis, Funding acquisition, Methodology, Supervision, Writing&#x2013;review and editing, Writing&#x2013;original draft. YW: Data curation, Formal Analysis, Methodology, Validation, Visualization, Writing&#x2013;original draft. JZ: Methodology, Visualization, Writing&#x2013;review and editing. YH: Methodology, Writing&#x2013;review and editing. HB: Methodology, Writing&#x2013;review and editing. Jun Bian: Supervision, Writing&#x2013;review and editing. RK: Supervision, Writing&#x2013;review and editing. XS: Supervision, Writing&#x2013;review and editing.</p>
</sec>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This work was supported by the National Key R&#x26;D Program of China under Grant No. 2021YFA1000402, the National Natural Science Foundation of China under Grant No. 61972320, and Xi&#x2019;an municipal bureau of science and technology under Grant No. 22YXYJ0057.</p>
</sec>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>RK was employed by Rewise (Hangzhou) Information Technology Co., Ltd.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s9" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Acharjee</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Larkman</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cardoso</surname>
<given-names>V. R.</given-names>
</name>
<name>
<surname>Gkoutos</surname>
<given-names>G. V.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A random forest based biomarker discovery and power analysis framework for diagnostics research</article-title>. <source>BMC Med. genomics</source> <volume>13</volume>, <fpage>178</fpage>. <pub-id pub-id-type="doi">10.1186/s12920-020-00826-6</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Amin</surname>
<given-names>D. N.</given-names>
</name>
<name>
<surname>Ngoyi</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Nhkwachi</surname>
<given-names>G.-M.</given-names>
</name>
<name>
<surname>Palomba</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Rottenberg</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>B&#xfc;scher</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>Identification of stage biomarkers for human african trypanosomiasis</article-title>. <source>Am. J. Trop. Med. Hyg.</source> <volume>82</volume>, <fpage>983</fpage>&#x2013;<lpage>990</lpage>. <pub-id pub-id-type="doi">10.4269/ajtmh.2010.09-0770</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Burrell</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>McGranahan</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Bartek</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Swanton</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>The causes and consequences of genetic heterogeneity in cancer evolution</article-title>. <source>Nature</source> <volume>501</volume>, <fpage>338</fpage>&#x2013;<lpage>345</lpage>. <pub-id pub-id-type="doi">10.1038/nature12625</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>De Picker</surname>
<given-names>L. J.</given-names>
</name>
<name>
<surname>Haarman</surname>
<given-names>B. C.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Applicability, potential and limitations of tspo pet imaging as a clinical immunopsychiatry biomarker</article-title>. <source>Eur. J. Nucl. Med. Mol. imaging</source> <volume>49</volume>, <fpage>164</fpage>&#x2013;<lpage>173</lpage>. <pub-id pub-id-type="doi">10.1007/s00259-021-05308-0</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Deva Magendhra Rao</surname>
<given-names>A. K.</given-names>
</name>
<name>
<surname>Patel</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Korivi Jyothiraj</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Meenakumari</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Sundersingh</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sridevi</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Identification of lnc rna s associated with early-stage breast cancer and their prognostic implications</article-title>. <source>Mol. Oncol.</source> <volume>13</volume>, <fpage>1342</fpage>&#x2013;<lpage>1355</lpage>. <pub-id pub-id-type="doi">10.1002/1878-0261.12489</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Harris</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Clark</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ireland</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lomax</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ashburner</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Foulger</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2004</year>). <article-title>The gene ontology (go) database and informatics resource</article-title>. <source>Nucleic acids Res.</source> <volume>32</volume>, <fpage>D258</fpage>&#x2013;<lpage>D261</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkh036</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Holland</surname>
<given-names>R. L.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>What makes a good biomarker?</article-title> <source>Adv. Precis. Med.</source> <volume>1</volume>, <fpage>66</fpage>. <pub-id pub-id-type="doi">10.18063/apm.2016.01.007</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Horvath</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2011</year>) <source>Weighted network analysis: applications in genomics and systems biology</source>. <publisher-name>Springer Science and Business Media</publisher-name>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jin</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>N. D.</given-names>
</name>
<name>
<surname>Talos</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Ecmarker: interpretable machine learning model identifies gene expression biomarkers predicting clinical outcomes and reveals molecular mechanisms of human disease in early stages</article-title>. <source>Bioinformatics</source> <volume>37</volume>, <fpage>1115</fpage>&#x2013;<lpage>1124</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa935</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Law</surname>
<given-names>C. W.</given-names>
</name>
<name>
<surname>Alhamdoosh</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Smyth</surname>
<given-names>G. K.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Rna-seq analysis is easy as 1-2-3 with limma, glimma and edger</article-title>. <source>F1000Research</source> <volume>5</volume>, <fpage>1408</fpage>. <pub-id pub-id-type="doi">10.12688/f1000research.9005.2</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ji</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Aihara</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Personalized characterization of diseases using sample-specific networks</article-title>. <source>Nucleic acids Res.</source> <volume>44</volume>, <fpage>e164</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkw772</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Love</surname>
<given-names>M. I.</given-names>
</name>
<name>
<surname>Huber</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Anders</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Moderated estimation of fold change and dispersion for rna-seq data with deseq2</article-title>. <source>Genome Biol.</source> <volume>15</volume>, <fpage>550</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-014-0550-8</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moein</surname>
<given-names>S. T.</given-names>
</name>
<name>
<surname>Hashemian</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Mansourafshar</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Khorram-Tousi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tabarsi</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Doty</surname>
<given-names>R. L.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Smell dysfunction: a biomarker for covid-19</article-title>. <source>Int. forum allergy and rhinology</source> <volume>10</volume>, <fpage>944</fpage>&#x2013;<lpage>950</lpage>. <pub-id pub-id-type="doi">10.1002/alr.22587</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Muller</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Joms</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Tozbikian</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>What&#x2019;s new in breast pathology 2022: who 5th edition and biomarker updates</article-title>. <source>J. pathology Transl. Med.</source> <volume>56</volume>, <fpage>170</fpage>&#x2013;<lpage>171</lpage>. <pub-id pub-id-type="doi">10.4132/jptm.2022.04.25</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ogata</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Goto</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sato</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Fujibuchi</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Bono</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Kanehisa</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>1999</year>). <article-title>Kegg: Kyoto encyclopedia of genes and genomes</article-title>. <source>Nucleic acids Res.</source> <volume>27</volume>, <fpage>29</fpage>&#x2013;<lpage>34</lpage>. <pub-id pub-id-type="doi">10.1093/nar/27.1.29</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Patil</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Bellary</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Machine learning approach in melanoma cancer stage detection</article-title>. <source>J. King Saud University-Computer Inf. Sci.</source> <volume>34</volume>, <fpage>3285</fpage>&#x2013;<lpage>3293</lpage>. <pub-id pub-id-type="doi">10.1016/j.jksuci.2020.09.002</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Perou</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>S&#xf8;rlie</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Eisen</surname>
<given-names>M. B.</given-names>
</name>
<name>
<surname>Van De Rijn</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Jeffrey</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Rees</surname>
<given-names>C. A.</given-names>
</name>
<etal/>
</person-group> (<year>2000</year>). <article-title>Molecular portraits of human breast tumours</article-title>. <source>nature</source> <volume>406</volume>, <fpage>747</fpage>&#x2013;<lpage>752</lpage>. <pub-id pub-id-type="doi">10.1038/35021093</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Robnik-&#x160;ikonja</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kononenko</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Theoretical and empirical analysis of relieff and rrelieff</article-title>. <source>Mach. Learn.</source> <volume>53</volume>, <fpage>23</fpage>&#x2013;<lpage>69</lpage>. <pub-id pub-id-type="doi">10.1023/a:1025667309714</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Identification of genes involved in the four stages of colorectal cancer: gene expression profiling</article-title>. <source>Mol. Cell. probes</source> <volume>37</volume>, <fpage>39</fpage>&#x2013;<lpage>47</lpage>. <pub-id pub-id-type="doi">10.1016/j.mcp.2017.11.004</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sung</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ferlay</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Siegel</surname>
<given-names>R. L.</given-names>
</name>
<name>
<surname>Laversanne</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Soerjomataram</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Jemal</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Global cancer statistics 2020: globocan estimates of incidence and mortality worldwide for 36 cancers in 185 countries</article-title>. <source>CA a cancer J. Clin.</source> <volume>71</volume>, <fpage>209</fpage>&#x2013;<lpage>249</lpage>. <pub-id pub-id-type="doi">10.3322/caac.21660</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Szklarczyk</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Kirsch</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Koutrouli</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Nastou</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Mehryary</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Hachilif</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>The string database in 2023: protein&#x2013;protein association networks and functional enrichment analyses for any sequenced genome of interest</article-title>. <source>Nucleic acids Res.</source> <volume>51</volume>, <fpage>D638</fpage>&#x2013;<lpage>D646</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkac1000</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tomczak</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Czerwi&#x144;ska</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Wiznerowicz</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Review the cancer genome atlas (tcga): an immeasurable source of knowledge</article-title>. <source>Contemp. Oncology/Wsp&#xf3;&#x142;czesna Onkol.</source> <volume>2015</volume>, <fpage>68</fpage>&#x2013;<lpage>77</lpage>. <pub-id pub-id-type="doi">10.5114/wo.2014.47136</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ubaldi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Valenti</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Borgese</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Collura</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Fantacci</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ferrera</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Strategies to develop radiomics and machine learning models for lung cancer stage and histology prediction using small data samples</article-title>. <source>Phys. Medica</source> <volume>90</volume>, <fpage>13</fpage>&#x2013;<lpage>22</lpage>. <pub-id pub-id-type="doi">10.1016/j.ejmp.2021.08.015</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Van der Kloet</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Tempels</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Ismail</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Van der Heijden</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Kasper</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Rojas-Cherto</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Discovery of early-stage biomarkers for diabetic kidney disease using ms-based metabolomics (finndiane study)</article-title>. <source>Metabolomics</source> <volume>8</volume>, <fpage>109</fpage>&#x2013;<lpage>119</lpage>. <pub-id pub-id-type="doi">10.1007/s11306-011-0291-6</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Huo</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Analysis of gene expression profiles of non-small cell lung cancer at different stages reveals significantly altered biological functions and candidate genes</article-title>. <source>Oncol. Rep.</source> <volume>37</volume>, <fpage>1736</fpage>&#x2013;<lpage>1746</lpage>. <pub-id pub-id-type="doi">10.3892/or.2017.5380</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xiong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Ucscxenashiny: an r/cran package for interactive analysis of ucsc xena data</article-title>. <source>Bioinformatics</source> <volume>38</volume>, <fpage>527</fpage>&#x2013;<lpage>529</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btab561</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ye</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Jing</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Mills</surname>
<given-names>G. B.</given-names>
</name>
<name>
<surname>Diao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Sex-associated molecular differences for cancer immunotherapy</article-title>. <source>Nat. Commun.</source> <volume>11</volume>, <fpage>1779</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-020-15679-x</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>