<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1532651</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2025.1532651</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A semi-supervised weighted SPCA- and convolution KAN-based model for drug response prediction</article-title>
<alt-title alt-title-type="left-running-head">Miao et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2025.1532651">10.3389/fgene.2025.1532651</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes" equal-contrib="yes">
<name>
<surname>Miao</surname>
<given-names>Rui</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1441450/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Zhong</surname>
<given-names>Bing-Jie</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Mei</surname>
<given-names>Xin-Yue</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Dong</surname>
<given-names>Xin</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ou</surname>
<given-names>Yang-Dong</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liang</surname>
<given-names>Yong</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1666077/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yu</surname>
<given-names>Hao-Yang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Ying</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Dong</surname>
<given-names>Zi-Han</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Basic Teaching Department</institution>, <institution>Zhuhai Campus of Zunyi Medical University</institution>, <addr-line>Zhu Hai</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Institute of Systems Engineering</institution>, <institution>Macau University of Science and Technology</institution>, <addr-line>Macau</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>School of Biomedical Engineering</institution>, <institution>Guangdong Medical University</institution>, <addr-line>Dongguan</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Peng Cheng Laboratory</institution>, <addr-line>Shenzhen</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/565086/overview">Lin Zhang</ext-link>, China University of Mining and Technology, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1233850/overview">Xiaobo Sun</ext-link>, Zhongnan University of Economics and Law, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1772675/overview">Conghao Wang</ext-link>, Nanyang Technological University, Singapore</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Rui Miao, <email>miaorui.research@gmail.com</email>
</corresp>
<fn fn-type="equal" id="fn001">
<label>
<sup>&#x2020;</sup>
</label>
<p>These authors have contributed equally to this work</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>21</day>
<month>03</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1532651</elocation-id>
<history>
<date date-type="received">
<day>02</day>
<month>12</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>24</day>
<month>02</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Miao, Zhong, Mei, Dong, Ou, Liang, Yu, Wang and Dong.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Miao, Zhong, Mei, Dong, Ou, Liang, Yu, Wang and Dong</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Motivation</title>
<p>Predicting the response of cell lines to characteristic drugs based on multi-omics gene information has become the core problem of precision oncology. At present, drug response prediction using multi-omics gene data faces the following three main challenges: first, how to design a gene probe feature extraction model with biological interpretation and high performance; second, how to develop multi-omics weighting modules for reasonably fusing genetic data of different lengths and noise conditions; third, how to construct deep learning models that can handle small sample sizes while minimizing the risk of possible overfitting.</p>
</sec>
<sec>
<title>Results</title>
<p>We propose an innovative drug response prediction model (NMDP). First, the NMDP model introduces an interpretable semi-supervised weighted SPCA module to solve the feature extraction problem in multi-omics gene data. Next, we construct a multi-omics data fusion framework based on sample similarity networks, bimodal tests, and variance information, which solves the data fusion problem and enables the NMDP model to focus on more relevant genomic data. Finally, we combine a one-dimensional convolution method and Kolmogorov&#x2013;Arnold networks (KANs) to predict the drug response. We conduct five sets of real data experiments and compare NMDP against seven advanced drug response prediction methods. The results show that NMDP achieves the best performance, with sensitivity and specificity reaching 0.92 and 0.93, respectively&#x2014;an improvement of 11%&#x2013;57% compared to other models. Bio-enrichment experiments strongly support the biological interpretation of the NMDP model and its ability to identify potential targets for drug activity prediction.</p>
</sec>
</abstract>
<kwd-group>
<kwd>drug response prediction</kwd>
<kwd>feature extraction</kwd>
<kwd>sparse PCA</kwd>
<kwd>Kolmogorov&#x2013;Arnold networks</kwd>
<kwd>data fusion</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Genomics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Precision oncology aims to leverage genomic information to identify patient groups with similar biological traits, enabling the delivery of the most suitable treatments (<xref ref-type="bibr" rid="B16">Dlamini et al., 2020</xref>; <xref ref-type="bibr" rid="B21">Garraway et al., 2013</xref>; <xref ref-type="bibr" rid="B24">Hodson, 2020</xref>; <xref ref-type="bibr" rid="B35">Prasad, 2016</xref>; <xref ref-type="bibr" rid="B36">Prasad et al., 2016</xref>). In clinical applications, this approach generally involves choosing targeted therapies based on the individual genomic profiles of patients (<xref ref-type="bibr" rid="B3">Ballester and Carmona, 2021</xref>). However, research reveals that only approximately 9% of patients experience effective outcomes from such targeted treatments, which greatly restricts the broad applicability of precision oncology (<xref ref-type="bibr" rid="B7">Barretina et al., 2012a</xref>; <xref ref-type="bibr" rid="B38">Rubio-Perez et al., 2015</xref>). Moreover, limited drug response prediction models for non-specific therapies mean that many patients miss out on the benefits of precision oncology and may even receive ineffective treatments. Fortunately, data from extensive pharmacogenomic screenings have shown that nearly all cancer cell lines and patient-derived xenografts (PDXs) respond to some form of targeted therapy or non-specific chemotherapy (<xref ref-type="bibr" rid="B8">Barretina et al., 2012b</xref>; <xref ref-type="bibr" rid="B19">Gao et al., 2015</xref>; <xref ref-type="bibr" rid="B20">Garnett et al., 2012</xref>). Thus, a primary challenge now is accurately aligning cancer patients with treatments that match their unique drug response profiles.</p>
<p>Currently, a significant research focus is predicting drug responses in cancer patients using single genomics data (<xref ref-type="bibr" rid="B1">Adam et al., 2020</xref>; <xref ref-type="bibr" rid="B17">Dong et al., 2015</xref>; <xref ref-type="bibr" rid="B18">Firoozbakht et al., 2022</xref>; <xref ref-type="bibr" rid="B42">Sheng et al., 2015</xref>). For instance, as demonstrated by Geeleher et al., a ridge regression model that utilizes gene expression data from the Genomics of Drug Sensitivity in Cancer (GDSC) database has shown effective application to clinical trial datasets for drugs including erlotinib, cisplatin, docetaxel, and bortezomib. The study also found that incorporating data from cancer cell lines other than breast cancer can improve the predictive performance of the docetaxel drug response model (<xref ref-type="bibr" rid="B22">Geeleher et al., 2014</xref>). Moreover, our preliminary research indicates that combining statistical methods based on individual genomic information from patients with machine learning techniques can construct highly performant drug response prediction models (<xref ref-type="bibr" rid="B27">Miao et al., 2020</xref>; <xref ref-type="bibr" rid="B49">Zheng et al., 2024</xref>; <xref ref-type="bibr" rid="B41">Sharma et al., 2024</xref>).</p>
<p>Recently, the increasing availability of multi-omics datasets for drug response has opened new avenues for machine learning models, enabling a deeper understanding of biological processes. Multi-omics data have shown notable success across various bioinformatics tasks, including survival prediction, cancer subtype classification, and target gene identification (<xref ref-type="bibr" rid="B46">Xu et al., 2024</xref>). As deep learning continues to progress rapidly, constructing predictive models that utilize multi-omics data through deep learning techniques becomes a primary research focus. Several multi-omics drug response models have been developed (<xref ref-type="bibr" rid="B4">Ballester et al., 2022</xref>; <xref ref-type="bibr" rid="B6">Baptista et al., 2021</xref>; <xref ref-type="bibr" rid="B11">Chen and Zhang, 2021</xref>; <xref ref-type="bibr" rid="B50">Zhou et al., 2024</xref>; <xref ref-type="bibr" rid="B37">Rashid, 2024</xref>; <xref ref-type="bibr" rid="B5">Baptista and Ferreira, 2023</xref>). For instance, Chiu et al. developed a deep learning model that utilizes autoencoders to combine diverse omics features for drug response prediction (<xref ref-type="bibr" rid="B13">Chiu et al., 2019</xref>). Similarly, Hossein et al. proposed a model that employs deep neural network fusion, combining hidden layer representations from different multi-omics networks to synthesize feature information effectively (<xref ref-type="bibr" rid="B39">Sharifi-Noghabi et al., 2019</xref>). Peng et al. proposed a two-space graph convolutional neural network (TSGCNN) that combines cell line and drug feature spaces to predict drug responses by leveraging both homogeneous and heterogeneous relationships (<xref ref-type="bibr" rid="B34">Peng et al., 2023</xref>). Similarly, Trac et al. proposed a GCN-based drug response prediction model for acute myeloid leukemia (AML), highlighting the versatility of graph-based neural networks in oncology research (<xref ref-type="bibr" rid="B43">Trac et al., 2023</xref>). Wang et al. proposed MOICVAE, a deep learning model that integrates multi-omics data using a variational autoencoder to improve drug sensitivity prediction (<xref ref-type="bibr" rid="B45">Wang et al., 2023</xref>). Meanwhile, Sharma et al. proposed DeepInsight-3D, architecture to fuse multi-omics data for anticancer drug response prediction, offering an advanced deep learning perspective for modeling complex interactions in diverse biological datasets (<xref ref-type="bibr" rid="B40">Sharma et al., 2023</xref>).</p>
<p>Currently, the multi-omics drug response prediction model faces three major challenges. First, genomic data typically involve small sample sizes, which increases the likelihood of overfitting in existing models (<xref ref-type="bibr" rid="B14">Deng et al., 2023</xref>). Developing an efficient and biologically interpretable feature selection method to select key genomic data is the first major challenge currently faced (<xref ref-type="bibr" rid="B14">Deng et al., 2023</xref>). Second, most genomic datasets for drug response prediction contain multiple independent genomic data types (<xref ref-type="bibr" rid="B30">Munquad et al., 2024</xref>). The data lengths and noise levels of these genomic datasets vary significantly, making the rational design of the multi-omics fusion method the second major challenge in constructing high-performance drug response medical models. Third, considering that drug response prediction is a complex biological problem and the dataset has only limited training samples, constructing a sufficiently high-performance prediction model based on a small sample of data remains the third major challenge.</p>
<p>In this study, we introduce an innovative model for predicting drug response (NMDP, <xref ref-type="fig" rid="F1">Figure 1</xref>). The NMDP model is composed of four main modules. 1) Key genome selection module: we propose an interpretable, semi-supervised weighted sparse PCA to identify essential biological features. 2) Similarity network construction module: this module addresses the challenge of aligning data across different omics. 3) Data fusion module: we introduce a weighted similarity network fusion approach, incorporating the dip test method and variance information. 4) Drug response prediction module: we integrate one-dimensional convolutional neural networks (CNNs) and the Kolmogorov&#x2013;Arnold network (KAN) method.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Overview of the NMDP model workflow. <bold>(A)</bold> Multi-omics data input. <bold>(B)</bold> Semi-supervised weighted sparse PCA. <bold>(C)</bold> Similarity calculation module. <bold>(D)</bold> Construct of the fusion feature matrix. <bold>(E)</bold> Prediction model. <bold>(F)</bold> Output.</p>
</caption>
<graphic xlink:href="fgene-16-1532651-g001.tif"/>
</fig>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<sec id="s2-1">
<title>2.1 Datasets</title>
<p>In this study, we use publicly available datasets to extract drug response and genomic data from cell lines. The first dataset is Genomics of Drug Sensitivity in Cancer (GDSC), which provides extensive data on drug response measurements. The second is the Gene Expression Omnibus (GEO) and Cell Model Passports, along with the European Bioinformatics Institute (EMBL-EBI) datasets. These two datasets provide the genomic data needed for this experiment.</p>
<p>It is important to note that the GDSC dataset comprises 624 unique drugs, 576,758 IC_50 values, and 978 cell lines. Genomic characteristics for each cell line include somatic copy number alterations (SCNAs) across 21,878 genes, RNA-Seq expression levels for 44,421 probes, and methylation levels for 365,860 CpG sites. For our study, we select 68 drugs: 14 FDA-approved targeted therapies, 49 drugs with known target genes not yet FDA-approved, and 5 nonspecific treatments (<xref ref-type="sec" rid="s11">Supplementary Tables S1&#x2013;S3</xref>).</p>
</sec>
<sec id="s2-2">
<title>2.2 Dataset of gene pathway data</title>
<p>The pathway data used in this study are sourced from the Pathway Commons database, which contains commonly used pathway datasets such as the Kyoto Encyclopedia of Genes and Genomes (KEGG) and Gene Ontology (GO).</p>
</sec>
<sec id="s2-3">
<title>2.3 Drug response data</title>
<p>In addition to the preprocessing already performed by the provider of this dataset, we also perform additional preprocessing. The following are the steps and criteria of our preprocessing: first, we remove samples with certain missing data, such as samples with a missing rate of more than 10%; second, we remove drugs with limited IC_50 test information, requiring the amount of IC_50 test data for each drug to be no less than 200 samples. Third, we use waterfall distribution to divide drug response data (<xref ref-type="bibr" rid="B15">Ding et al., 2018</xref>). Waterfall distribution is a method that sorts drugs based on their IC_50 values and uses a linear model to fit the data, which is used to determine whether a drug is effective. Specifically, the drugs are sorted according to the true IC_50 information. A linear model is then constructed to fit the distribution, and Pearson&#x2019;s coefficient is used to evaluate the degree of fit of the model. If the fit is higher than 0.95, then the median is chosen as the cut-off point. If the fit is less than 0.95, a new monadic linear function is created, and the parameters of the function are determined by the smallest and largest points of IC_50. Finally, the point furthest away from the unary linear function in the IC_50 curve is calculated as the demarcation point. Ultimately, we classify divide drugs into two categories: responsive and non-responsive. In addition, to ensure that the data are balanced, we ensure that the response group constitutes at least 25% of the total data.</p>
</sec>
<sec id="s2-4">
<title>2.4 Methods</title>
<p>In this section, we provide a detailed overview of the architecture and algorithm flow of the NMDP model. This NMDP method transforms the sparse PCA model from a non-supervised to a semi-supervised approach, improving the ability of feature selection (key genome selection module). Second: Similarity network building module: in this module, we construct a sample similarity network based on the Spearman and Kendall correlation coefficients. Third: Data fusion module: we develop a data fusion algorithm based on the dip and variance tests. Fourth: Drug response prediction module: in this module, we propose a drug response prediction model based on one-dimensional convolution and KANs.</p>
<sec id="s2-4-1">
<title>2.4.1 Key genome selection module</title>
<sec id="s2-4-1-1">
<title>2.4.1.1 ESPCA method</title>
<p>Before introducing the NMDP model, we first define the sparse PCA (SPCA) and edge sparse PCA (ESPCA) models. Suppose we have an <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> feature matrix <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the number of samples and <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the number of gene probes. The definition of SPCA is given by <xref ref-type="disp-formula" rid="e1">Formula 1</xref>:<disp-formula id="e1">
<mml:math id="m5">
<mml:mrow>
<mml:munder>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>z</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munder>
<mml:msup>
<mml:mi>u</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mi>X</mml:mi>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>.</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>Here, <inline-formula id="inf5">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf6">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represent <inline-formula id="inf7">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf8">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> norms, respectively. <inline-formula id="inf9">
<mml:math id="m10">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents principal component (PC) loading, which has the dimension as the number of gene probes. <inline-formula id="inf10">
<mml:math id="m11">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the retention number of gene probes. In most cases, the SVD method is used to solve <xref ref-type="disp-formula" rid="e1">Formula 1</xref>. Therefore, the formula can also be written as <xref ref-type="disp-formula" rid="e2">Formula 2</xref>:<disp-formula id="e2">
<mml:math id="m12">
<mml:mrow>
<mml:munder>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>z</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munder>
<mml:msup>
<mml:mi>u</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mi>X</mml:mi>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>.</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo>&#x2225;</mml:mo>
<mml:mi>u</mml:mi>
<mml:mo>&#x2225;</mml:mo>
</mml:mrow>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
</p>
<p>In this case, <inline-formula id="inf11">
<mml:math id="m13">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the weight information corresponding to the sample, with dimensions matching the number of samples. ESPCA builds upon SPCA by incorporating improvements. Its main contribution is the integration of pathway structure information from the genome as <italic>a priori</italic> knowledge. Suppose that the known pathway structure information (edge set) is represented as <inline-formula id="inf13">
<mml:math id="m15">
<mml:mrow>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. At this point, the researcher introduces <inline-formula id="inf14">
<mml:math id="m16">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> regulon, which is represented as <xref ref-type="disp-formula" rid="e3">Formula 3</xref>:<disp-formula id="e3">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:munder>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>z</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2200;</mml:mo>
<mml:msup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2286;</mml:mo>
<mml:mi>V</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:munder>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="&#x7c;">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>Here, <inline-formula id="inf15">
<mml:math id="m18">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">G</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf16">
<mml:math id="m19">
<mml:mrow>
<mml:mi>V</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the vertex set derived from the <inline-formula id="inf17">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> regulon. Therefore, ESPCA can also be represented as <xref ref-type="disp-formula" rid="e4">Formula 4</xref> (<xref ref-type="bibr" rid="B29">Min et al., 2018</xref>):<disp-formula id="e4">
<mml:math id="m21">
<mml:mrow>
<mml:munder>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>z</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munder>
<mml:msup>
<mml:mi>u</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mi>X</mml:mi>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>.</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo>&#x2225;</mml:mo>
<mml:mi>u</mml:mi>
<mml:mo>&#x2225;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
</p>
</sec>
<sec id="s2-4-1-2">
<title>2.4.1.2 Semi-supervised weighted edge sparse PCA</title>
<p>The existing SPCA and ESPCA methods are pure non-supervised methods; this method has a great advantage in data analysis with small samples and high dimensions. However, two primary issues arise: first, the method cannot utilize existing grouping information, which may reduce its effectiveness. Second, for the problem of drug response, the existing sparse PCA method selects the exact same key gene probe for all types of drugs; it obviously does not accord with the common sense of biology. In this study, we propose a novel semi-supervised weighted edge sparse PCA. This method mainly includes a weighted parameter <inline-formula id="inf18">
<mml:math id="m22">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, which is calculated using a machine learning model. The parameter <inline-formula id="inf19">
<mml:math id="m23">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> leverages known grouping information on drug responses. Each time the model completes a cycle, we calculate <inline-formula id="inf20">
<mml:math id="m24">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> based on the currently selected key gene probes and weight <inline-formula id="inf21">
<mml:math id="m25">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Finally, we can select different key gene probes for each drug. The specific steps are shown in <xref ref-type="disp-formula" rid="e5">Formulas 5</xref>&#x2013;<xref ref-type="disp-formula" rid="e12">12</xref>.</p>
<p>In general, the semi-supervised weighted edge sparse PCA method proposed in this paper can be expressed as <xref ref-type="disp-formula" rid="e5">Formula 5</xref>:<disp-formula id="e5">
<mml:math id="m26">
<mml:mrow>
<mml:munder>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>z</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munder>
<mml:msup>
<mml:mi>u</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mi>X</mml:mi>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>.</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo>&#x2225;</mml:mo>
<mml:mi>u</mml:mi>
<mml:mo>&#x2225;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
<p>Here, <inline-formula id="inf22">
<mml:math id="m27">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is a sparse regulon representing the edge group proposed by ESPCA and <inline-formula id="inf23">
<mml:math id="m28">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the regularization parameter. The regulon is given by <xref ref-type="disp-formula" rid="e6">Formula 6</xref>:<disp-formula id="e6">
<mml:math id="m29">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:munder>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>z</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2200;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>w</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>w</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2286;</mml:mo>
<mml:mi>V</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>w</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:munder>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>w</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
</p>
<p>Here, <inline-formula id="inf24">
<mml:math id="m30">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>w</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> represents a subset of vertices selected from the edge set, with <inline-formula id="inf25">
<mml:math id="m31">
<mml:mrow>
<mml:mfenced open="|" close="|" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>w</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> representing the count of vertices within this subset. Additionally, support <inline-formula id="inf26">
<mml:math id="m32">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> represents the collection of non-zero elements in the sparse vector <inline-formula id="inf27">
<mml:math id="m33">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Then, we specifically explain how to calculate <inline-formula id="inf28">
<mml:math id="m34">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>w</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, supposing <inline-formula id="inf29">
<mml:math id="m35">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>h</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mi>m</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. At the beginning of the algorithm, <inline-formula id="inf30">
<mml:math id="m36">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is randomly initialized. We use <inline-formula id="inf31">
<mml:math id="m37">
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>X</mml:mi>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to calculate the weight of <inline-formula id="inf32">
<mml:math id="m38">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Based on <inline-formula id="inf33">
<mml:math id="m39">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, we use <xref ref-type="disp-formula" rid="e7">Formula 7</xref> to calculate the edge weight <inline-formula id="inf34">
<mml:math id="m40">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>h</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> corresponding to <inline-formula id="inf35">
<mml:math id="m41">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>h</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>:<disp-formula id="e7">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>h</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:msubsup>
<mml:mi>u</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mi>u</mml:mi>
<mml:mi>j</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:msqrt>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
</p>
<p>Finally, the edge weight can be represented as <inline-formula id="inf36">
<mml:math id="m43">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>w</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi mathvariant="normal">h</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi>l</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>. In this paper, we used a greedy principle based on the random sampling method, previously developed by our team, to sparsify <inline-formula id="inf37">
<mml:math id="m44">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, as represented in <xref ref-type="disp-formula" rid="e8">Formula 8</xref> (<xref ref-type="bibr" rid="B28">Miao et al., 2022</xref>):<disp-formula id="e8">
<mml:math id="m45">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">P</mml:mi>
<mml:msub>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>w</mml:mi>
</mml:msub>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="" separators="&#x7c;">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>w</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2229;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2260;</mml:mo>
<mml:mo>&#x2205;</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>
</p>
<p>Here, <inline-formula id="inf38">
<mml:math id="m46">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">P</mml:mi>
<mml:mi mathvariant="script">G</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the sparse projection, with <inline-formula id="inf39">
<mml:math id="m47">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">P</mml:mi>
<mml:msub>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf40">
<mml:math id="m48">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf41">
<mml:math id="m49">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the random selection of k elements from the set <inline-formula id="inf42">
<mml:math id="m50">
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf43">
<mml:math id="m51">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the number of non-zero elements selected in the sparsification process. If gene <inline-formula id="inf44">
<mml:math id="m52">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is selected, then <inline-formula id="inf45">
<mml:math id="m53">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">P</mml:mi>
<mml:mi mathvariant="script">G</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>; otherwise, <inline-formula id="inf46">
<mml:math id="m54">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">P</mml:mi>
<mml:mi mathvariant="script">G</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>In this case, we can obtain a sparse gene weight vector <inline-formula id="inf47">
<mml:math id="m55">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">P</mml:mi>
<mml:msub>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>w</mml:mi>
</mml:msub>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Since existing sparse PCA models are non-supervised, identical input gene expression information results in the same <inline-formula id="inf48">
<mml:math id="m56">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> for each drug. In order to find a more suitable key gene set for different drugs, we design a linear evaluator based on machine learning, denoted as <inline-formula id="inf49">
<mml:math id="m57">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. For example, linear models or random forests can be used as evaluators. Here, <inline-formula id="inf50">
<mml:math id="m58">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the parameterized model, while the classification label corresponds to the drug response grouping information. Each time the sparse PCA model completes a cycle, we extract a new genome key expression matrix <inline-formula id="inf52">
<mml:math id="m60">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>X</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf53">
<mml:math id="m61">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>X</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> based on <inline-formula id="inf54">
<mml:math id="m62">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf55">
<mml:math id="m63">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the number of non-zero gene probes contained in <inline-formula id="inf56">
<mml:math id="m64">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> at that time. Next, <inline-formula id="inf57">
<mml:math id="m65">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>X</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> is input into <inline-formula id="inf58">
<mml:math id="m66">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula>, as represented in <xref ref-type="disp-formula" rid="e9">Formula 9</xref>:<disp-formula id="e9">
<mml:math id="m67">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:munder>
<mml:mrow>
<mml:mi>arg</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>min</mml:mi>
</mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:munder>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>X</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>;</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>Here, <inline-formula id="inf59">
<mml:math id="m68">
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the loss function. <inline-formula id="inf60">
<mml:math id="m69">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the optimizer of the model. <inline-formula id="inf61">
<mml:math id="m70">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> represents the parameter after model training. Once training is complete, an importance score <inline-formula id="inf62">
<mml:math id="m71">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is calculated for each gene probe associated with <inline-formula id="inf63">
<mml:math id="m72">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>X</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula>. Finally, we obtain <inline-formula id="inf64">
<mml:math id="m73">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. In order to ensure the stability in the weighting process, we perform a normalization step on <inline-formula id="inf65">
<mml:math id="m74">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>,scaling the values to the range <inline-formula id="inf66">
<mml:math id="m75">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&#x2013;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. Finally, we update <inline-formula id="inf67">
<mml:math id="m76">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> based on <inline-formula id="inf68">
<mml:math id="m77">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, as represented in <xref ref-type="disp-formula" rid="e10">Formula 10</xref>:<disp-formula id="e10">
<mml:math id="m78">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:msub>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:msub>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>
</p>
<p>In this case, if the gene probe corresponding to <inline-formula id="inf69">
<mml:math id="m79">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is not included in the set of <inline-formula id="inf70">
<mml:math id="m80">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, then <inline-formula id="inf71">
<mml:math id="m81">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. We use <xref ref-type="disp-formula" rid="e11">Formulas 11</xref>, <xref ref-type="disp-formula" rid="e12">12</xref> to cross-update <inline-formula id="inf72">
<mml:math id="m82">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>:<disp-formula id="e11">
<mml:math id="m83">
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>&#x2190;</mml:mo>
<mml:mfrac>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>
<disp-formula id="e12">
<mml:math id="m84">
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>&#x2190;</mml:mo>
<mml:mfrac>
<mml:mover accent="true">
<mml:mi>v</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mi>w</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mover accent="true">
<mml:mi>v</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mi>u</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(12)</label>
</disp-formula>
</p>
<p>
<statement content-type="algorithm" id="Algorithm_1">
<label>Algorithm 1</label>
<p>Semi-supervised weighted edge SPCA.<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf73">
<mml:math id="m85">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi>u</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>X</mml:mi>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf74">
<mml:math id="m86">
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>:</mml:mo>
<mml:mtext>for&#x2009;any&#x2009;weight&#x2009;of&#x2009;edge&#x2009;</mml:mtext>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;in&#x2009;</mml:mtext>
<mml:msub>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi mathvariant="normal">w</mml:mi>
</mml:msub>
<mml:mtext>&#x2009;do</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf75">
<mml:math id="m87">
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>:</mml:mo>
<mml:msubsup>
<mml:mi>w</mml:mi>
<mml:mi>h</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:msubsup>
<mml:mi>u</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mi>u</mml:mi>
<mml:mi>j</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:msqrt>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mi>G</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>a</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>d</mml:mi>
<mml:mi>y</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>w</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>k</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf76">
<mml:math id="m88">
<mml:mrow>
<mml:mn>4</mml:mn>
<mml:mo>:</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="script">G</mml:mi>
</mml:mrow>
<mml:mi mathvariant="normal">w</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mi>w</mml:mi>
<mml:mi>h</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf77">
<mml:math id="m89">
<mml:mrow>
<mml:mn>5</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi mathvariant="bold-italic">e</mml:mi>
<mml:mi mathvariant="bold-italic">n</mml:mi>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="bold-italic">f</mml:mi>
<mml:mi mathvariant="bold-italic">o</mml:mi>
<mml:mi mathvariant="bold-italic">r</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf78">
<mml:math id="m90">
<mml:mrow>
<mml:mn>6</mml:mn>
<mml:mo>:</mml:mo>
<mml:mtext>Let&#x2009;</mml:mtext>
<mml:msubsup>
<mml:mtext>norm</mml:mtext>
<mml:msubsup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>w</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x2225;</mml:mo>
<mml:msubsup>
<mml:mi>e</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mrow>
<mml:mo>&#x2225;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mo>&#x2225;</mml:mo>
</mml:mrow>
<mml:msubsup>
<mml:mi>e</mml:mi>
<mml:mi>l</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mo>&#x2225;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf79">
<mml:math id="m91">
<mml:mrow>
<mml:mn>7</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi>I</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>supp</mml:mtext>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mtext>norm</mml:mtext>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mtext>&#x2009;Extract&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mtext>&#x2009;edges</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf80">
<mml:math id="m92">
<mml:mrow>
<mml:mn>8</mml:mn>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>J</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>sample</mml:mtext>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf81">
<mml:math id="m93">
<mml:mrow>
<mml:mn>9</mml:mn>
<mml:mo>:</mml:mo>
<mml:mtext>if&#x2009;</mml:mtext>
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#x3e;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mtext>&#x2009;then&#x2009;</mml:mtext>
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3c1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf82">
<mml:math id="m94">
<mml:mrow>
<mml:mn>10</mml:mn>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:msubsup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>w</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>V</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>w</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf83">
<mml:math id="m95">
<mml:mrow>
<mml:mn>11</mml:mn>
<mml:mo>:</mml:mo>
<mml:mtext>for&#x2009;any&#x2009;gene&#x2009;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mtext>&#x2009;in&#x2009;</mml:mtext>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:msubsup>
<mml:mi mathvariant="script">G</mml:mi>
<mml:mi>w</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:msub>
<mml:mtext>&#x2009;do</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf84">
<mml:math id="m96">
<mml:mrow>
<mml:mn>12</mml:mn>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf85">
<mml:math id="m97">
<mml:mrow>
<mml:mn>13</mml:mn>
<mml:mo>:</mml:mo>
<mml:mtext>end&#x2009;for</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf86">
<mml:math id="m98">
<mml:mrow>
<mml:mn>14</mml:mn>
<mml:mo>:</mml:mo>
<mml:mover accent="true">
<mml:mi>X</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf87">
<mml:math id="m99">
<mml:mrow>
<mml:mn>15</mml:mn>
<mml:mo>:</mml:mo>
<mml:mtext>Class</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>RandomForestClassifier&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf88">
<mml:math id="m100">
<mml:mrow>
<mml:mn>16</mml:mn>
<mml:mo>:</mml:mo>
<mml:mtext>Class</mml:mtext>
<mml:mo>.</mml:mo>
<mml:mtext>&#x2009;fit&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mtext>&#x2009;Class</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mo>\</mml:mo>
<mml:mo>&#x23;</mml:mo>
<mml:mtext>Train&#x2009;and&#x2009;test&#x2009;the&#x2009;classifier</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf89">
<mml:math id="m101">
<mml:mrow>
<mml:mn>17</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>Class</mml:mtext>
<mml:mo>.</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>feature</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>importances</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf90">
<mml:math id="m102">
<mml:mrow>
<mml:mn>18</mml:mn>
<mml:mo>:</mml:mo>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2a;</mml:mo>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mo>&#x23;</mml:mo>
<mml:mtext>&#x2009;Weight&#x2009;</mml:mtext>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf91">
<mml:math id="m103">
<mml:mrow>
<mml:mn>19</mml:mn>
<mml:mo>:</mml:mo>
<mml:mtext>return&#x2009;</mml:mtext>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf92">
<mml:math id="m104">
<mml:mrow>
<mml:mn>20</mml:mn>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo>&#x2225;</mml:mo>
<mml:mover accent="true">
<mml:mi>u</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>&#x2225;</mml:mo>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf93">
<mml:math id="m105">
<mml:mrow>
<mml:mn>21</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi>v</mml:mi>
<mml:mo>&#x2190;</mml:mo>
<mml:mfrac>
<mml:mover accent="true">
<mml:mi>v</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mi>w</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mover accent="true">
<mml:mi>v</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf94">
<mml:math id="m106">
<mml:mrow>
<mml:mn>22</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="&#x7c;">
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mn>0.0001</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>n</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>p</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
</list>
</p>
</statement>
</p>
</sec>
</sec>
<sec id="s2-4-2">
<title>2.4.2 Similarity network building modules</title>
<p>For the same drug, we can get at least three different genomics data. The experiments in this paper mainly include gene expression data, copy volume data, and methylation data. Each omics performs sparse PCA operations independently. Finally, we can obtain three key feature matrices, namely, <inline-formula id="inf95">
<mml:math id="m107">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf96">
<mml:math id="m108">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf97">
<mml:math id="m109">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf98">
<mml:math id="m110">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the number of key gene probes retained by each of the three omics.</p>
<p>Because of the inconsistency of the data lengths for each omics, data cannot be aligned. Therefore, we calculate a sample similarity subnet for each omics based on the concept of the sample similarity network. In this paper, we use two similarity measurement methods. We use the Spearman correlation coefficient to calculate the sample similarity subnet of gene expression and methylation omics, as represented in <xref ref-type="disp-formula" rid="e13">Formula 13</xref>:<disp-formula id="e13">
<mml:math id="m111">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c1;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:mfrac>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(13)</label>
</disp-formula>
</p>
<p>Here, for the <inline-formula id="inf99">
<mml:math id="m112">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>th sample, <inline-formula id="inf100">
<mml:math id="m113">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represent the expression information on the <inline-formula id="inf101">
<mml:math id="m114">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>th gene expression in each sample, with <inline-formula id="inf102">
<mml:math id="m115">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> representing the total number of gene probes. The symbols <inline-formula id="inf103">
<mml:math id="m116">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo>,</mml:mo>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> indicate the average gene expression levels for each sample.</p>
<p>The Kendall correlation coefficient is used for the copy number dataset, as provided in <xref ref-type="disp-formula" rid="e14">Formula 14</xref>:<disp-formula id="e14">
<mml:math id="m117">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>u</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(14)</label>
</disp-formula>
</p>
<p>Here, the <inline-formula id="inf104">
<mml:math id="m118">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>th sample can be showed as a set of two elements containing p gene probe. <inline-formula id="inf105">
<mml:math id="m119">
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the number of consistent elements. <inline-formula id="inf106">
<mml:math id="m120">
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the number of inconsistent elements. <inline-formula id="inf107">
<mml:math id="m121">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the total number of gene probes in each sample.</p>
</sec>
<sec id="s2-4-3">
<title>2.4.3 Data fusion module</title>
<p>After completing the construction of the sample similarity subnet, we can obtain three feature matrices, namely, <inline-formula id="inf108">
<mml:math id="m122">
<mml:mrow>
<mml:msup>
<mml:mi>S</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:msup>
<mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf109">
<mml:math id="m123">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:msup>
<mml:mi>C</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf110">
<mml:math id="m124">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:msup>
<mml:mi>M</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. Then, we propose a subnet fusion algorithm based on the dip test and variance estimation using <xref ref-type="disp-formula" rid="e15">Formula 15</xref>:<disp-formula id="e15">
<mml:math id="m125">
<mml:mrow>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">&#x3b1;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mi>S</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">&#x3b1;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">&#x3b1;</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(15)</label>
</disp-formula>where <inline-formula id="inf111">
<mml:math id="m126">
<mml:mrow>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> represent the feature representation after fusion. <inline-formula id="inf112">
<mml:math id="m127">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf113">
<mml:math id="m128">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> are defined as the amount of statistical information corresponding to the genomics data matrix. Theoretically, our goal is to retain as much of the feature matrix as possible, prioritizing genomics with higher statistical significance for drug response prediction. To achieve this, we use two statistical methods to assess the amount of information in the data. The first method is the single peak test, which aims to retain similarity matrices that exhibit more typical bimodal distributions. A bimodal distribution is a statistical concept that represents a dataset in more than two regions. In gene expression analysis, if the data show a bimodal part, it indicates a significant statistical difference within the sample. In this paper, we assess the bimodal property of data using the dip test method, originally proposed by <xref ref-type="bibr" rid="B23">Hartigan et al. (1985)</xref>. We assume that <inline-formula id="inf114">
<mml:math id="m129">
<mml:mrow>
<mml:mi>&#x3c1;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> follows <xref ref-type="disp-formula" rid="e16">Formula 16</xref> for any bounded functions <inline-formula id="inf115">
<mml:math id="m130">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Let <inline-formula id="inf116">
<mml:math id="m131">
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> be the class of unimodal distribution functions.<disp-formula id="e16">
<mml:math id="m132">
<mml:mrow>
<mml:mi>&#x3c1;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2061;</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">sup</mml:mi>
<mml:mi>x</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="&#x7c;">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(16)</label>
</disp-formula>
</p>
<p>We define <inline-formula id="inf117">
<mml:math id="m133">
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> as a typical unimodal distribution function and <inline-formula id="inf118">
<mml:math id="m134">
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> as a dip distribution function. We can obtain <xref ref-type="disp-formula" rid="e17">Formulas 17</xref>, <xref ref-type="disp-formula" rid="e18">18</xref> as follows:<disp-formula id="e17">
<mml:math id="m135">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c1;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(17)</label>
</disp-formula>
<disp-formula id="e18">
<mml:math id="m136">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3c1;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(18)</label>
</disp-formula>
</p>
<p>It is important to note that <inline-formula id="inf119">
<mml:math id="m137">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> for <inline-formula id="inf120">
<mml:math id="m138">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, indicating that the dip quantifies deviation from unimodality. Assume that the result of the dip function is <inline-formula id="inf121">
<mml:math id="m139">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, as shown in <xref ref-type="disp-formula" rid="e19">Equation 19</xref>:<disp-formula id="e19">
<mml:math id="m140">
<mml:mrow>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mi>p</mml:mi>
<mml:mo>&#x3e;</mml:mo>
<mml:mn>0.95</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>u</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mn>0.05</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(19)</label>
</disp-formula>
</p>
<p>Another statistical method is the variance test. In addition to information about the probability distribution of the samples, our goal is to retain a matrix of sample similarity features that preserves as much discrete information as possible. The formula for the variance <inline-formula id="inf122">
<mml:math id="m141">
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> information is provided in <xref ref-type="disp-formula" rid="e20">Formula 20</xref>:<disp-formula id="e20">
<mml:math id="m142">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>X</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(20)</label>
</disp-formula>where <inline-formula id="inf123">
<mml:math id="m143">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the variable, <inline-formula id="inf124">
<mml:math id="m144">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>X</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> is the sample mean, and n denotes the sample size. Suppose that the result of the variance of the <inline-formula id="inf125">
<mml:math id="m145">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> feature of the <inline-formula id="inf126">
<mml:math id="m146">
<mml:mrow>
<mml:msub>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> histology is <inline-formula id="inf127">
<mml:math id="m147">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Then, <inline-formula id="inf128">
<mml:math id="m148">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> of <inline-formula id="inf129">
<mml:math id="m149">
<mml:mrow>
<mml:msub>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> histology can be expressed as <xref ref-type="disp-formula" rid="e21">Formula 21</xref>:<disp-formula id="e21">
<mml:math id="m150">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(21)</label>
</disp-formula>
</p>
<p>The computed <inline-formula id="inf130">
<mml:math id="m151">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> accounts for the possibility that <inline-formula id="inf131">
<mml:math id="m152">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> having a large parameter. Therefore, we normalize <inline-formula id="inf132">
<mml:math id="m153">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> using <xref ref-type="disp-formula" rid="e22">Formulas 22</xref>, <xref ref-type="disp-formula" rid="e23">23</xref> as follows:<disp-formula id="e22">
<mml:math id="m154">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>k</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="italic">Min</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(22)</label>
</disp-formula>
<disp-formula id="e23">
<mml:math id="m155">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi mathvariant="italic">Max</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="italic">Min</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(23)</label>
</disp-formula>
</p>
<p>Here, <inline-formula id="inf133">
<mml:math id="m156">
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf134">
<mml:math id="m157">
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> are user-defined parameters, representing the normalized range of data. <inline-formula id="inf135">
<mml:math id="m158">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents a scaling factor used to normalize the raw data <inline-formula id="inf136">
<mml:math id="m159">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to a user-defined range. <inline-formula id="inf137">
<mml:math id="m160">
<mml:mrow>
<mml:mi mathvariant="italic">Max</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf138">
<mml:math id="m161">
<mml:mrow>
<mml:mi mathvariant="italic">Min</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represent the maximum and minimum values of <inline-formula id="inf139">
<mml:math id="m162">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, respectively.</p>
</sec>
<sec id="s2-4-4">
<title>2.4.4 Drug response prediction module</title>
<p>Finally, we obtain the feature matrix <inline-formula id="inf140">
<mml:math id="m163">
<mml:mrow>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. Although the problem of the high dimensionality of data has been largely alleviated after genomic feature extraction and similarity network computation, researchers still need a powerful enough deep learning model to achieve high performance and avoid overfitting. However, considering the limitation of the number of samples, researchers still need a sufficiently powerful drug response prediction deep learning model to avoid model overfitting. In this study, we construct a deep learning model based on one-dimensional convolution and KANs to predict drug response (<xref ref-type="fig" rid="F2">Figure 2</xref>). One-dimensional convolution can further localize the features of the samples and remove potential noise. Experimental results indicate that one-dimensional convolution significantly enhances the model&#x2019;s prediction performance. KANs, proposed by <xref ref-type="bibr" rid="B26">Liu et al. (2024)</xref>, aim to replace the traditional fully connected neural network layer. The network is based on the Kolmogorov&#x2013;Arnold theorem, which states that any continuous function <inline-formula id="inf141">
<mml:math id="m164">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> in n-dimensional real space, where <inline-formula id="inf142">
<mml:math id="m165">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> can be represented as a combination of a single-variable continuous function <inline-formula id="inf143">
<mml:math id="m166">
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and a series of continuous bivariate functions <inline-formula id="inf144">
<mml:math id="m167">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf145">
<mml:math id="m168">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Specifically, the theorem is expressed in <xref ref-type="disp-formula" rid="e24">Formula 24</xref>:<disp-formula id="e24">
<mml:math id="m169">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(24)</label>
</disp-formula>
</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Structure of the deep learning model used in the NMDP model.</p>
</caption>
<graphic xlink:href="fgene-16-1532651-g002.tif"/>
</fig>
<p>The theorem shows that even a complex function in a high-dimensional space can be reconstructed using a series of lower-dimensional function operations. Specifically, a KAN layer with <inline-formula id="inf146">
<mml:math id="m170">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> dimensional inputs and <inline-formula id="inf147">
<mml:math id="m171">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> dimensional outputs can be defined as a matrix of one-dimensional functions, as represented in <xref ref-type="disp-formula" rid="e25">Formula 25</xref>:<disp-formula id="e25">
<mml:math id="m172">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3d5;</mml:mi>
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>q</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(25)</label>
</disp-formula>where the function <inline-formula id="inf148">
<mml:math id="m173">
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is defined as shown in <xref ref-type="disp-formula" rid="e26">Formula 26</xref> and consists of a B-spline curve and a residual activation function <inline-formula id="inf149">
<mml:math id="m174">
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, all multiplied by a learnable parameter <inline-formula id="inf150">
<mml:math id="m175">
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. The function <inline-formula id="inf151">
<mml:math id="m176">
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is defined as shown in <xref ref-type="disp-formula" rid="e26">Formula 26</xref>:<disp-formula id="e26">
<mml:math id="m177">
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(26)</label>
</disp-formula>
</p>
<p>The main advantage of KANs is that they can achieve results beyond fully connected neural networks while using fewer parameters. This is especially important for the drug response prediction problem. Due to the limitation of the sample size, it is unlikely that we can construct a deep learning model that contains a huge neural network. To summarize, the module can be expressed using <xref ref-type="disp-formula" rid="e27">Formulas 27</xref>, <xref ref-type="disp-formula" rid="e28">28</xref> as follows:<disp-formula id="e27">
<mml:math id="m178">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>X</mml:mi>
<mml:mo>&#xb4;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>O</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(27)</label>
</disp-formula>
<disp-formula id="e28">
<mml:math id="m179">
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>K</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>X</mml:mi>
<mml:mo>&#xb4;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(28)</label>
</disp-formula>
</p>
</sec>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<p>The procedure in this article consists of six distinct steps. Initially, experiments were performed using 14 FDA-approved targeted therapy drugs already authorized for clinical use. In the second step, we broadened the model evaluation by testing it with 49 targeted therapy drugs not approved by the FDA. In the third step, we conducted experiments on five chemotherapeutic agents (non-targeted therapeutics) in order to verify that the NMDP model has good scalability. We used seven state-of-the-art AI models for comparison, namely, TSGCNN, MOICVAE, MOLI, netDx, netDx&#x2013;elastic network, deep autoencoder, and netDx&#x2013;SVR. Five evaluation indicators were used, namely, sensitivity, specificity, precision, accuracy, and F1 score. The details of comparison models are provided in <xref ref-type="sec" rid="s11">Supplementary Materials</xref>.</p>
<p>In the fourth step, we selected the GDSC1 dataset for training and testing and the GDSC2 dataset for validation. We selected 14 FDA-approved drugs to perform and calculate the mean value. The consistency and reliability of the results were ensured by calculating the mean value.</p>
<p>The fifth step included conducting ablation experiments to determine the importance of each sub-module of the NMDP model. We randomly selected 10 drugs for analysis and averaged the results. All experiments were performed using the GDSC2 dataset (<xref ref-type="sec" rid="s11">Supplementary Table S7</xref>). We conducted four independent experiments: the first experiment was conducted to remove the multi-omics weighting module; the second experiment, to remove the convolution module; the third experiment, to remove the sample similarity network; and the last experiment, to replace KANs with MLPs.</p>
<p>Ultimately, we used the Metascape platform to examine the biological pathways associated with the gene probes chosen by the NMDP model (<xref ref-type="bibr" rid="B52">Zhou et al., 2019</xref>). The details of the indicators are provided in <xref ref-type="sec" rid="s11">Supplementary Materials</xref>.</p>
<sec id="s3-1">
<title>3.1 FDA-approved targeted therapy drugs</title>
<p>Based on the results presented in <xref ref-type="fig" rid="F3">Figure 3</xref> and <xref ref-type="sec" rid="s11">Supplementary Table S4</xref>, experiments show that the NMDP model is much better than the advanced deep learning model. Notably, the NMDP model achieves an average sensitivity of 0.92 and a specificity of 0.93. Among the models for comparison, the MOICVAE model ranks highest, with a sensitivity of 0.77 and a specificity of 0.91. The deep autoencoder model, however, performs the lowest, with sensitivity and specificity values of 0.53 and 0.44, respectively.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Results of 14 FDA-approved drugs for each model. <bold>(A)</bold> Sensitivity and specificity of the NMDP model; <bold>(B)</bold> sensitivity and specificity of the MOLI model; <bold>(C)</bold> sensitivity and specificity of the netDx model; <bold>(D)</bold> sensitivity and specificity of the TSGCNN model; <bold>(E)</bold> sensitivity and specificity of the MOICVAE model; and <bold>(F)</bold> accuracy of each model.</p>
</caption>
<graphic xlink:href="fgene-16-1532651-g003.tif"/>
</fig>
<p>Based on <xref ref-type="fig" rid="F3">Figure 3C</xref>, it is evident that the deep autoencoder model exhibits overfitting across multiple drugs. Moreover, the NMDP model demonstrates minimal fluctuation across 14 drugs, indicating its superior stability (<xref ref-type="fig" rid="F3">Figure 3F</xref>). Compared to other models, all except the MOLI model show relatively high levels of fluctuation, suggesting weaker stability in those models. The NMDP model achieves values of 0.93, 0.92, 0.92, and 0.92 for average accuracy and F1 score, outperforming the MOICVAE model by 10% in each metric (<xref ref-type="table" rid="T1">Table 1</xref>). When compared to the deep autoencoder model, it shows improvements of 31%, 48%, 52%, and 50%, respectively.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Results of each model of the 14 drugs approved by the FDA.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="left"/>
<th align="center">NMDP</th>
<th align="center">MOLI</th>
<th align="center">Deep autoencoder</th>
<th align="center">netDx&#x2013;KNN</th>
<th align="center">netDx&#x2013;ElasticNet</th>
<th align="center">netDx&#x2013;SVR</th>
<th align="center">TSGCNN</th>
<th align="center">MOICVAE</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Accuracy</td>
<td align="left">All</td>
<td align="center">0.93</td>
<td align="center">0.84</td>
<td align="center">0.64</td>
<td align="center">0.70</td>
<td align="center">0.71</td>
<td align="center">0.74</td>
<td align="center">0.78</td>
<td align="center">0.86</td>
</tr>
<tr>
<td align="left">F1 score</td>
<td align="left">Responsive</td>
<td align="center">0.92</td>
<td align="center">0.83</td>
<td align="center">0.48</td>
<td align="center">0.72</td>
<td align="center">0.67</td>
<td align="center">0.71</td>
<td align="center">0.74</td>
<td align="center">0.74</td>
</tr>
<tr>
<td align="left"/>
<td align="left">Non-responsive</td>
<td align="center">0.92</td>
<td align="center">0.83</td>
<td align="center">0.44</td>
<td align="center">0.56</td>
<td align="center">0.58</td>
<td align="center">0.66</td>
<td align="center">0.78</td>
<td align="center">0.86</td>
</tr>
<tr>
<td align="left"/>
<td align="left">All</td>
<td align="center">0.92</td>
<td align="center">0.83</td>
<td align="center">0.46</td>
<td align="center">0.64</td>
<td align="center">0.63</td>
<td align="center">0.68</td>
<td align="center">0.77</td>
<td align="center">0.80</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>According to <xref ref-type="fig" rid="F4">Figure 4</xref>, the NMDP model demonstrated excellent performance in predictive accuracy. It is worth mentioning that out of these 14 drugs, the prediction accuracy for 13 of them exceeded 90%.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>NMDP model precision results of 14 FDA-approved drugs.</p>
</caption>
<graphic xlink:href="fgene-16-1532651-g004.tif"/>
</fig>
</sec>
<sec id="s3-2">
<title>3.2 FDA non-approved targeted therapy drugs</title>
<p>Across the 49 drugs not approved by the FDA, we observe similar outcomes. Experimental findings indicate that the NMDP model achieves average sensitivity and specificity values of 0.92 and 0.93, respectively, outperforming the comparison models by 11%&#x2013;57% (<xref ref-type="sec" rid="s11">Supplementary Figure S1</xref>; <xref ref-type="sec" rid="s11">Supplementary Table S5</xref>). <xref ref-type="sec" rid="s11">Supplementary Figure S1F</xref> illustrates the NMDP model&#x2019;s high stability, with only 5 out of the 49 drugs showing precision below 0.9 (see <xref ref-type="fig" rid="F5">Figure 5</xref>). In addition, the average precision of the NMDP method can reach 0.95. F1 score can reach 0.92, surpassing the MOLI model by 14%, 10%, and 13% (<xref ref-type="sec" rid="s11">Supplementary Table S1</xref>). Among the seven models compared, MOLI achieves the highest performance. Nevertheless, its sensitivity, specificity, precision, and accuracy for response, non-response, and all samples are only 0.80, 0.88, and 0.86, respectively, with F1 scores of 0.83, 0.83, 0.83, 0.82, and 0.84.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>NMDP model precision results of five non-specific therapeutic drugs.</p>
</caption>
<graphic xlink:href="fgene-16-1532651-g005.tif"/>
</fig>
</sec>
<sec id="s3-3">
<title>3.3 Non-specific therapeutic drugs</title>
<p>In the study of five non-specific therapeutic drugs, we achieve optimal outcomes in three experiments. Results indicate that the NMDP model achieves 0.93 for average sensitivity, surpassing the comparison models by 19%&#x2013;42% (<xref ref-type="sec" rid="s11">Supplementary Figure S2</xref>; <xref ref-type="sec" rid="s11">Supplementary Table S6</xref>). Additionally, the NMDP model demonstrates a precision close to 0.95 across the five drugs (<xref ref-type="fig" rid="F6">Figure 6</xref>). The model also shows outstanding performance in F1 score and accuracy, reaching 0.93 (<xref ref-type="sec" rid="s11">Supplementary Table S2</xref>). Compared to the other models, the MOLI model achieves the best results, but its average F1 score and accuracy reach only 0.78. The experimental results show that the NMDP model has good expansibility.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>NMDP model precision results of five non-specific therapeutic drugs.</p>
</caption>
<graphic xlink:href="fgene-16-1532651-g006.tif"/>
</fig>
</sec>
<sec id="s3-4">
<title>3.4 External independent validation results</title>
<p>To evaluate the model&#x2019;s performance and test its generalization ability, we design this external independent validation experiment. The experimental outcomes demonstrate that the NMDP model exhibits superior generalization capabilities. Specifically, the NMDP model achieves an overall prediction accuracy of 0.77 and precision, recall, F1 score, and accuracy of 0.73, 0.77, 0.74, and 0.77, respectively (<xref ref-type="table" rid="T2">Table 2</xref>). It is worth noting that a slight decrease in accuracy is observed on the validation set compared to the results on the test set. This may be due to the noise difference between the datasets. Overall, the NMDP model exhibits robustness and reliability, with the capability for widespread application.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>External independent validation results.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="left">Precision</th>
<th align="left">Recall</th>
<th align="left">F1-score</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Responsive</td>
<td align="left">0.72</td>
<td align="left">0.74</td>
<td align="left">0.73</td>
</tr>
<tr>
<td align="left">Non-responsive</td>
<td align="left">0.83</td>
<td align="left">0.81</td>
<td align="left">0.82</td>
</tr>
<tr>
<td align="left">Accuracy</td>
<td align="left"/>
<td align="left"/>
<td align="left">0.77</td>
</tr>
<tr>
<td align="left">Macro average</td>
<td align="left">0.73</td>
<td align="left">0.77</td>
<td align="left">0.74</td>
</tr>
<tr>
<td align="left">Weighted average</td>
<td align="left">0.8</td>
<td align="left">0.78</td>
<td align="left">0.79</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-5">
<title>3.5 Ablation experiment</title>
<p>The experimental results show that the model feature extraction effect is weakened by removing the multi-omics weighting module and the convolution module, but the convolution module has a greater impact on the model. The sample similarity network module has the greatest impact, further verifying the importance of similarity across samples. When KANs are replaced with MLPs, the performance of the model improves but still does not surpass that of the original model. This indicates that KANs have a unique advantage in capturing complex relationships, especially when dealing with multi-omics data. Taken together, the results of the ablation experiments fully indicate that the sample similarity network and convolution module are the key factors in improving the model performance. Among them, the sample similarity network module has the greatest impact, and we believe that the main reason is that, even after the feature filtering of the sparse PCA model, the three modules are still able to save more than 9,000 gene probes collectively, and the excessively high data dimensions make it easy for the model to fall into an overfitting state.</p>
</sec>
<sec id="s3-6">
<title>3.6 Enrichment analysis</title>
<p>To validate the biological interpretability of the NMDP model, we conduct bio-enrichment analysis using gene selection results for erlotinib across different omics types obtained from the first principal component (PC) in the NMDP model. Erlotinib is an FDA-approved non-small cell lung cancer drug, with EGFR as its primary target (<xref ref-type="bibr" rid="B44">Tsao et al., 2005</xref>; <xref ref-type="bibr" rid="B51">Zhou et al., 2021</xref>). The analysis yields promising results as the NMDP model successfully identifies lung cancer-related pathways and gene probes. For instance, in copy number omics, we discover pathways corresponding to genes like <italic>EGFR</italic>, <italic>CDK6</italic>, <italic>RASSF5</italic>, <italic>BRAF</italic>, and <italic>CCND1</italic>, which are associated with non-small cell lung cancer (<xref ref-type="bibr" rid="B10">Chen et al., 2018</xref>; <xref ref-type="bibr" rid="B47">Xue et al., 2019</xref>; <xref ref-type="bibr" rid="B48">Zhao et al., 2018</xref>) (<xref ref-type="fig" rid="F7">Figure 7A</xref>). In methylation omics, we observe the developmental process pathway GO:0032502, involving genes such as <italic>EGFR</italic>, <italic>FGFR2</italic>, <italic>GATA6</italic>, <italic>ASCL1</italic>, <italic>BMP4</italic>, and <italic>FOXA1</italic>, indicating relevance to lung development (<xref ref-type="bibr" rid="B2">Bach et al., 2018</xref>; <xref ref-type="bibr" rid="B25">Ju et al., 2019</xref>; <xref ref-type="bibr" rid="B31">Murai et al., 2015</xref>) (<xref ref-type="fig" rid="F7">Figure 7B</xref>). In Seq omics, we identify pathways related to the KEGG pathway, which include genes like <italic>EGFR</italic>, <italic>MAPK1</italic>, <italic>KRAS</italic>, <italic>CCND1</italic>, <italic>HRAS</italic>, <italic>NRAS</italic>, <italic>PLCG1</italic>, and <italic>GRB2</italic>, which show strong connections to non-small cell lung cancer (<xref ref-type="bibr" rid="B9">Betticher et al., 1996</xref>; <xref ref-type="bibr" rid="B32">Park et al., 2020</xref>; <xref ref-type="bibr" rid="B33">P&#x105;zik et al., 2021</xref>) (<xref ref-type="fig" rid="F7">Figure 7C</xref>). Remarkably, EGFR, the target gene of erlotinib, is consistently identified across these three omics types.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>
<bold>(A)</bold> Pathway results from the first PC of copy number; <bold>(B)</bold> pathway results from the first PC of methylation data; and <bold>(C)</bold> pathway results from the first PC of sequencing data.</p>
</caption>
<graphic xlink:href="fgene-16-1532651-g007.tif"/>
</fig>
<p>Overall, the NMDP model demonstrates superior performance compared to other models across all metrics.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<p>With advancements in bioassay technology, a growing number of large-scale drug response datasets are being released, creating new possibilities for building drug response prediction models. In recent years, researchers have proposed a number of AI-based drug response prediction models (<xref ref-type="bibr" rid="B12">Chiu et al., 2020</xref>). However, drug response prediction data are mostly characterized by the typical features of multi-omics, small samples, high dimensionality, and high noise. Designing feature selection and multi-omics fusion methods based on regularization ideas becomes very important. In addition, considering the potential overfitting problem, it is difficult for researchers to build AI-based drug response prediction models with many parameters.</p>
<p>In light of the aforementioned issues, we propose the NMDP model, which integrates semi-supervised weighted SPCA, similarity networks, dip tests, and KANs. Unlike the traditional unsupervised sparse PCA model, the NMDP model proposes an independent evaluator that converts the sparse PCA model from a traditional unsupervised to a semi-supervised model. This improvement allows the NMDP model to use known dataset grouping information, ultimately allowing the model to stably select different potential target genes for different target drugs. The experimental results show that the NMDP model inherits the advantages of the sparse PCA model, such as good biological interpretability and strong denoising ability, further enhances the feature selection ability of the model in multi-omics gene data, and greatly strengthens the stability of the model in high-dimensional small-sample cases. Sample similarity networks further address the dimensionality challenge of the samples while helping the model perform multi-omics data alignment. We introduce a fusion algorithm that utilizes both dip test and variance data with weighted integration, which allows the model to focus on important histological information, thus improving prediction accuracy. Finally, we propose a one-dimensional convolution combined with KANs for drug response prediction modeling. The model achieves efficient prediction with a small number of parameters, thereby effectively avoiding the overfitting problem.</p>
<p>To enhance the validation of the model, we also conduct external validation experiments to assess the generalization capability of the model using independent datasets. The experimental findings indicate that the NMDP model performs consistently on different datasets, validating its robustness and reliability. In addition, we conduct ablation experiments to evaluate the contribution of each component to the model performance. The results of the ablation experiments show that removing the multi-omics weighting module and the convolution module significantly degrades the model&#x2019;s performance, and in particular, the sample similarity network module plays the most crucial role in influencing the model&#x2019;s effectiveness. This further emphasizes the importance of inter-sample similarity and the unique advantage of KANs in capturing complex relationships. Bioenrichment experiments fully validate the biointerpretability of the model, suggesting that the NMDP model could help researchers in drug development.</p>
<p>We also acknowledge some limitations to this study: our research is confined to predicting the response to a single drug, without considering the effects of combination drug therapies. Moreover, the weighted edge sparse PCA method has high time complexity, which leads to slower model computations. In future work, we plan to improve the model&#x2019;s ability to predict responses to drug combinations and optimize its computational efficiency.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The data presented in the study are deposited in the European Genome-phenome Archive (EGA) repository, accession numbers EGAD00001001039, EGAD00001004201, EGAD00010000644; and in the Gene Expression Omnibus (GEO) repository, accession number GSE68379.</p>
</sec>
<sec sec-type="author-contributions" id="s6">
<title>Author contributions</title>
<p>RM: data curation, methodology, software, writing&#x2013;original draft, and writing&#x2013;review and editing. B-JZ: writing&#x2013;original draft and writing&#x2013;review and editing. X-YM: methodology, software, and writing&#x2013;original draft. XD: software and writing&#x2013;original draft. Y-DO: software and writing&#x2013;original draft. YL: writing&#x2013;review and editing. H-YY: writing&#x2013;review and editing. YW: writing&#x2013;review and editing. Z-HD: writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s7">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. The study is supported by the Guangdong Provincial Department of Education youth innovative talent project (no. 2023KQNCX155), Post-doctoral training project of Zunyi Medical University (no. 2023F-ZH- 019), the Key Construction Discipline of Immunology and Pathogen Biology Fund, Zunyi Medical University Zhuhai Campus, China (no. ZHGF 2024-1), and the Science and Technology Project of Guizhou Province (Project Number: Guizhou Science and Technology Support [2021] General 446).</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s11">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2025.1532651/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2025.1532651/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table2.XLSX" id="SM1" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table3.XLSX" id="SM2" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table7.DOCX" id="SM3" mimetype="application/DOCX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table6.XLSX" id="SM4" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table4.XLSX" id="SM5" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table1.XLSX" id="SM6" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table5.XLSX" id="SM7" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Adam</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Ramp&#xe1;&#x161;ek</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Safikhani</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Smirnov</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Haibe-Kains</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Goldenberg</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Machine learning approaches to drug response prediction: challenges and recent progress</article-title>. <source>NPJ Precis. Oncol.</source> <volume>4</volume> (<issue>1</issue>), <fpage>19</fpage>. <pub-id pub-id-type="doi">10.1038/s41698-020-0122-1</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bach</surname>
<given-names>D.-H.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>An</surname>
<given-names>Y. J.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>H. J.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>S. K.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Bmp4 upregulation is associated with acquired drug resistance and fatty acid metabolism in egfr-mutant non-small-cell lung cancer cells</article-title>. <source>Mol. Therapy-Nucleic Acids</source> <volume>12</volume>, <fpage>817</fpage>&#x2013;<lpage>828</lpage>. <pub-id pub-id-type="doi">10.1016/j.omtn.2018.07.016</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ballester</surname>
<given-names>P. J.</given-names>
</name>
<name>
<surname>Carmona</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Artificial intelligence for the next generation of precision oncology</article-title>. <source>NPJ Precis. Oncol.</source> <volume>5</volume> (<issue>1</issue>), <fpage>79</fpage>. <pub-id pub-id-type="doi">10.1038/s41698-021-00216-w</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ballester</surname>
<given-names>P. J.</given-names>
</name>
<name>
<surname>Stevens</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Haibe-Kains</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>R. S.</given-names>
</name>
<name>
<surname>Aittokallio</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2022</year>). <source>Artificial intelligence for drug response prediction in disease models, bbab450</source>. <publisher-name>Oxford University Press</publisher-name>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Baptista</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ferreira</surname>
<given-names>P. G.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A systematic evaluation of deep learning methods for the prediction of drug synergy in</article-title>. <source>Cancer</source> <volume>19</volume> (<issue>3</issue>), <fpage>e1010200</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1010200</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Baptista</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ferreira</surname>
<given-names>P. G.</given-names>
</name>
<name>
<surname>Rocha</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Deep learning for drug response prediction in cancer</article-title>. <source>Briefings Bioinforma.</source> <volume>22</volume> (<issue>1</issue>), <fpage>360</fpage>&#x2013;<lpage>379</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbz171</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barretina</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Caponigro</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Stransky</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Venkatesan</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Margolin</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2012a</year>). <article-title>The Cancer Cell Line Encyclopedia enables predictive modelling of anticancer drug sensitivity</article-title>. <source>Cancer Cell Line Encycl. enables Predict. Model. anticancer drug Sensit. Nat.</source> <volume>483</volume>, <fpage>603</fpage>&#x2013;<lpage>607</lpage>. <pub-id pub-id-type="doi">10.1038/nature11003</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barretina</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Caponigro</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Stransky</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Venkatesan</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Margolin</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2012b</year>). <article-title>The cancer cell line encyclopedia enables predictive modelling of anticancer drug sensitivity</article-title>. <source>Nature</source> <volume>483</volume> (<issue>7391</issue>), <fpage>603</fpage>&#x2013;<lpage>607</lpage>. <pub-id pub-id-type="doi">10.1038/nature11003</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Betticher</surname>
<given-names>D. C.</given-names>
</name>
<name>
<surname>Heighway</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hasleton</surname>
<given-names>P. S.</given-names>
</name>
<name>
<surname>Altermatt</surname>
<given-names>H. J.</given-names>
</name>
<name>
<surname>Ryder</surname>
<given-names>W. D. J.</given-names>
</name>
<name>
<surname>Cerny</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>1996</year>). <article-title>Prognostic significance of Ccnd1 (cyclin D1) overexpression in primary resected non-small-cell lung cancer</article-title>. <source>Br. J. cancer</source> <volume>73</volume> (<issue>3</issue>), <fpage>294</fpage>&#x2013;<lpage>300</lpage>. <pub-id pub-id-type="doi">10.1038/bjc.1996.52</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Snhg8 is identified as a key regulator in non-small-cell lung cancer progression sponging to mir-542-3p by targeting ccnd1/cdk6</article-title>. <source>OncoTargets Ther.</source> <volume>11</volume>, <fpage>6081</fpage>&#x2013;<lpage>6090</lpage>. <pub-id pub-id-type="doi">10.2147/OTT.S170482</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A survey and systematic assessment of computational methods for drug response prediction</article-title>. <source>Briefings Bioinforma.</source> <volume>22</volume> (<issue>1</issue>), <fpage>232</fpage>&#x2013;<lpage>246</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbz164</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chiu</surname>
<given-names>Y.-C.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.-I. H.</given-names>
</name>
<name>
<surname>Gorthi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mostavi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Deep learning of pharmacogenomics resources: moving towards precision oncology</article-title>. <source>Briefings Bioinforma.</source> <volume>21</volume> (<issue>6</issue>), <fpage>2066</fpage>&#x2013;<lpage>2083</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbz144</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chiu</surname>
<given-names>Y.-C.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.-I. H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gorthi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.-Ju</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Predicting drug response of tumors from integrated genomic profiles by deep neural networks</article-title>. <source>BMC Med. genomics</source> <volume>12</volume> (<issue>1</issue>), <fpage>18</fpage>&#x2013;<lpage>55</lpage>. <pub-id pub-id-type="doi">10.1186/s12920-018-0460-9</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Deng</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>Da</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A cofunctional grouping-based approach for non-redundant feature gene selection in unannotated single-cell rna-seq analysis</article-title>. <source>Brief. Bioinform.</source> <volume>24</volume>, <fpage>bbad042</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbad042</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ding</surname>
<given-names>M. Q.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Cooper</surname>
<given-names>G. F.</given-names>
</name>
<name>
<surname>Young</surname>
<given-names>J. D.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Precision oncology beyond targeted therapy: combining omics data with machine learning matches the majority of cancer cells to effective therapeutics</article-title>. <source>Mol. cancer Res.</source> <volume>16</volume> (<issue>2</issue>), <fpage>269</fpage>&#x2013;<lpage>278</lpage>. <pub-id pub-id-type="doi">10.1158/1541-7786.MCR-17-0378</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dlamini</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Francies</surname>
<given-names>F. Z.</given-names>
</name>
<name>
<surname>Hull</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Marima</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Artificial intelligence (ai) and big data in cancer and precision oncology</article-title>. <source>Comput. Struct. Biotechnol. J.</source> <volume>18</volume>, <fpage>2300</fpage>&#x2013;<lpage>2311</lpage>. <pub-id pub-id-type="doi">10.1016/j.csbj.2020.08.019</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dong</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Anticancer drug sensitivity prediction in cell lines from baseline gene expression through recursive feature selection</article-title>. <source>BMC cancer</source> <volume>15</volume>, <fpage>489</fpage>&#x2013;<lpage>512</lpage>. <pub-id pub-id-type="doi">10.1186/s12885-015-1492-6</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Firoozbakht</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Yousefi</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Schwikowski</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>An overview of machine learning methods for monotherapy drug response prediction</article-title>. <source>Briefings Bioinforma.</source> <volume>23</volume> (<issue>1</issue>), <fpage>bbab408</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbab408</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Korn</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Ferretti</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Monahan</surname>
<given-names>J. E.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>High-throughput screening using patient-derived tumor xenografts to predict clinical trial drug response</article-title>. <source>Nat. Med.</source> <volume>21</volume> (<issue>11</issue>), <fpage>1318</fpage>&#x2013;<lpage>1325</lpage>. <pub-id pub-id-type="doi">10.1038/nm.3954</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Garnett</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Edelman</surname>
<given-names>E. J.</given-names>
</name>
<name>
<surname>Heidorn</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Greenman</surname>
<given-names>C. D.</given-names>
</name>
<name>
<surname>Dastur</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lau</surname>
<given-names>K. W.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Systematic identification of genomic markers of drug sensitivity in cancer cells</article-title>. <source>Nature</source> <volume>483</volume> (<issue>7391</issue>), <fpage>570</fpage>&#x2013;<lpage>575</lpage>. <pub-id pub-id-type="doi">10.1038/nature11005</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Garraway</surname>
<given-names>L. A.</given-names>
</name>
<name>
<surname>Verweij</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ballman</surname>
<given-names>K. V.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Precision oncology: an overview</article-title>. <source>J. Clin. Oncol.</source> <volume>31</volume> (<issue>15</issue>), <fpage>1803</fpage>&#x2013;<lpage>1805</lpage>. <pub-id pub-id-type="doi">10.1200/JCO.2013.49.4799</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Geeleher</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Cox</surname>
<given-names>N. J.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>R. S.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Clinical drug response can Be predicted using baseline gene expression levels and <italic>in vitro</italic> drug sensitivity in cell lines</article-title>. <source>Genome Biol.</source> <volume>15</volume>, <fpage>R47</fpage>&#x2013;<lpage>R12</lpage>. <pub-id pub-id-type="doi">10.1186/gb-2014-15-3-r47</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hartigan</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Pamela</surname>
<given-names>M. J.</given-names>
</name>
</person-group> (<year>1985</year>). &#x201c;<article-title>The annals of statistics hartigan</article-title>,&#x201d; in <source>The dip test of unimodality</source>, <fpage>70</fpage>&#x2013;<lpage>84</lpage>.</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hodson</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Precision oncology</article-title>. <source>Nature</source> <volume>585</volume> (<issue>7826</issue>), <fpage>S1</fpage>. <pub-id pub-id-type="doi">10.1038/d41586-020-02673-y</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ju</surname>
<given-names>F.-J.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>F.-Q.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>H.-L.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Association between Bmp4 expression and pathology, ct characteristics and prognosis of non-small cell lung cancer</article-title>. <source>Eur. Rev. Med. and Pharmacol. Sci.</source> <volume>23</volume>, <fpage>5787</fpage>&#x2013;<lpage>5794</lpage>. <pub-id pub-id-type="doi">10.26355/eurrev_201907_18317</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Vaidya</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ruehle</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Halverson</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Solja&#x10d;i&#x107;</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Kan: Kolmogorov-arnold networks</article-title>. <comment>arXiv preprint arXiv:2404.19756</comment>.</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Miao</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.-H.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>L.-Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Z.-Yi</given-names>
</name>
<name>
<surname>He</surname>
<given-names>M.-F.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Beyond the limitation of targeted therapy: improve the application of targeted drugs combining genomic data with machine learning</article-title>. <source>Pharmacol. Res.</source> <volume>159</volume>, <fpage>104932</fpage>. <pub-id pub-id-type="doi">10.1016/j.phrs.2020.104932</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Miao</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.-Y.</given-names>
</name>
<name>
<surname>Lo</surname>
<given-names>S.-L.</given-names>
</name>
<name>
<surname>Xin-Yue</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Dynamic meta-data network sparse pca for cancer subtype biomarker screening</article-title>. <source>Front. Genet.</source> <volume>13</volume>, <fpage>869906</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2022.869906</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Min</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Edge-group sparse pca for network-guided high dimensional data analysis</article-title>. <source>Bioinformatics</source> <volume>34</volume> (<issue>20</issue>), <fpage>3479</fpage>&#x2013;<lpage>3487</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty362</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Munquad</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bikas</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Das</surname>
<given-names>J. H.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Uncovering the subtype-specific disease module and the development of drug response prediction models for glioma</article-title>. <source>Heliyon</source> <volume>10</volume> (<issue>5</issue>), <fpage>e27190</fpage>. <pub-id pub-id-type="doi">10.1016/j.heliyon.2024.e27190</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Murai</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Koinuma</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Shinozaki-Ushiku</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Fukayama</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Miyaozono</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Ehata</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Ezh2 promotes progression of small cell lung cancer by suppressing the tgf-&#x392;-smad-ascl1 pathway</article-title>. <source>Cell Discov.</source> <volume>1</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.1038/celldisc.2015.26</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Park</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Cho</surname>
<given-names>S.-Y.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Oh</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Combined blockade of polo-like kinase and pan-raf is effective against nras-mutant non-small cell lung cancer cells</article-title>. <source>Cancer Lett.</source> <volume>495</volume>, <fpage>135</fpage>&#x2013;<lpage>144</lpage>. <pub-id pub-id-type="doi">10.1016/j.canlet.2020.09.018</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>P&#x105;zik</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Michalska</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>&#x17b;ebrowska-Nawrocka</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zawadzka</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>&#x141;ochowski</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Balcerczak</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Clinical significance of hras and kras genes expression in patients with non&#x2013;small-cell lung cancer-preliminary findings</article-title>. <source>BMC cancer</source> <volume>21</volume>, <fpage>130</fpage>&#x2013;<lpage>213</lpage>. <pub-id pub-id-type="doi">10.1186/s12885-021-07858-w</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peng</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Lan</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Improving drug response prediction based on two-space graph convolution</article-title>. <source>Comput. Biol. Med.</source> <volume>158</volume>, <fpage>106859</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2023.106859</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Prasad</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Perspective: the precision-oncology illusion</article-title>. <source>Nature</source> <volume>537</volume> (<issue>7619</issue>), <fpage>S63</fpage>. <pub-id pub-id-type="doi">10.1038/537S63a</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Prasad</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Fojo</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Brada</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Precision oncology: origins, optimism, and potential</article-title>. <source>Lancet Oncol.</source> <volume>17</volume> (<issue>2</issue>), <fpage>e81</fpage>&#x2013;<lpage>e86</lpage>. <pub-id pub-id-type="doi">10.1016/S1470-2045(15)00620-8</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rashid</surname>
<given-names>Md M.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Advancing drug-response prediction using multi-modal and-omics machine learning integration (momlin): a case study on breast cancer clinical data</article-title>. <source>Brief. Bioinform.</source> <volume>25</volume> (<issue>4</issue>), <fpage>bbae300</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbae300</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rubio-Perez</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Tamborero</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Schroeder</surname>
<given-names>M. P.</given-names>
</name>
<name>
<surname>Antol&#xed;n</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Deu-Pons</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Perez-Llamas</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>
<italic>In silico</italic> prescription of anticancer drugs to cohorts of 28 tumor types reveals targeting opportunities</article-title>. <source>Cancer Cell</source> <volume>27</volume> (<issue>3</issue>), <fpage>382</fpage>&#x2013;<lpage>396</lpage>. <pub-id pub-id-type="doi">10.1016/j.ccell.2015.02.007</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sharifi-Noghabi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zolotareva</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Collins</surname>
<given-names>C. C.</given-names>
</name>
<name>
<surname>Ester</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Moli: multi-omics late integration with deep neural networks for drug response prediction</article-title>. <source>Bioinformatics</source> <volume>35</volume> (<issue>14</issue>), <fpage>i501</fpage>&#x2013;<lpage>i509</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz318</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sharma</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lysenko</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Boroevich</surname>
<given-names>K. A.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>DeepInsight-3D architecture for anti-cancer drug response prediction with deep-learning on multi-omics</article-title>. <source>Deepinsight-3d Archit. Anti-Cancer Drug Response Predict. Deep-Learning Multi-Omics</source> <volume>13</volume> (<issue>1</issue>), <fpage>2483</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-023-29644-3</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sharma</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lysenko</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Boroevich</surname>
<given-names>K. A.</given-names>
</name>
</person-group> (<year>2024</year>). <source>Advances in ai and machine learning for predictive medicine</source>, <fpage>1</fpage>&#x2013;<lpage>11</lpage>.</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sheng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>S. T. C.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Optimal drug prediction from personal genomics profiles</article-title>. <source>IEEE J. Biomed. Health Inf.</source> <volume>19</volume> (<issue>4</issue>), <fpage>1264</fpage>&#x2013;<lpage>1270</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2015.2412522</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Trac</surname>
<given-names>Q. T.</given-names>
</name>
<name>
<surname>Pawitan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Mou</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Erkers</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>&#xd6;stling</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Bohlin</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Prediction model for drug response of acute myeloid leukemia patients</article-title>. <source>NPJ Precis. Oncol.</source> <volume>7</volume> (<issue>1</issue>), <fpage>32</fpage>. <pub-id pub-id-type="doi">10.1038/s41698-023-00374-z</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tsao</surname>
<given-names>M.-S.</given-names>
</name>
<name>
<surname>Sakurada</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cutz</surname>
<given-names>J.-C.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>C.-Qi</given-names>
</name>
<name>
<surname>Kamel-Reid</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Squire</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2005</year>). <article-title>Erlotinib in lung cancer&#x2014;molecular and clinical predictors of outcome</article-title>. <source>N. Engl. J. Med.</source> <volume>353</volume> (<issue>2</issue>), <fpage>133</fpage>&#x2013;<lpage>144</lpage>. <pub-id pub-id-type="doi">10.1056/NEJMoa050736</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>The prediction of drug sensitivity by multi-omics fusion reveals the heterogeneity of drug response in pan-cancer</article-title>. <source>Comput. Biol. Med.</source> <volume>163</volume>, <fpage>107220</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2023.107220</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hou</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Detecting anomalous anatomic regions in spatial transcriptomics with stands</article-title>. <source>Nat. Commun.</source> <volume>15</volume> (<issue>1</issue>), <fpage>8223</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-024-52445-9</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xue</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Meehan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X. Q. D.</given-names>
</name>
<name>
<surname>Fiset</surname>
<given-names>P. O.</given-names>
</name>
<name>
<surname>Rieker</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Smarca4 loss is synthetic lethal with cdk4/6 inhibition in non-small cell lung cancer</article-title>. <source>Nat. Commun.</source> <volume>10</volume> (<issue>1</issue>), <fpage>557</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-019-08380-1</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Retracted article: dual roles of mir-374a by modulated C-jun respectively targets ccnd1-inducing pi3k/akt signal and pten-suppressing wnt/&#x392;-catenin signaling in non-small-cell lung cancer</article-title>. <source>Cell Death and Dis.</source> <volume>9</volume> (<issue>2</issue>), <fpage>78</fpage>. <pub-id pub-id-type="doi">10.1038/s41419-017-0103-7</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Global and cross-modal feature aggregation for multi-omics data classification and application on drug response prediction</article-title>. <source>Inf. Fusion</source> <volume>102</volume>, <fpage>102077</fpage>. <pub-id pub-id-type="doi">10.1016/j.inffus.2023.102077</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Multi-omics fusion based on attention mechanism for survival and drug response prediction in digestive system tumors</article-title>, <source>Neurocomputing</source> <volume>572</volume>, <fpage>127168</fpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2023.127168</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>C.-R.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.-P.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>G.-Y.</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>J.-W.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Bevacizumab plus erlotinib in Chinese patients with untreated, egfr-mutated, advanced nsclc (Artemis-Ctong1509): a multicenter phase 3 study</article-title>. <source>Cancer Cell</source> <volume>39</volume> (<issue>9</issue>), <fpage>1279</fpage>&#x2013;<lpage>1291. e3</lpage>. <pub-id pub-id-type="doi">10.1016/j.ccell.2021.07.005</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Pache</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Khodabakhshi</surname>
<given-names>A. H.</given-names>
</name>
<name>
<surname>Tanaseichuk</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Metascape provides a biologist-oriented resource for the analysis of systems-level datasets</article-title>. <source>Nat. Commun.</source> <volume>10</volume> (<issue>1</issue>), <fpage>1523</fpage>&#x2013;<lpage>1610</lpage>. <pub-id pub-id-type="doi">10.1038/s41467-019-09234-6</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>