<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1132370</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2023.1132370</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A self-training subspace clustering algorithm based on adaptive confidence for gene expression data</article-title>
<alt-title alt-title-type="left-running-head">Li et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2023.1132370">10.3389/fgene.2023.1132370</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Dan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2217423/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liang</surname>
<given-names>Hongnan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2151983/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Qin</surname>
<given-names>Pan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/635358/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Wang</surname>
<given-names>Jia</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/703145/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Faculty of Electronic Information and Electrical Engineering</institution>, <institution>Dalian University of Technology</institution>, <addr-line>Dalian</addr-line>, <addr-line>Liaoning</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Breast Surgery</institution>, <institution>The Second Hospital of Dalian Medical University</institution>, <addr-line>Dalian</addr-line>, <addr-line>Liaoning</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1495169/overview">Cong Liu</ext-link>, Columbia University, United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/704184/overview">Hong Peng</ext-link>, Chinese Academy of Sciences (CAS), China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1272427/overview">Jiaogen Zhou</ext-link>, Huaiyin Normal University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Pan Qin, <email>qp112cn@dlut.edu.cn</email>; Jia Wang, <email>wangjia77@hotmail.com</email>
</corresp>
<fn fn-type="other">
<p>This article was submitted to Computational Genomics, a section of the journal Frontiers in Genetics</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>21</day>
<month>03</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>14</volume>
<elocation-id>1132370</elocation-id>
<history>
<date date-type="received">
<day>27</day>
<month>12</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>07</day>
<month>03</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Li, Liang, Qin and Wang.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Li, Liang, Qin and Wang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Gene clustering is one of the important techniques to identify co-expressed gene groups from gene expression data, which provides a powerful tool for investigating functional relationships of genes in biological process. Self-training is a kind of important semi-supervised learning method and has exhibited good performance on gene clustering problem. However, the self-training process inevitably suffers from mislabeling, the accumulation of which will lead to the degradation of semi-supervised learning performance of gene expression data. To solve the problem, this paper proposes a self-training subspace clustering algorithm based on adaptive confidence for gene expression data (SSCAC), which combines the low-rank representation of gene expression data and adaptive adjustment of label confidence to better guide the partition of unlabeled data. The superiority of the proposed SSCAC algorithm is mainly reflected in the following aspects. 1) In order to improve the discriminative property of gene expression data, the low-rank representation with distance penalty is used to mine the potential subspace structure of data. 2) Considering the problem of mislabeling in self-training, a semi-supervised clustering objective function with label confidence is proposed, and a self-training subspace clustering framework is constructed on this basis. 3) In order to mitigate the negative impact of mislabeled data, an adaptive adjustment strategy based on gravitational search algorithm is proposed for label confidence. Compared with a variety of state-of-the-art unsupervised and semi-supervised learning algorithms, the SSCAC algorithm has demonstrated its superiority through extensive experiments on two benchmark gene expression datasets.</p>
</abstract>
<kwd-group>
<kwd>self-training</kwd>
<kwd>subspace clustering</kwd>
<kwd>label confidence</kwd>
<kwd>adaptive adjustment</kwd>
<kwd>gravitational search algorithm</kwd>
<kwd>gene expression data</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>The recent development of biological experiments has generated vast amounts of gene expression data. Thus, comprehending and interpreting the enormous number of genes has become a significant challenge (<xref ref-type="bibr" rid="B5">Diniz et al., 2019</xref>; <xref ref-type="bibr" rid="B20">Ma&#xe2;touk et al., 2019</xref>; <xref ref-type="bibr" rid="B15">Li and Yang, 2020</xref>; <xref ref-type="bibr" rid="B30">Summers et al., 2020</xref>; <xref ref-type="bibr" rid="B25">Nisar et al., 2021</xref>; <xref ref-type="bibr" rid="B4">Dang et al., 2022</xref>). Semi-supervised learning (<xref ref-type="bibr" rid="B2">Chapelle et al., 2006</xref>) is a focused issue in the analysis of gene expression data, the research branches mainly include semi-supervised gene clustering (<xref ref-type="bibr" rid="B37">Yu et al., 2014</xref>; <xref ref-type="bibr" rid="B38">Yu et al., 2016</xref>; <xref ref-type="bibr" rid="B35">Xia et al., 2018</xref>; <xref ref-type="bibr" rid="B17">Liu et al., 2021</xref>), semi-supervised gene classification (<xref ref-type="bibr" rid="B9">Huang and Feng, 2012</xref>; <xref ref-type="bibr" rid="B39">Zhang et al., 2021</xref>), semi-supervised gene selection (<xref ref-type="bibr" rid="B21">Mahendran et al., 2020</xref>), and semi-supervised gene dimensionality reduction (<xref ref-type="bibr" rid="B7">Feng et al., 2021</xref>). In this paper, we focus on the semi-supervised gene clustering problem for identify co-expressed gene groups, which can provide a useful basis for the further investigation of gene function and gene regulation in the field of functional genomics (<xref ref-type="bibr" rid="B20">Ma&#xe2;touk et al., 2019</xref>). When clustering gene expression data, practical dataset usually exists in the form of a large amount of unlabeled data and a small amount of labeled data. However, unsupervised clustering algorithms inherently lack the ability to utilize the label information in exploring the pattern of gene expression data, and the clustering results are often unsatisfactory. Comparatively speaking, semi-supervised clustering can make full use of prior knowledge, such as pairwise information or class labels, to guide the partition of unlabeled data, thus can improve the clustering quality of gene expression data.</p>
<p>Most of the existing semi-supervised learning methods use raw data directly for analysis (<xref ref-type="bibr" rid="B8">Gan et al., 2013</xref>; <xref ref-type="bibr" rid="B34">Wu et al., 2018</xref>; <xref ref-type="bibr" rid="B14">Li et al., 2019</xref>). In recent years, many scholars have found in their research that the intrinsic structure of data is often smaller than its actual dimensionality, and it may be easier to mine the cluster structure of data in subspaces (<xref ref-type="bibr" rid="B1">Basri and Jacobs, 2003</xref>). Subspace-based low-dimensional feature representation of data has been successfully applied to various applications, such as image segmentation (<xref ref-type="bibr" rid="B16">Liu et al., 2013</xref>; <xref ref-type="bibr" rid="B6">Fei et al., 2017</xref>; <xref ref-type="bibr" rid="B36">Xu et al., 2023</xref>) and biological data analysis (<xref ref-type="bibr" rid="B29">Shi et al., 2019</xref>; <xref ref-type="bibr" rid="B32">Wang et al., 2019</xref>; <xref ref-type="bibr" rid="B40">Zheng et al., 2019</xref>; <xref ref-type="bibr" rid="B19">Lu et al., 2020</xref>; <xref ref-type="bibr" rid="B31">Sun et al., 2021</xref>; <xref ref-type="bibr" rid="B10">Huang and Wu, 2022</xref>). One of the representative algorithm is low-rank representation (LRR) (<xref ref-type="bibr" rid="B16">Liu et al., 2013</xref>), which assumes that the dataset is sampled from multiple mutually orthogonal subspaces in the data space, and uses rank to measure the sparsity of matrix. LRR only focuses on the global structure of data, and ignores the local structure hidden in data. To overcome this drawback, <xref ref-type="bibr" rid="B32">Wang et al. (2019)</xref> introduced mixed-norm and Laplacian regularization into LRR to identify differentially expressed genes for tumor clustering. <xref ref-type="bibr" rid="B19">Lu et al. (2020)</xref> incorporated the constraints of the non-negative symmetric low-rank matrix and graph regularization for cancer clustering. To preserve the neighbor relationship among data, <xref ref-type="bibr" rid="B6">Fei et al. (2017)</xref> proposed a low-rank representation algorithm with distance penalty (LRRADP), which adds a distance penalty term on the basis of LRR to ensure that the representation vectors of the neighboring data in the original data space are still close in the representation space, thereby enhancing the locality of the model and data discriminability. Aiming at guaranteeing block diagonal property of LRR, <xref ref-type="bibr" rid="B36">Xu et al. (2023)</xref> presented a projective block diagonal representation approach, which rapidly pursues a representation matrix with block diagonal structure. By assuming that cells with the same type are in the same subspace, <xref ref-type="bibr" rid="B40">Zheng et al. (2019)</xref> proposed a self-expression clustering method with non-negative and low-rank constraints for cell type detection. Besides, to effectively integrate multiple omics data, various multi-view subspace clustering algorithms based on LRR were developed for cancer subtyping (<xref ref-type="bibr" rid="B29">Shi et al., 2019</xref>; <xref ref-type="bibr" rid="B31">Sun et al., 2021</xref>; <xref ref-type="bibr" rid="B10">Huang and Wu, 2022</xref>).</p>
<p>As an essential semi-supervised learning method, self-training (<xref ref-type="bibr" rid="B24">Nie et al., 2012</xref>; <xref ref-type="bibr" rid="B8">Gan et al., 2013</xref>; <xref ref-type="bibr" rid="B34">Wu et al., 2018</xref>; <xref ref-type="bibr" rid="B35">Xia et al., 2018</xref>; <xref ref-type="bibr" rid="B14">Li et al., 2019</xref>) has been successfully applied to various applications including the analysis of gene expression data. Self-training can be regarded as a kind of self-learning method, which consists of two main steps (<xref ref-type="bibr" rid="B14">Li et al., 2019</xref>): semi-supervised learning using labeled data to update the predicted labels of unlabeled data; expansion of labeled dataset by selecting unlabeled data as newly labeled data based on some rules. These two steps are repeated until some stopping criteria are reached. For the task of self-training classification, <xref ref-type="bibr" rid="B8">Gan et al. (2013)</xref> suggested utilizing unlabeled and labeled data to reveal the true data space structure by cluster analysis, along with a semi-supervised fuzzy c-means technique, to improve self-training. However, the algorithm is not appropriate for non-spherically distributed data (<xref ref-type="bibr" rid="B34">Wu et al., 2018</xref>; <xref ref-type="bibr" rid="B14">Li et al., 2019</xref>). To overcome this weakness, <xref ref-type="bibr" rid="B34">Wu et al. (2018)</xref> proposed a method of self-training based on density peak of data (STDP), which uses clustering by fast search and find of density peaks (DPC) (<xref ref-type="bibr" rid="B28">Rodriguez and Laio, 2014</xref>) to build the density-pointing relationship between data, and newly labeled data are selected to iteratively strengthen the classification performance of SVM, KNN, and CART on this basis. Although STDP achieves good classification results for non-spherically distributed data, the problem of mislabeling in the self-training process is not considered. In fact, mislabeling of newly labeled data in a self-training approach is an unavoidable and very intractable problem (<xref ref-type="bibr" rid="B14">Li et al., 2019</xref>). Iterative self-training based on these mislabeled data will further reinforce the misinformation and generate more mislabels, leading to mistaken reinforcement (<xref ref-type="bibr" rid="B35">Xia et al., 2018</xref>; <xref ref-type="bibr" rid="B13">Li and Zhu, 2020</xref>). To solve this problem, researchers have proposed various self-training methods based on partial noise filters in recent years, including multi-label self-training with editing (<xref ref-type="bibr" rid="B33">Wei et al., 2013</xref>), dynamic safety assessment self-training based on semi-supervised learning and data editing (<xref ref-type="bibr" rid="B18">Liu et al., 2019</xref>), etc. To further exploit unlabeled data in the filter and overcome the parameter dependence problem, <xref ref-type="bibr" rid="B14">Li et al. (2019)</xref> proposed a self-training method based on density peaks and an extended parameter-free local noise filter (STDPNF), which can filter out part of mislabeled newly labeled data. However, as with other self-training algorithms using local noise filters, STDPNF still needs to entirely solve the problem of mislabeling.</p>
<p>On the other hand, for the self-training clustering task, <xref ref-type="bibr" rid="B24">Nie et al. (2012)</xref> proposed an active self-training clustering (ASTC), which utilizes Gaussian fields and harmonic functions (GFHF) (<xref ref-type="bibr" rid="B41">Zhu et al., 2003</xref>) to achieve label propagation. ASTC considers the probability of data being partitioned into various clusters as Bayesian posterior probability, and iteratively selects unlabeled data with large probability values as newly labeled data to optimize the label fitness process of GFHF and improve the label prediction accuracy. To address the problem of partitioning cancer gene expression data, <xref ref-type="bibr" rid="B35">Xia et al. (2018)</xref> proposed a self-training subspace clustering algorithm under low-rank representation (SSC-LRR), which introduces LRR to extract subspace structures from cancer gene expression data, iteratively clusters low-rank representation matrix and noise matrix using the K-means algorithm, and selects unlabeled data with the same clustering labels on the two matrices as newly labeled data for self-training learning. SSC-LRR achieves encouraging cancer classification on several benchmark gene expression datasets, and the advantage of low-rank representation in extracting discriminative features from data is analyzed through experimental results.</p>
<p>Despite the success of the above self-training methods, mislabeling a considerable amount of newly labeled data is inevitable (<xref ref-type="bibr" rid="B35">Xia et al., 2018</xref>; <xref ref-type="bibr" rid="B14">Li et al., 2019</xref>), and its accumulation will lead to the problem of mistaken reinforcement and seriously affect the performance of the self-training methods. In fact, in the self-training clustering problem on gene expression data, different newly labeled data should have different label confidences. The higher the semi-supervised learning value of a newly labeled datum, the more likely this datum has a correctly predicted label, so it should be assigned a higher label confidence. Based on the above analysis, for gene expression data with partial labels, a self-training subspace clustering algorithm based on adaptive confidence (SSCAC) is proposed in this paper, with the following main contributions. Firstly, a self-training subspace clustering framework based on GFHF is designed in this paper, which reveals the subspace structure of gene expression data through low-rank representation, and achieves iterative semi-supervised learning of unlabeled data using the label propagation capability of GFHF on the basis of the constructed similarity matrix. Secondly, to tackle the problem of mislabeling, an improved GFHF objective function with label confidence and the corresponding adaptive adjustment strategy of label confidence based on the gravitational search algorithm (<xref ref-type="bibr" rid="B27">Rashedi et al., 2009</xref>) are proposed. The negative impact of mislabeled data can be mitigated by reducing the label confidences of low-value newly labeled data, and the clustering accuracy on gene expression data can thus be improved.</p>
</sec>
<sec id="s2">
<title>2 Proposed algorithm</title>
<p>Although existing self-training methods have improved the partition accuracy of unlabeled data to some extent, the mislabeling problem of newly labeled data is still one of the important challenges in self-training methods (<xref ref-type="bibr" rid="B35">Xia et al., 2018</xref>; <xref ref-type="bibr" rid="B14">Li et al., 2019</xref>), which makes it difficult to accurately identify co-expressed gene groups on gene expression data with partial labels. During the iterative self-training, the falsely predicted labels will be accumulated gradually and lead to the problem of mistaken reinforcement. One major reason is that once the newly labeled data are selected, self-training methods always fully trust their predicted labels in the semi-supervised classification or clustering process, i.e., it is implicitly assumed that all newly labeled data have the same label confidence. This will obviously make both correctly and incorrectly labeled data act on the semi-supervised learning task with equal strength, and ignore the difference in value of different newly labeled data for semi-supervised learning. In view of this, a self-training subspace clustering algorithm based on adaptive confidence for gene expression data (SSCAC) is proposed in this paper. The proposed algorithm uses density relationships to select newly labeled data, and constructs a self-training subspace clustering framework based on GFHF and the low-rank representation with distance penalty. SSCAC differs from the existing self-training methods in that the semi-supervised clustering objective function with label confidence and the adaptive adjustment strategy of label confidences. The proposed algorithm aims to weaken the supervisory guidance of low-value newly labeled data by reducing their label confidences, thus alleviating the problem of mislabeling in the self-training process and improving the generalization ability of the algorithm.</p>
<sec id="s2-1">
<title>2.1 SSCAC objective function</title>
<p>Currently, low-rank representation has achieved good clustering results as a typical representation model for learning the subspace structure of gene expression data (<xref ref-type="bibr" rid="B35">Xia et al., 2018</xref>; <xref ref-type="bibr" rid="B29">Shi et al., 2019</xref>; <xref ref-type="bibr" rid="B32">Wang et al., 2019</xref>; <xref ref-type="bibr" rid="B40">Zheng et al., 2019</xref>; <xref ref-type="bibr" rid="B19">Lu et al., 2020</xref>; <xref ref-type="bibr" rid="B31">Sun et al., 2021</xref>; <xref ref-type="bibr" rid="B10">Huang and Wu, 2022</xref>). In this paper, the proposed SSCAC algorithm constructs a self-training subspace clustering framework based on the low-rank representation with distance penalty (LRRADP) (<xref ref-type="bibr" rid="B6">Fei et al., 2017</xref>) by using the high coordination between Gaussian fields and harmonic functions (GFHF) (<xref ref-type="bibr" rid="B41">Zhu et al., 2003</xref>) and low-rank representation.</p>
<p>In a semi-supervised learning framework, the dataset is usually formulated as <bold>
<italic>X</italic>
</bold> &#x3d; [<bold>
<italic>x</italic>
</bold>
<sub>1</sub>, <bold>
<italic>x</italic>
</bold>
<sub>2</sub>, &#x2026;, <bold>
<italic>x</italic>
</bold>
<sub>
<italic>l</italic>
</sub>, <bold>
<italic>x</italic>
</bold>
<sub>
<italic>l</italic>&#x2b;1</sub>, &#x2026;, <bold>
<italic>x</italic>
</bold>
<sub>
<italic>n</italic>
</sub>] &#x3d; [<bold>
<italic>X</italic>
</bold>
<sub>
<italic>L</italic>
</sub>, <bold>
<italic>X</italic>
</bold>
<sub>
<italic>U</italic>
</sub>] &#x2208; <bold>
<italic>R</italic>
</bold>
<sup>
<italic>m</italic>&#xd7;<italic>n</italic>
</sup>, where <inline-formula id="inf1">
<mml:math id="m1">
<mml:msubsup>
<mml:mrow>
<mml:mfenced open="" close="|">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> and <inline-formula id="inf2">
<mml:math id="m2">
<mml:msubsup>
<mml:mrow>
<mml:mfenced open="" close="|">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> are labeled and unlabeled data, <bold>
<italic>X</italic>
</bold>
<sub>
<italic>L</italic>
</sub> and <bold>
<italic>X</italic>
</bold>
<sub>
<italic>U</italic>
</sub> are the labeled and unlabeled datasets, <italic>c</italic> is the number of clusters, the corresponding label set is <bold>
<italic>L</italic>
</bold>
<sub>
<italic>a</italic>
</sub> &#x3d; {1, &#x2026;, <italic>c</italic>}, the label of datum <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> is <italic>y</italic>
<sub>
<italic>i</italic>
</sub> &#x2208; <bold>
<italic>L</italic>
</bold>
<sub>
<italic>a</italic>
</sub>. In order to make different newly labeled data act on semi-supervised gene clustering with different strengths, this paper introduces label confidence to GFHF semi-supervised clustering, and the proposed SSCAC objective function is:<disp-formula id="e1">
<mml:math id="m3">
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:munderover accentunder="false" accent="true">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:msup>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x221e;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:munderover accentunder="false" accent="true">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:msup>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:math>
<label>(1)</label>
</disp-formula>where <inline-formula id="inf3">
<mml:math id="m4">
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> is the label prediction matrix, vector <bold>
<italic>F</italic>
</bold>
<sub>
<italic>i</italic>
</sub> denotes the attribution of <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> to each cluster; <inline-formula id="inf4">
<mml:math id="m5">
<mml:mi mathvariant="bold-italic">Y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">Y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">B</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> is the binary label indication matrix, vector <bold>
<italic>Y</italic>
</bold>
<sub>
<italic>i</italic>
</sub> corresponds to the label of <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub>, if the label <italic>y</italic>
<sub>
<italic>i</italic>
</sub> &#x3d; <italic>k</italic> then <italic>Y</italic>
<sub>
<italic>ik</italic>
</sub> &#x3d; 1, otherwise <italic>Y</italic>
<sub>
<italic>ik</italic>
</sub> &#x3d; 0; <italic>&#x3bb;</italic>
<sub>
<italic>&#x221e;</italic>
</sub> is a very large constant; <italic>&#x3bc;</italic>
<sub>
<italic>i</italic>
</sub> &#x2208; (0, 1] is the label confidence of the labeled datum <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub>(<italic>i</italic> &#x3d; 1, 2, &#x2026;, <italic>l</italic>); <italic>W</italic>
<sub>
<italic>ij</italic>
</sub> is the element of the LRRADP affinity matrix <bold>
<italic>W</italic>
</bold> obtained by:<disp-formula id="e2">
<mml:math id="m6">
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
</mml:math>
<label>(2)</label>
</disp-formula>
</p>
<p>In LRRADP, <bold>
<italic>Z</italic>
</bold> &#x2208; <bold>
<italic>R</italic>
</bold>
<sup>
<italic>n</italic>&#xd7;<italic>n</italic>
</sup> is the low-rank representation matrix and <bold>
<italic>Z</italic>
</bold>
<sub>
<italic>i</italic>
</sub> is the vector of coefficients of datum <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> represented by other data; <bold>
<italic>E</italic>
</bold> &#x2208; <bold>
<italic>R</italic>
</bold>
<sup>
<italic>m</italic>&#xd7;<italic>n</italic>
</sup> is the noise matrix. The iterative update equations are as follows (<xref ref-type="bibr" rid="B6">Fei et al., 2017</xref>):<disp-formula id="e3">
<mml:math id="m7">
<mml:mtable class="array">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>arg</mml:mi>
<mml:munder>
<mml:mrow>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;">
<mml:mrow>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">&#x39b;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">&#x39b;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
<label>(3)</label>
</disp-formula>
<disp-formula id="e4">
<mml:math id="m8">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>arg</mml:mi>
<mml:munder>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">E</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:msubsup>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;">
<mml:mrow>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">&#x39b;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:math>
<label>(4)</label>
</disp-formula>where &#x2016;<bold>
<italic>Z</italic>
</bold>&#x2016;<sub>&#x2a;</sub> &#x3d; <italic>&#x2211;</italic>
<sub>
<italic>i</italic>
</sub>
<italic>&#x3c3;</italic>
<sub>
<italic>i</italic>
</sub>(<bold>
<italic>Z</italic>
</bold>) is the nuclear norm of <bold>
<italic>Z</italic>
</bold>, which is used as a convex approximation of matrix rank, <italic>&#x3c3;</italic>
<sub>
<italic>i</italic>
</sub>(<bold>
<italic>Z</italic>
</bold>) denotes the <italic>i</italic>-th singular value of <bold>
<italic>Z</italic>
</bold>; &#x2016;.&#x2016;<sub>1</sub> and &#x2016;.&#x2016;<sub>2</sub> are the <italic>l</italic>
<sub>1</sub>-norm and <italic>l</italic>
<sub>2</sub>-norm, respectively; auxiliary variable <bold>
<italic>H</italic>
</bold>, Lagrange multipliers <bold>&#x39b;</bold>
<sub>1</sub>, <bold>&#x39b;</bold>
<sub>2</sub> and penalty parameter <italic>&#x3b2;</italic> are determined by the following equations:<disp-formula id="e5">
<mml:math id="m9">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>arg</mml:mi>
<mml:munder>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">H</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">&#x39e;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold-italic">D</mml:mi>
<mml:mo>&#x2297;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:msubsup>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">&#x39b;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:math>
<label>(5)</label>
</disp-formula>
<disp-formula id="e6">
<mml:math id="m10">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">&#x39b;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">&#x39b;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(6)</label>
</disp-formula>
<disp-formula id="e7">
<mml:math id="m11">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">&#x39b;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">&#x39b;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(7)</label>
</disp-formula>
<disp-formula id="e8">
<mml:math id="m12">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>min</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c1;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(8)</label>
</disp-formula>
</p>
<p>In the update equations, <italic>&#x3bb;</italic>
<sub>1</sub> &#x3e; 0 and <italic>&#x3bb;</italic>
<sub>2</sub> &#x3e; 0 are balance parameters to trade off among the low-rank representation, noise and adaptive distance penalty.</p>
<p>In the SSCAC objective function defined by Eq. <xref ref-type="disp-formula" rid="e1">1</xref>, the first term is the same as that of the original GFHF, which ensures the smoothness of data labels on the LRRADP graph. The second term is the label fitness term, which incorporates the label confidence <italic>&#x3bc;</italic>
<sub>
<italic>i</italic>
</sub> and applies it to the label indication vector <bold>
<italic>Y</italic>
</bold>
<sub>
<italic>i</italic>
</sub> of the labeled datum <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub>. Actually, the objective function of GFHF is a special case of that of SSCAC with <italic>&#x3bc;</italic>
<sub>
<italic>i</italic>
</sub> &#x3d; 1 for each labeled datum <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub>(<italic>i</italic> &#x3d; 1, 2, &#x2026;, <italic>l</italic>). That is, the SSCAC objective function is the extension of that of GHFH, which further considers the label confidences of the labeled data and can be applied to self-training clustering. Minimizing Eq. <xref ref-type="disp-formula" rid="e1">1</xref> can achieve both the manifold smoothness of the partition results in subspaces and the maximum matching between the predicted label and the label of labeled data under the effect of label confidence.</p>
<p>In the self-training process of SSCAC, newly labeled data are selected based on density-pointing relationships between data (<xref ref-type="bibr" rid="B34">Wu et al., 2018</xref>; <xref ref-type="bibr" rid="B14">Li et al., 2019</xref>) and added to the labeled dataset <bold>
<italic>X</italic>
</bold>
<sub>
<italic>L</italic>
</sub> to guide the next iteration of self-training learning. The newly labeled data selection strategy will be detailed in the next Subsection. The rules for setting the label confidence <italic>&#x3bc;</italic>
<sub>
<italic>i</italic>
</sub> in Eq. <xref ref-type="disp-formula" rid="e1">1</xref> are as follows: 1) if <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> is an initially labeled datum, set the label confidence <italic>&#x3bc;</italic>
<sub>
<italic>i</italic>
</sub> &#x3d; 1 with complete confidence in its label accuracy; 2) if <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> is a newly labeled datum of the current iteration of self-training, <italic>&#x3bc;</italic>
<sub>
<italic>i</italic>
</sub> is initialized to a random number within (0,1], and then adaptively adjusted based on the semi-supervised learning value of <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub>. The specific strategy is detailed in <xref ref-type="sec" rid="s2-3">Section 2.3</xref>; 3) only the label confidences of the newly labeled data selected in the current iteration are adjusted, the adjusted confidences remain unchanged in the subsequent iterations of self-training.</p>
<p>The advantage of adding the label confidence in Eq. <xref ref-type="disp-formula" rid="e1">1</xref> is that the value can effectively regulate the supervision strength of newly labeled data on semi-supervised gene clustering, which improves the clustering accuracy on gene expression data. The analysis is as follows: 1) if the newly labeled datum <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> is mislabeled, i.e., the position of 1 in the label indication vector <bold>
<italic>Y</italic>
</bold>
<sub>
<italic>i</italic>
</sub> does not match that of the actual label, the label prediction vector <bold>
<italic>F</italic>
</bold>
<sub>
<italic>i</italic>
</sub> will be predicted in the wrong direction under the effect of the second term of Eq. <xref ref-type="disp-formula" rid="e1">1</xref>, and the larger the label confidence <italic>&#x3bc;</italic>
<sub>
<italic>i</italic>
</sub>, the larger the prediction bias. In the first term of Eq. <xref ref-type="disp-formula" rid="e1">1</xref>, the elements corresponding to data in the same subspace in the LRRADP similarity matrix <bold>
<italic>W</italic>
</bold> are relatively large and those corresponding to data in different subspaces are small, so that labels are mainly propagated among data in the same subspace, then the mislabeled datum <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> will lead to the label prediction bias of unlabeled data in the same gene clustering. Therefore, reducing the label confidence of mislabeled datum <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> can effectively mitigate its negative impact on semi-supervised gene clustering; 2) if the newly labeled datum <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> has correct label, the second term of Eq. <xref ref-type="disp-formula" rid="e1">1</xref> can guide <bold>
<italic>F</italic>
</bold>
<sub>
<italic>i</italic>
</sub> to obtain correct prediction, and then realize correct label propagation for unlabeled data in the same subspace under the effect of the first term of Eq. <xref ref-type="disp-formula" rid="e1">1</xref>. Obviously, increasing the label confidence of correctly labeled datum is beneficial to improve the partition accuracy of unlabeled data.</p>
<p>The matrix form of the SSCAC objective function is:<disp-formula id="e9">
<mml:math id="m13">
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mi mathvariant="bold-italic">L</mml:mi>
<mml:mi mathvariant="bold-italic">F</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
<mml:mo>&#x2297;</mml:mo>
<mml:mi mathvariant="bold-italic">Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mi mathvariant="bold-italic">U</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
<mml:mo>&#x2297;</mml:mo>
<mml:mi mathvariant="bold-italic">Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(9)</label>
</disp-formula>where <bold>
<italic>L</italic>
</bold> &#x2208; <bold>
<italic>R</italic>
</bold>
<sup>
<italic>n</italic>&#xd7;<italic>n</italic>
</sup> is the graph Laplacian matrix, <bold>
<italic>L</italic>
</bold> &#x3d; <bold>
<italic>D</italic>
</bold> &#x2212; <bold>
<italic>W</italic>
</bold>, <bold>
<italic>D</italic>
</bold> is a diagonal matrix, <italic>D</italic>
<sub>
<italic>ii</italic>
</sub> &#x3d; <italic>&#x2211;</italic>
<sub>
<italic>j</italic>
</sub>
<italic>W</italic>
<sub>
<italic>i</italic>,<italic>j</italic>
</sub>;<bold>
<italic>U</italic>
</bold> &#x2208; <bold>
<italic>R</italic>
</bold>
<sup>
<italic>n</italic>&#xd7;<italic>n</italic>
</sup> is also a diagonal matrix, the first <italic>l</italic> and the remaining <italic>n</italic> &#x2212; <italic>l</italic> diagonal elements are <italic>&#x3bb;</italic>
<sub>
<italic>&#x221e;</italic>
</sub> and 0, respectively; &#x2297; denotes the Hadamard product; <bold>
<italic>&#x3bc;</italic>
</bold> &#x2208; <bold>
<italic>R</italic>
</bold>
<sup>
<italic>n</italic>&#xd7;<italic>c</italic>
</sup>, if the label of <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub>(<italic>i</italic> &#x3d; 1, 2, &#x2026;, <italic>l</italic>) is <italic>k</italic>(<italic>k</italic> &#x3d; 1, 2, &#x2026;, <italic>c</italic>), then the <italic>k</italic>-th element in the <italic>i</italic>-th row vector is the label confidence of <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub>, and all the other elements in the row vector are 0. For each unlabeled datum, all elements in the corresponding row vector are set to 0. By setting the derivative of Eq. <xref ref-type="disp-formula" rid="e9">9</xref> with respect to <bold>
<italic>F</italic>
</bold> to zero, the following equation can be easily obtained:<disp-formula id="e10">
<mml:math id="m14">
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold-italic">L</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="bold-italic">U</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mi mathvariant="bold-italic">U</mml:mi>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
<mml:mi mathvariant="bold-italic">Y</mml:mi>
</mml:math>
<label>(10)</label>
</disp-formula>
</p>
<p>Then the predicted label of the unlabeled datum <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> can be assigned by:<disp-formula id="e11">
<mml:math id="m15">
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>arg</mml:mi>
<mml:munder>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
<label>(11)</label>
</disp-formula>
</p>
</sec>
<sec id="s2-2">
<title>2.2 Newly labeled data selection strategy based on density relationships</title>
<p>In the self-training process, how to select newly labeled data from the unlabeled dataset <bold>
<italic>X</italic>
</bold>
<sub>
<italic>U</italic>
</sub> and iteratively expand the labeled dataset <bold>
<italic>X</italic>
</bold>
<sub>
<italic>L</italic>
</sub> is an important issue. Most self-training learning methods (<xref ref-type="bibr" rid="B24">Nie et al., 2012</xref>; <xref ref-type="bibr" rid="B35">Xia et al., 2018</xref>) rely entirely on the performance of learning models and ignore the potential density information in datasets. Relatively speaking, the strategy based on the data density relationships is not restricted by the distribution of initially labeled data and entire data space (<xref ref-type="bibr" rid="B34">Wu et al., 2018</xref>), and is more suitable for self-training learning on non-spherically distributed data.</p>
<p>In the self-training process of SSCAC, newly labeled data are selected based on density-pointing relationships between data (<xref ref-type="bibr" rid="B34">Wu et al., 2018</xref>; <xref ref-type="bibr" rid="B14">Li et al., 2019</xref>). The strategy utilizes clustering by fast search and find of density peaks (DPC) (<xref ref-type="bibr" rid="B28">Rodriguez and Laio, 2014</xref>), and for each datum <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub>, its local density <italic>&#x3c1;</italic>
<sub>
<italic>i</italic>
</sub> can be defined as:<disp-formula id="e12">
<mml:math id="m16">
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mi>&#x3c7;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mi>&#x3c7;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="{" close="">
<mml:mrow>
<mml:mtable class="cases">
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mi>x</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mspace width="1em"/>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mi>x</mml:mi>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mspace width="1em"/>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(12)</label>
</disp-formula>where <italic>d</italic>
<sub>
<italic>ij</italic>
</sub> is the Euclidean distance between <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> and <bold>
<italic>x</italic>
</bold>
<sub>
<italic>j</italic>
</sub>, and <italic>d</italic>
<sub>
<italic>c</italic>
</sub> is the cut-off distance. It can be seen that the value of local density <italic>&#x3c1;</italic>
<sub>
<italic>i</italic>
</sub> is the number of data whose distance from <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> is less than <italic>d</italic>
<sub>
<italic>c</italic>
</sub>. In addition, DPC defines the minimum distance between <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> and other data with higher local densities as follows:<disp-formula id="e13">
<mml:math id="m17">
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="{" close="">
<mml:mrow>
<mml:mtable class="array">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mo>&#x2200;</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2265;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2003;others</mml:mtext>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(13)</label>
</disp-formula>
</p>
<p>The newly labeled data selection strategy calculates <italic>&#x3c1;</italic>
<sub>
<italic>i</italic>
</sub> and <italic>&#x3b4;</italic>
<sub>
<italic>i</italic>
</sub> for each datum <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> and make <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> point to its nearest datum <bold>
<italic>x</italic>
</bold>
<sub>
<italic>j</italic>
</sub> with a higher local density, then <bold>
<italic>x</italic>
</bold>
<sub>
<italic>j</italic>
</sub> is called the &#x201c;next&#x201d; datum of <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> and <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> is the &#x201c;previous&#x201d; datum of <bold>
<italic>x</italic>
</bold>
<sub>
<italic>j</italic>
</sub>. Then, the strategy constructs the density-pointing relationships of low-density data to high-density data by selecting the &#x201c;next&#x201d; and &#x201c;previous&#x201d; unlabeled data of labeled data in batches and set their selection orders. Specifically, all the &#x201c;next&#x201d; data of data in the original labeled dataset <bold>
<italic>X</italic>
</bold>
<sub>
<italic>L</italic>
</sub> are firstly selected from the unlabeled dataset <bold>
<italic>X</italic>
</bold>
<sub>
<italic>U</italic>
</sub>, and their selection orders are set to 1. That is, these data are viewed as the ones that should be labeled in the first iteration of self-training and used as the newly labeled data to expand the labeled dataset. In the next iteration, all the &#x201c;next&#x201d; data of the newly labeled data of the previous iteration are selected from <bold>
<italic>X</italic>
</bold>
<sub>
<italic>U</italic>
</sub>, and their selection orders increase by 1. This step repeats until there exists no &#x201c;next&#x201d; data of the newly labeled data of the previous iteration in <bold>
<italic>X</italic>
</bold>
<sub>
<italic>U</italic>
</sub>. If there are still unselected data in <bold>
<italic>X</italic>
</bold>
<sub>
<italic>U</italic>
</sub>, the selection orders of these remaining data can be set according to the &#x201c;previous&#x201d; relationships using the similar process. It can be seen that the unlabeled data with the same selection orders form the newly labeled dataset of the same iteration of self-training, on which basis the proposed SSCAC algorithm can expand the labeled dataset <bold>
<italic>X</italic>
</bold>
<sub>
<italic>L</italic>
</sub> iteratively and realize self-training clustering.</p>
</sec>
<sec id="s2-3">
<title>2.3 Adaptive adjustment of label confidence based on gravitational search algorithm</title>
<p>According to the analysis of the SSCAC objective function in the previous subsection, it is obvious that the value of different newly labeled data should vary for semi-supervised learning. If the newly labeled datum <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> is mislabeled, its incorrect label will propagate to the unlabeled data in the same subspace, making these data together with <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> have significant differences in the label prediction vectors from those of the correctly labeled data in that subspace. In this case, Eq. <xref ref-type="disp-formula" rid="e1">1</xref> will inevitably result in a large function value, and <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> can be regarded as a low-value newly labeled datum. Conversely, the newly labeled datum <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> with correct label can propagate its correct label in the subspace it belongs to, so that the unlabeled data in this subspace will obtain similar label prediction vectors to those of the correctly labeled data. In this case, the objective function value of Eq. <xref ref-type="disp-formula" rid="e1">1</xref> will be relatively small, and <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> can be regarded as a high-value newly labeled datum. Therefore, the SSCAC algorithm proposed in this paper measures the semi-supervised learning value of newly labeled data by the objective function value of Eq. <xref ref-type="disp-formula" rid="e1">1</xref>, and on this basis, achieves the adaptive adjustment of label confidence.</p>
<p>Gravitational search algorithm (GSA) (<xref ref-type="bibr" rid="B27">Rashedi et al., 2009</xref>) is an optimization method based on the law of gravity, which is easy to implement and requires fewer parameters. It has been proven in the literature that GSA outperforms heuristic search algorithms such as PSO and GA (<xref ref-type="bibr" rid="B23">Mirjalili et al., 2012</xref>; <xref ref-type="bibr" rid="B12">Kumar et al., 2013</xref>). The search particles in GSA are a set of individuals that attract each other and generate motion in the solution space, the position of the individual is the solution of the optimization problem. Under the influence of gravity, the individuals move toward the individuals with heavier masses, which correspond to better solutions. To distinguish from the iterations of self-training learning, the iteration index of GSA is referred to as time in this paper. In the <italic>r</italic>-th iteration of self-training, let <italic>I</italic> be the number of newly labeled data, <inline-formula id="inf5">
<mml:math id="m18">
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula> be the set of newly labeled data, we use the label confidences of these newly labeled data to compose the label confidence vector. Specifically, the label confidence vector can be represented as the positions of particles when optimized by GSA, the position of GSA particle <italic>a</italic> at time <italic>t</italic> is defined by:<disp-formula id="e14">
<mml:math id="m19">
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mi>a</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1,2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>N</mml:mi>
</mml:math>
<label>(14)</label>
</disp-formula>where <italic>N</italic> is the population size, <italic>&#x3bc;</italic>
<sub>
<italic>i</italic>,<italic>a</italic>
</sub>(<italic>t</italic>) is the label confidence of the <italic>i</italic>-th newly labeled datum <inline-formula id="inf6">
<mml:math id="m20">
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> in particle <italic>a</italic> at time <italic>t</italic>, which is initialized to a random number within (0,1].</p>
<p>Based on the SSCAC objective function given in Eq. <xref ref-type="disp-formula" rid="e9">9</xref>, the GSA fitness function of particle <italic>a</italic> at time <italic>t</italic> is defined as:<disp-formula id="e15">
<mml:math id="m21">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">fi</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mi mathvariant="bold-italic">L</mml:mi>
<mml:mi mathvariant="bold-italic">F</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2297;</mml:mo>
<mml:mi mathvariant="bold-italic">Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mi mathvariant="bold-italic">U</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold-italic">F</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2297;</mml:mo>
<mml:mi mathvariant="bold-italic">Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(15)</label>
</disp-formula>
</p>
<p>For the <italic>i</italic>-th newly labeled datum <inline-formula id="inf7">
<mml:math id="m22">
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula>, the force acting on particle <italic>a</italic> from particle <italic>b</italic> at time <italic>t</italic> is expressed as:<disp-formula id="e16">
<mml:math id="m23">
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>G</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b5;</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(16)</label>
</disp-formula>where <italic>G</italic>(<italic>t</italic>) is gravitational constant at time <italic>t</italic>, <italic>M</italic>
<sub>
<italic>a</italic>
</sub>(<italic>t</italic>) and <italic>M</italic>
<sub>
<italic>b</italic>
</sub>(<italic>t</italic>) are the inertial masses of particle <italic>a</italic> and particle <italic>b</italic>, <italic>R</italic>
<sub>
<italic>ab</italic>
</sub>(<italic>t</italic>) is the Euclidean distance between particle <italic>a</italic> and particle <italic>b</italic>, and <italic>&#x25b;</italic> is a tiny constant to avoid zero denominator. The following equation can be used to determine the gravitational constant <italic>G</italic>(<italic>t</italic>):<disp-formula id="e17">
<mml:math id="m24">
<mml:mi>G</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:math>
<label>(17)</label>
</disp-formula>where <italic>G</italic>
<sub>0</sub> is the initial value of the gravitational coefficient, <italic>&#x3b1;</italic> is the decay coefficient, <italic>G</italic>
<sub>0</sub> and <italic>&#x3b1;</italic> are usually taken as 100 and 20 (<xref ref-type="bibr" rid="B27">Rashedi et al., 2009</xref>), and <italic>T</italic> is the maximum time.</p>
<p>During the motion of a particle, the inertial mass <italic>M</italic>
<sub>
<italic>a</italic>
</sub>(<italic>t</italic>) of particle <italic>a</italic> can be updated according to the adapted value:<disp-formula id="e18">
<mml:math id="m25">
<mml:msub>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">fi</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="normal">w</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">b</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="normal">w</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(18)</label>
</disp-formula>
<disp-formula id="e19">
<mml:math id="m26">
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mo movablelimits="false" form="prefix">&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(19)</label>
</disp-formula>where <italic>m</italic>
<sub>
<italic>a</italic>
</sub>(<italic>t</italic>) is the intermediate variable, best(<italic>t</italic>) and worst(<italic>t</italic>) are the best and worst fitness values among all particles at time <italic>t</italic>, respectively. In this paper, the particle position that makes the fitness value Eq. <xref ref-type="disp-formula" rid="e15">15</xref> obtain the minimum value is selected as the label confidence of the newly labeled data. Here, best(<italic>t</italic>) and worst(<italic>t</italic>) are respectively given by:<disp-formula id="e20">
<mml:math id="m27">
<mml:mi mathvariant="normal">b</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:munder>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mfenced open="{" close="}">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">fi</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(20)</label>
</disp-formula>
<disp-formula id="e21">
<mml:math id="m28">
<mml:mi mathvariant="normal">w</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:munder>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mfenced open="{" close="}">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">fi</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(21)</label>
</disp-formula>
</p>
<p>According to Newtonian gravity and the laws of motion, the gravitational force on particle <italic>a</italic> in the <italic>i</italic>-th dimension at time <italic>t</italic> is the sum of the gravitational forces from all other particles.<disp-formula id="e22">
<mml:math id="m29">
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:munderover accentunder="false" accent="true">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>b</mml:mi>
<mml:mo>&#x2260;</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(22)</label>
</disp-formula>where rand<sub>
<italic>b</italic>
</sub> is a random number within [0,1]. According to Newton&#x2019;s second law, the acceleration of particle <italic>a</italic> in the <italic>i</italic>-th dimension is:<disp-formula id="e23">
<mml:math id="m30">
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(23)</label>
</disp-formula>
</p>
<p>Therefore, the velocity and position of particle <italic>a</italic> in the <italic>i</italic>-th dimension at the next time are updated by:<disp-formula id="e24">
<mml:math id="m31">
<mml:mfenced open="{" close="">
<mml:mrow>
<mml:mtable class="array">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(24)</label>
</disp-formula>where rand<sub>
<italic>a</italic>
</sub> is a random number within [0,1], the initial velocity <italic>v</italic>
<sub>
<italic>i</italic>,<italic>a</italic>
</sub>(0) is 0.</p>
<p>When time <italic>t</italic> reaches <italic>T</italic>, the position of the particle that obtains the minimum fitness value is used as the label confidence vector <bold>
<italic>&#x3bc;</italic>
</bold>
<sup>
<italic>r</italic>
</sup> for the newly labeled data <bold>
<italic>X</italic>
</bold>
<sup>
<italic>r</italic>
</sup> at the <italic>r</italic>-th iteration of self-training. Then, we can update <bold>
<italic>&#x3bc;</italic>
</bold>, <bold>
<italic>U</italic>
</bold> and <bold>
<italic>Y</italic>
</bold> in Eq. <xref ref-type="disp-formula" rid="e9">9</xref> based on the obtained label confidence vector <bold>
<italic>&#x3bc;</italic>
</bold>
<sup>
<italic>r</italic>
</sup>, the newly labeled data <bold>
<italic>X</italic>
</bold>
<sup>
<italic>r</italic>
</sup>, their predicted labels respectively, and guide the subsequent iterations of self-training. It can be seen that the proposed strategy can adaptively adjust the label confidence based on the semi-supervised learning value of the newly labeled data. By reducing the label confidences of low-value newly labeled data, we can effectively reduce their effect on semi-supervised learning and thus alleviate the problem of mistaken reinforcement in the self-training gene clustering.</p>
</sec>
<sec id="s2-4">
<title>2.4 The procedure of the proposed SSCAC algorithm</title>
<p>For a set of gene expression data <bold>
<italic>X</italic>
</bold> &#x3d; [<bold>
<italic>x</italic>
</bold>
<sub>1</sub>, <bold>
<italic>x</italic>
</bold>
<sub>2</sub>, &#x2026;, <bold>
<italic>x</italic>
</bold>
<sub>
<italic>l</italic>
</sub>, <bold>
<italic>x</italic>
</bold>
<sub>
<italic>l</italic>&#x2b;1</sub>, &#x2026;, <bold>
<italic>x</italic>
</bold>
<sub>
<italic>n</italic>
</sub>] &#x3d; [<bold>
<italic>X</italic>
</bold>
<sub>
<italic>L</italic>
</sub>, <bold>
<italic>X</italic>
</bold>
<sub>
<italic>U</italic>
</sub>] &#x2208; <bold>
<italic>R</italic>
</bold>
<sup>
<italic>m</italic>&#xd7;<italic>n</italic>
</sup>, the detailed procedure of SSCAC is given in <xref ref-type="statement" rid="Algorithm_1">Algorithm 1</xref>, and the framework of SSCAC is shown in <xref ref-type="fig" rid="F1">Figure 1</xref>. In SSCAC, the stopping condition is set to <bold>
<italic>X</italic>
</bold>
<sub>
<italic>U</italic>
</sub> &#x3d; &#x2205; or the clustering accuracy no longer increases as suggested in the literature (<xref ref-type="bibr" rid="B26">Qu et al., 2019</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>A framework of SSCAC.</p>
</caption>
<graphic xlink:href="fgene-14-1132370-g001.tif"/>
</fig>
<p>
<statement content-type="algorithm" id="Algorithm_1">
<label>Algorithm 1</label>
<p>Note that when the stopping condition is that the clustering accuracy no longer increases, the labels of the remaining data in <bold>
<italic>X</italic>
</bold>
<sub>
<italic>U</italic>
</sub> are obtained based on <bold>
<italic>F</italic>
</bold>.<list list-type="simple">
<list-item>
<p>Step 1: Set the parameters, including maximum value of penalty parameter <italic>&#x3b2;</italic>
<sub>
<italic>max</italic>
</sub>, iteration stop parameter <italic>&#x3be;</italic>, constant <italic>&#x3c1;</italic>, balance parameters <italic>&#x3bb;</italic>
<sub>1</sub> and <italic>&#x3bb;</italic>
<sub>2</sub> of the LRRADP algorithm, and population size <italic>N</italic>, maximum time <italic>T</italic>, constant <italic>&#x25b;</italic> of the GSA algorithm.</p>
</list-item>
<list-item>
<p>Step 2: For each datum <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> in <bold>
<italic>X</italic>
</bold>, initialize its selection order <bold>
<italic>O</italic>
</bold>(<italic>i</italic>) &#x3d; 0, calculate <italic>&#x3c1;</italic>
<sub>
<italic>i</italic>
</sub>, <italic>&#x3b4;</italic>
<sub>
<italic>i</italic>
</sub> according to Eqs <xref ref-type="disp-formula" rid="e12">12</xref>, <xref ref-type="disp-formula" rid="e13">13</xref>, and find the &#x201c;next&#x201d; and &#x201c;previous&#x201d; data of <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> based on <italic>&#x3c1;</italic>
<sub>
<italic>i</italic>
</sub>, <italic>&#x3b4;</italic>
<sub>
<italic>i</italic>
</sub>. Set the iteration index of the unlabeled data selection <italic>r</italic> &#x3d; 1, then set the selection order of unlabeled data by the following steps.</p>
<list list-type="simple">
<list-item>
<p>1) For each datum <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> in <bold>
<italic>X</italic>
</bold>
<sub>
<italic>U</italic>
</sub>, if <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> is the &#x201c;next&#x201d; datum of a datum in <bold>
<italic>X</italic>
</bold>
<sub>
<italic>L</italic>
</sub>, set its selection order <bold>
<italic>O</italic>
</bold>(<italic>i</italic>) &#x3d; <italic>r</italic>.</p>
</list-item>
<list-item>
<p>2) Set <italic>r</italic> &#x3d; <italic>r</italic> &#x2b; 1. For each unselected datum <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> in <bold>
<italic>X</italic>
</bold>
<sub>
<italic>U</italic>
</sub>, if <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> is the &#x201c;next&#x201d; datum of a datum whose selection order is <italic>r</italic> &#x2212; 1, set <bold>
<italic>O</italic>
</bold>(<italic>i</italic>) &#x3d; <italic>r</italic>.</p>
</list-item>
<list-item>
<p>3) If there still exists &#x201c;next&#x201d; data of data whose selection orders are <italic>r</italic> in <bold>
<italic>X</italic>
</bold>
<sub>
<italic>U</italic>
</sub>, then return to 2); otherwise, set <italic>r</italic> &#x3d; <italic>r</italic> &#x2b; 1 and go to 4).</p>
</list-item>
<list-item>
<p>4) For each unselected datum <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> in <bold>
<italic>X</italic>
</bold>
<sub>
<italic>U</italic>
</sub>, if <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> is the &#x201c;previous&#x201d; datum of the selected data, then set <bold>
<italic>O</italic>
</bold>(<italic>i</italic>) &#x3d; <italic>r</italic>.</p>
</list-item>
<list-item>
<p>5) Set <italic>r</italic> &#x3d; <italic>r</italic> &#x2b; 1. For each unselected datum <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> in <bold>
<italic>X</italic>
</bold>
<sub>
<italic>U</italic>
</sub>, if <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> is the &#x201c;previous&#x201d; datum of a datum whose selection order is <italic>r</italic> &#x2212; 1, set <bold>
<italic>O</italic>
</bold>(<italic>i</italic>) &#x3d; <italic>r</italic>.</p>
</list-item>
<list-item>
<p>6) If there still exists &#x201c;previous&#x201d; data of data whose selection orders are <italic>r</italic> in <bold>
<italic>X</italic>
</bold>
<sub>
<italic>U</italic>
</sub>, then return to 5); otherwise, get the vector <bold>
<italic>O</italic>
</bold> of selection order for unlabeled data and go to Step3.</p>
</list-item>
</list>
</list-item>
<list-item>
<p>Step 3: Initialize <bold>
<italic>Z</italic>
</bold> &#x3d; <bold>
<italic>H</italic>
</bold> &#x3d; <bold>
<italic>E</italic>
</bold> &#x3d; <bold>&#x39b;</bold>
<sub>1</sub> &#x3d; <bold>&#x39b;</bold>
<sub>2</sub> &#x3d; 0, <italic>&#x3b2;</italic>
<sub>0</sub> &#x3d; 1. Set the iteration index of the LRRADP algotirhm <italic>p</italic> &#x3d; 0, calculate Eqs <xref ref-type="disp-formula" rid="e3">3</xref>&#x2013;<xref ref-type="disp-formula" rid="e8">8</xref> iteratively until <inline-formula id="inf8">
<mml:math id="m32">
<mml:mfenced open="&#x2016;" close="&#x2016;">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>/</mml:mo>
<mml:mfenced open="&#x2016;" close="&#x2016;">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2265;</mml:mo>
<mml:mi>&#x3be;</mml:mi>
</mml:math>
</inline-formula> to obtain the low-rank representation matrix <bold>
<italic>Z</italic>
</bold> of <bold>
<italic>X</italic>
</bold>, and get the similarity matrix <bold>
<italic>W</italic>
</bold> &#x3d; (<bold>
<italic>Z</italic>
</bold> &#x2b; <bold>
<italic>Z</italic>
</bold>
<sup>
<italic>T</italic>
</sup>)/2. Set the iteration index of self-training <italic>r</italic> &#x3d; 1, initialize <bold>
<italic>U</italic>
</bold> and <bold>
<italic>Y</italic>
</bold> based on initial <bold>
<italic>X</italic>
</bold>
<sub>
<italic>L</italic>
</sub>, set label confidence <bold>
<italic>&#x3bc;</italic>
</bold>
<sub>
<italic>i</italic>
</sub> &#x3d; 1 for each datum in <bold>
<italic>X</italic>
</bold>
<sub>
<italic>L</italic>
</sub>, get initial predicted labels according to Eqs <xref ref-type="disp-formula" rid="e10">10</xref>, <xref ref-type="disp-formula" rid="e11">11</xref>.</p>
</list-item>
<list-item>
<p>Step 4: For the <italic>r</italic>-th iteration of self-training, initialize the newly labeled dataset <bold>
<italic>X</italic>
</bold>
<sup>
<italic>r</italic>
</sup> &#x3d; &#x2205;. For each datum <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> whose <bold>
<italic>O</italic>
</bold>(<italic>i</italic>) &#x3d; <italic>r</italic>, label <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> according to its predicted label <inline-formula id="inf9">
<mml:math id="m33">
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>, set <inline-formula id="inf10">
<mml:math id="m34">
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x222a;</mml:mo>
<mml:mfenced open="{" close="}">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>Step 5: Determine the label confidence vector <bold>
<italic>&#x3bc;</italic>
</bold>
<sup>
<italic>r</italic>
</sup> for the newly labeled data <bold>
<italic>X</italic>
</bold>
<sup>
<italic>r</italic>
</sup> by the following steps.</p>
<list list-type="simple">
<list-item>
<p>1) For each particle <italic>a</italic>(1 &#x2264; <italic>a</italic> &#x2264; <italic>N</italic>), randomly generate each element of its initial position <inline-formula id="inf11">
<mml:math id="m35">
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> within (0,1]. Set the particle search time <italic>t</italic> &#x3d; 1.</p>
</list-item>
<list-item>
<p>2) For each particle <italic>a</italic>, calculate its fitness value at time <italic>t</italic> according to Eq. <xref ref-type="disp-formula" rid="e15">15</xref>, update its position <inline-formula id="inf12">
<mml:math id="m36">
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> according to Eqs <xref ref-type="disp-formula" rid="e16">16</xref>&#x2013;<xref ref-type="disp-formula" rid="e24">24</xref>.</p>
</list-item>
<list-item>
<p>3) If <italic>t</italic> &#x3c; <italic>T</italic>, then set <italic>t</italic> &#x3d; <italic>t</italic> &#x2b; 1 and return to 2); otherwise, the position of the particle with minimum fitness value is used as the label confidence vector <bold>
<italic>&#x3bc;</italic>
</bold>
<sup>
<italic>r</italic>
</sup> and go to Step 6.</p>
</list-item>
</list>
</list-item>
<list-item>
<p>Step 6: Set <bold>
<italic>X</italic>
</bold>
<sub>
<italic>L</italic>
</sub> &#x3d; <bold>
<italic>X</italic>
</bold>
<sub>
<italic>L</italic>
</sub> &#x222a; <bold>
<italic>X</italic>
</bold>
<sup>
<italic>r</italic>
</sup>, <bold>
<italic>X</italic>
</bold>
<sub>
<italic>U</italic>
</sub> &#x3d; <bold>
<italic>X</italic>
</bold>
<sub>
<italic>U</italic>
</sub> &#x2212; <bold>
<italic>X</italic>
</bold>
<sup>
<italic>r</italic>
</sup>, update <bold>
<italic>&#x3bc;</italic>
</bold>, <bold>
<italic>U</italic>
</bold> and <bold>
<italic>Y</italic>
</bold>. Update the label prediction matrix <bold>
<italic>F</italic>
</bold> and predicted labels of the data according to Eqs <xref ref-type="disp-formula" rid="e10">10</xref>, <xref ref-type="disp-formula" rid="e11">11</xref>. If <bold>
<italic>X</italic>
</bold>
<sub>
<italic>U</italic>
</sub> &#x3d; &#x2205; or the clustering accuracy no longer increases compared with the previous iteration, stop and output the clustering result; otherwise, set <italic>r</italic> &#x3d; <italic>r</italic> &#x2b; 1 and return to Step 4.</p>
</list-item>
</list>
</p>
</statement>
</p>
</sec>
</sec>
<sec id="s3">
<title>3 Experimental results and analysis</title>
<sec id="s3-1">
<title>3.1 Experimental setup</title>
<p>In this paper, comparative experiments are conducted in two benchmark gene expression datasets, as shown in <xref ref-type="table" rid="T1">Table 1</xref>. The Gal dataset (<xref ref-type="bibr" rid="B11">Ideker et al., 2001</xref>) is composed of gene expression measurements for 205 genes involved in galactose use in <italic>Saccharomyces cerevisiae</italic>. The gene expression profiles were measured with four replicate assays across 20 time points and the expression patterns reflect four functional categories. Yeast is a UCI dataset, which aims to predict the localization sites of proteins in cells and contains 1,484 yeast genes with eight methods of predicting protein localization sites in dimensions. Besides, we also demonstrate the applications of the proposed algorithm in other datasets, details of the datasets are tabulated in <xref ref-type="sec" rid="s9">Supplementary Table S1</xref>, and the clustering results can be seen in <xref ref-type="sec" rid="s9">Supplementary Tables S2, S3</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>The description of experimental datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Index</th>
<th align="center">Datesets</th>
<th align="center">Types</th>
<th align="center">Number of genes(<italic>n</italic>)</th>
<th align="center">Number of features(<italic>m</italic>)</th>
<th align="center">Classes(<italic>c</italic>)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">1</td>
<td align="center">Gal</td>
<td align="center">Gene expression</td>
<td align="center">205</td>
<td align="center">80</td>
<td align="center">4</td>
</tr>
<tr>
<td align="left">2</td>
<td align="center">Yeast</td>
<td align="center">Gene expression</td>
<td align="center">1,484</td>
<td align="center">8</td>
<td align="center">10</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>To verify the effectiveness of the SSCAC algorithm proposed in this paper for gene expression data, SSCAC is compared with three unsupervised clustering algorithms and four semi-supervised learning algorithms, including the K-means clustering based on the original gene expression data <bold>
<italic>X</italic>
</bold>, the K-means clustering based on the low-rank representation matrix <bold>
<italic>Z</italic>
</bold> (LRR &#x2b; Kmeans) (<xref ref-type="bibr" rid="B35">Xia et al., 2018</xref>), the NCut clustering based on the LRR similarity matrix <bold>
<italic>W</italic>
</bold> (LRR &#x2b; NCut) (<xref ref-type="bibr" rid="B16">Liu et al., 2013</xref>), SSC-LRR (<xref ref-type="bibr" rid="B35">Xia et al., 2018</xref>), STDP (<xref ref-type="bibr" rid="B34">Wu et al., 2018</xref>), STDPNF (<xref ref-type="bibr" rid="B14">Li et al., 2019</xref>), and LRRADP &#x2b; GFHF (<xref ref-type="bibr" rid="B6">Fei et al., 2017</xref>) algorithms, where SSC-LRR, STDP, and STDPNF are self-training methods. To illustrate the effectiveness of the filter in the STDPNF algorithm, both STDP and STDPNF use KNN as the base classifier. Based on the suggestion of the literature (<xref ref-type="bibr" rid="B35">Xia et al., 2018</xref>), the balance parameter <italic>&#x3bb;</italic> in LRR and SSC-LRR algorithms is tuned within [2<sup>&#x2212;3</sup>, 2<sup>4</sup>], and the parameter value corresponding to the optimal clustering result is selected, so we set <italic>&#x3bb;</italic> &#x3d; 0.1 for all the datasets. In LRRADP &#x2b; GFHF and SSCAC, we set the balance parameters <italic>&#x3bb;</italic>
<sub>1</sub> &#x3d; 100, <italic>&#x3bb;</italic>
<sub>2</sub> &#x3d; 1 and <italic>&#x3bb;</italic>
<sub>
<italic>&#x221e;</italic>
</sub> &#x3d; 1 &#xd7; 10<sup>5</sup>, and the maximum value of penalty parameter <italic>&#x3b2;</italic>
<sub>max</sub> &#x3d; 10<sup>4</sup>, iteration stop parameter <italic>&#x3be;</italic> &#x3d; 10<sup>&#x2212;5</sup>, constant <italic>&#x3c1;</italic> &#x3d; 1.01. And we set the maximum time of the adaptive adjustment of label confidence <italic>T</italic> &#x3d; 100, population size <italic>N</italic> &#x3d; 50, and constant <italic>&#x25b;</italic> &#x3d; 2.2204e &#x2212; 16 in SSCAC, the cut-off distance <italic>d</italic>
<sub>
<italic>c</italic>
</sub> is the corresponding value of data distance sorted in ascending order of 2%, and the other parameters in comparison methods are set as suggested in the original studies. Similar to literature (<xref ref-type="bibr" rid="B24">Nie et al., 2012</xref>; <xref ref-type="bibr" rid="B6">Fei et al., 2017</xref>), the experiments in this paper form the initial labeled dataset <bold>
<italic>X</italic>
</bold>
<sub>
<italic>L</italic>
</sub> by randomly selecting 10% of the data in each dataset, and the rest of the data form the unlabeled dataset <bold>
<italic>X</italic>
</bold>
<sub>
<italic>U</italic>
</sub>. All algorithms are run 10 times with randomly selected initial labeled data, and the algorithm performance is evaluated using the mean value of the results.</p>
</sec>
<sec id="s3-2">
<title>3.2 Evaluation metrics</title>
<p>To assess the partition performance, we use two popular metrics, accuracy (ACC) and Normalized mutual information (NMI).</p>
<p>(1) ACC is calculated by<disp-formula id="e25">
<mml:math id="m37">
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mo movablelimits="false" form="prefix">&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mi>&#x3b4;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(25)</label>
</disp-formula>where <italic>y</italic>
<sub>
<italic>i</italic>
</sub> and <inline-formula id="inf13">
<mml:math id="m38">
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> denote the true label and predicted label of <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub>, respectively, <inline-formula id="inf14">
<mml:math id="m39">
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula> denotes the mapping match between the true label and the predicted label, and <inline-formula id="inf15">
<mml:math id="m40">
<mml:mi>&#x3b4;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:math>
</inline-formula> when <inline-formula id="inf16">
<mml:math id="m41">
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula>, otherwise, it is 0. The closer the value of ACC is to 1, the higher the partition accuracy is.</p>
<p>(2) NMI is calculated by<disp-formula id="e26">
<mml:math id="m42">
<mml:mi>N</mml:mi>
<mml:mi>M</mml:mi>
<mml:mi>I</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>I</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(26)</label>
</disp-formula>where <italic>A</italic> and <italic>B</italic> denote the vectors consisting of the true and predicted labels corresponding to the partition results, respectively. <italic>I</italic>(<italic>A</italic>, <italic>B</italic>) denotes the mutual information measure, <italic>H</italic>(<italic>A</italic>) and <italic>H</italic>(<italic>B</italic>) denote the entropy of <italic>A</italic> and <italic>B</italic>, respectively. The value of <italic>NMI</italic> is between 0 and 1, and a larger value of <italic>NMI</italic> indicates a better partition performance.</p>
</sec>
<sec id="s3-3">
<title>3.3 Comparative results and analysis</title>
<p>
<xref ref-type="table" rid="T2">Table 2</xref> shows the ACC and NMI results of eight algorithms on two benchmark gene expression datasets. The optimal and suboptimal results are marked with bold and italics, respectively.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>ACC and NMI results of each algorithm on two benchmark gene expression datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Datasets</th>
<th align="left">Evaluation metrics</th>
<th align="center">K-means</th>
<th align="left">LRR &#x2b; Kmeans</th>
<th align="center">LRR &#x2b; NCut</th>
<th align="center">SSC&#x2212;LRR</th>
<th align="center">STDP</th>
<th align="center">STDPNF</th>
<th align="left">LRRADP &#x2b; GFHF</th>
<th align="center">SSCAC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="2" align="left">Gal</td>
<td align="center">ACC</td>
<td align="center">0.8517</td>
<td align="center">0.8639</td>
<td align="center">0.8585</td>
<td align="center">0.8912</td>
<td align="center">0.9059</td>
<td align="center">0.9024</td>
<td align="center">
<italic>0.9205</italic>
</td>
<td align="center">
<bold>0.9371</bold>
</td>
</tr>
<tr>
<td align="center">NMI</td>
<td align="center">0.8006</td>
<td align="center">0.8043</td>
<td align="center">0.7493</td>
<td align="center">
<italic>0.8079</italic>
</td>
<td align="center">0.8014</td>
<td align="center">0.7858</td>
<td align="center">0.7562</td>
<td align="center">
<bold>0.8113</bold>
</td>
</tr>
<tr>
<td rowspan="2" align="left">Yeast</td>
<td align="center">ACC</td>
<td align="center">0.3647</td>
<td align="center">0.3726</td>
<td align="center">0.3760</td>
<td align="center">0.3261</td>
<td align="center">0.4816</td>
<td align="center">0.4387</td>
<td align="center">
<italic>0.4926</italic>
</td>
<td align="center">
<bold>0.4987</bold>
</td>
</tr>
<tr>
<td align="center">NMI</td>
<td align="center">0.2652</td>
<td align="center">0.2543</td>
<td align="center">0.1421</td>
<td align="center">0.2024</td>
<td align="center">0.2708</td>
<td align="center">
<italic>0.2781</italic>
</td>
<td align="center">0.2715</td>
<td align="center">
<bold>0.2782</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold and italic values indicates the optimal value and suboptimal values.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>From the results in <xref ref-type="table" rid="T2">Table 2</xref>, it can be seen that.<list list-type="simple">
<list-item>
<p>(1) For the two benchmark gene expression datasets, the clustering results of the SSCAC algorithm proposed in this paper are significantly better than those of the comparison algorithms, indicating the effectiveness of the proposed self-training subspace clustering framework and the adaptive adjustment strategy of label confidence. In addition, the performance of the semi-supervised learning methods outperforms that of the unsupervised clustering algorithms in general, reflecting the advantages of the semi-supervised learning methods.</p>
</list-item>
<list-item>
<p>(2) Among the unsupervised clustering algorithms, LRR &#x2b; Kmeans and LRR &#x2b; NCut perform better overall than the K-means algorithm based on the original gene expression data <bold>
<italic>X</italic>
</bold>. Compared with K-means, LRR &#x2b; Kmeans and LRR &#x2b; NCut improve ACC by an average of 1.80% and 1.95% for two benchmark gene expression datasets. This is because the low-rank representation matrix <bold>
<italic>Z</italic>
</bold> and the similarity matrix <bold>
<italic>W</italic>
</bold> can better reflect the properties of the gene expression data in the low-dimensional subspace, thus more discriminative features can be extracted from the data (<xref ref-type="bibr" rid="B35">Xia et al., 2018</xref>). Compared with LRR, the LRRADP used in the proposed SSCAC algorithm further enhances the locality of the model and can better capture the subspace structure of gene expression data. This advantage of SSCAC will be further demonstrated and analyzed in <xref ref-type="sec" rid="s3-7">Section 3.7</xref>.</p>
</list-item>
<list-item>
<p>(3) Compared with the self-training algorithms SSC-LRR, STDP, and STDPNF, the SSCAC algorithm proposed in this paper has significant advantages. One of the main reasons is that the compared self-training methods implicitly assume that all newly labeled data have the same label confidence. As pointed out in the literature (<xref ref-type="bibr" rid="B22">Mellor et al., 2015</xref>; <xref ref-type="bibr" rid="B35">Xia et al., 2018</xref>; <xref ref-type="bibr" rid="B14">Li et al., 2019</xref>), the problem of mislabeling is inevitable, so setting the same label confidence for both mislabeled and correctly labeled data will lead to continuous reinforcement of incorrect labels during label propagation. Besides, the proposed SSCAC algorithm also outperforms the semi-supervised LRRADP &#x2b; GFHF, the analysis and comparison will be detailed in the following ablation study.</p>
</list-item>
</list>
</p>
<p>In order to verify the convergence of the proposed updating strategy of confidence vector in SSCAC, convergence analysis experiments regarding the number of iterations versus fitness value are done for two benchmark gene expression datasets, Gal and Yeast. As shown in <xref ref-type="fig" rid="F2">Figure 2</xref>, the fitness values flatten out with increasing iteration number and finally converge in approximately 100 iterations. Then, the position of the particle that obtains the minimum fitness value is used as the label confidence vector for the newly labeled data, on which basis SSCAC yields superior clustering results.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Convergence analysis of proposed updating strategy of confidence vector in SSCAC. <bold>(A)</bold> Gal; <bold>(B)</bold> Yeast.</p>
</caption>
<graphic xlink:href="fgene-14-1132370-g002.tif"/>
</fig>
</sec>
<sec id="s3-4">
<title>3.4 Ablation study</title>
<p>In order to validate the effectiveness of label confidence, we also conduct an ablation study. The ablation algorithm is referred to as SSCNAC, i.e., SSCAC without label confidence. In SSCNAC, the same label confidence <italic>&#x3bc;</italic>
<sub>
<italic>i</italic>
</sub> &#x3d; 1 is implicitly set for each newly labeled datum <bold>
<italic>x</italic>
</bold>
<sub>
<italic>i</italic>
</sub> in the self-training process, thus SSCNAC is a self-training subspace clustering algorithm based on original GFHF. The parameter setting of SSCNAC is the same as that of SSCAC, and the performance of SSCNAC and SSCAC in terms of ACC and NMI is reported in <xref ref-type="table" rid="T3">Table 3</xref>. The optimal values of <xref ref-type="table" rid="T3">Table 3</xref> are shown in bold. From <xref ref-type="table" rid="T3">Table 3</xref>, it can be seen that the proposed SSCAC algorithm achieves better clustering performance over SSCNAC. As with other self-training algorithms, SSCNAC performs self-training with complete confidence in the label accuracy of newly labeled data, and therefore suffers from the problem of mislabeling. Comparatively speaking, the proposed SSCAC algorithm introduces label confidences into the semi-supervised clustering objective function and adaptively adjusts them based on semi-supervised learning values, thus can effectively mitigate the negative impact of mislabeled data on self-training learning. This advantage of SSCAC will be further demonstrated in <xref ref-type="table" rid="T4">Table 4</xref>.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Comparison of ACC and NMI of SSCNAC and SSCAC on two benchmark gene expression datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Datasets</th>
<th align="left">Evaluation metrics</th>
<th align="center">SSCNAC</th>
<th align="center">SSCAC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="2" align="left">Gal</td>
<td align="center">ACC</td>
<td align="center">0.9361</td>
<td align="center">
<bold>0.9371</bold>
</td>
</tr>
<tr>
<td align="center">NMI</td>
<td align="center">
<bold>0.8153</bold>
</td>
<td align="center">0.8113</td>
</tr>
<tr>
<td rowspan="2" align="left">Yeast</td>
<td align="center">ACC</td>
<td align="center">0.4966</td>
<td align="center">
<bold>0.4987</bold>
</td>
</tr>
<tr>
<td align="center">NMI</td>
<td align="center">0.2761</td>
<td align="center">
<bold>0.2782</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values indicates the optimal value.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>The newly labeled data selected in the last three iterations of SSCAC on Gal.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Iterations of self-training</th>
<th align="center">Newly labeled data</th>
<th align="center">Real labels</th>
<th align="center">Predicted labels</th>
<th align="center">Label confidences</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">7</td>
<td align="center">
<bold>
<italic>x</italic>
</bold>
<sub>23</sub>
</td>
<td align="center">3</td>
<td align="center">3</td>
<td align="center">0.0090</td>
</tr>
<tr>
<td align="left">7</td>
<td align="center">
<bold>
<italic>x</italic>
</bold>
<sub>65</sub>
</td>
<td align="center">3</td>
<td align="center">3</td>
<td align="center">1.0000</td>
</tr>
<tr>
<td align="left">7</td>
<td align="center">
<bold>
<italic>x</italic>
</bold>
<sub>92</sub>
</td>
<td align="center">2</td>
<td align="center">1</td>
<td align="center">0.1821</td>
</tr>
<tr>
<td align="left">7</td>
<td align="center">
<bold>
<italic>x</italic>
</bold>
<sub>101</sub>
</td>
<td align="center">1</td>
<td align="center">1</td>
<td align="center">0.7275</td>
</tr>
<tr>
<td align="left">7</td>
<td align="center">
<bold>
<italic>x</italic>
</bold>
<sub>108</sub>
</td>
<td align="center">2</td>
<td align="center">1</td>
<td align="center">0.1990</td>
</tr>
<tr>
<td align="left">7</td>
<td align="center">
<bold>
<italic>x</italic>
</bold>
<sub>128</sub>
</td>
<td align="center">4</td>
<td align="center">4</td>
<td align="center">1.0000</td>
</tr>
<tr>
<td align="left">7</td>
<td align="center">
<bold>
<italic>x</italic>
</bold>
<sub>152</sub>
</td>
<td align="center">3</td>
<td align="center">3</td>
<td align="center">1.0000</td>
</tr>
<tr>
<td align="left">7</td>
<td align="center">
<bold>
<italic>x</italic>
</bold>
<sub>165</sub>
</td>
<td align="center">1</td>
<td align="center">1</td>
<td align="center">0.9626</td>
</tr>
<tr>
<td align="left">8</td>
<td align="center">
<bold>
<italic>x</italic>
</bold>
<sub>24</sub>
</td>
<td align="center">1</td>
<td align="center">1</td>
<td align="center">0.8331</td>
</tr>
<tr>
<td align="left">8</td>
<td align="center">
<bold>
<italic>x</italic>
</bold>
<sub>40</sub>
</td>
<td align="center">1</td>
<td align="center">1</td>
<td align="center">0.1622</td>
</tr>
<tr>
<td align="left">8</td>
<td align="center">
<bold>
<italic>x</italic>
</bold>
<sub>99</sub>
</td>
<td align="center">1</td>
<td align="center">1</td>
<td align="center">0.7804</td>
</tr>
<tr>
<td align="left">8</td>
<td align="center">
<bold>
<italic>x</italic>
</bold>
<sub>163</sub>
</td>
<td align="center">1</td>
<td align="center">1</td>
<td align="center">0.8924</td>
</tr>
<tr>
<td align="left">9</td>
<td align="center">
<bold>
<italic>x</italic>
</bold>
<sub>30</sub>
</td>
<td align="center">3</td>
<td align="center">3</td>
<td align="center">0.6928</td>
</tr>
<tr>
<td align="left">9</td>
<td align="center">
<bold>
<italic>x</italic>
</bold>
<sub>183</sub>
</td>
<td align="center">1</td>
<td align="center">1</td>
<td align="center">0.8974</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Moreover, from <xref ref-type="table" rid="T2">Tables 2</xref>, <xref ref-type="table" rid="T3">3</xref>, we can also observe that the clustering results of SSCNAC outperform those of LRRADP &#x2b; GFHF, with an average improvement of 1.25% and 4.75% in ACC and NMI, respectively. In essence, the SSCNAC algorithm with fixed-label confidence is a direct extension of LRRADP &#x2b; GFHF on self-training, which gives SSCNAC the ability to learn from unlabeled data in self-training framework and therefore has better generalization performance. The above results demonstrate the positive role of unlabeled data in self-training learning and the effectiveness of the proposed self-training subspace clustering framework based on GFHF for gene expression data.</p>
</sec>
<sec id="s3-5">
<title>3.5 Analysis of hyper-parameters</title>
<p>In the proposed SSCAC algorithm, <italic>&#x3bb;</italic>
<sub>1</sub> and <italic>&#x3bb;</italic>
<sub>2</sub> are balance parameters to trade off among the low-rank representation, noise and adaptive distance penalty. <xref ref-type="fig" rid="F3">Figure 3</xref> shows the impact of the two hyper-parameters on the performace of SSCAC. As can be observed, the proposed SSCAC algorithm is comparatively unaffected by hyper-parameters that are close to the ideal. To be more precise, we advise setting <italic>&#x3bb;</italic>
<sub>1</sub> &#x3d; 100 and <italic>&#x3bb;</italic>
<sub>2</sub> &#x3d; 1.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Impact analysis of the hyper-parameters on the performance of SSCAC. <bold>(A)</bold> Gal; <bold>(B)</bold> Yeast.</p>
</caption>
<graphic xlink:href="fgene-14-1132370-g003.tif"/>
</fig>
</sec>
<sec id="s3-6">
<title>3.6 Analysis of the impact of initially labeled data ratio</title>
<p>In order to analyze the impact of initially labeled data size on algorithm performance, we increase the initially labeled data ratio from 10% to 90% and conducted experiments, all algorithms are run 10 times. The average ACC curves of semi-supervised SSC-LRR, STDP, STDPNF, LRRADP &#x2b; GFHF, SSCNAC and SSCAC algorithms are given in <xref ref-type="fig" rid="F4">Figure 4</xref>.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>ACC of each algorithm with different ratio of initially labeled data. <bold>(A)</bold> Gal; <bold>(B)</bold> Yeast.</p>
</caption>
<graphic xlink:href="fgene-14-1132370-g004.tif"/>
</fig>
<p>It can be seen from <xref ref-type="fig" rid="F4">Figure 4</xref> that, in general, the partition accuracy of each algorithm increases along with the size of initially labeled data, the reason is that the increase in available label information helps to obtain models that fit the data more closely. It can also be seen from <xref ref-type="fig" rid="F4">Figure 4</xref> that in all cases, the ACC values of the proposed SSCAC algorithm are higher than those of the comparison algorithms, and this advantage of SSCAC is more evident in the cases when the initially labeled data ratios are relatively low. This is because that in such cases, the newly labeled data occupies a larger proportion of the labeled dataset and therefore plays a dominant role in the self-training process. Thus, the adaptive adjustment strategy of label confidence of SSCAC can reduce the influence of mislabeled data to a greater extent. As the proportion of initially labeled data increases, the dominant role of the newly labeled data in the self-training process decreases, and the semi-supervised learning performance of each algorithm tends to be similar. The above results and analysis indicate that the SSCAC algorithm proposed in this paper is more suitable for solving the semi-supervised clustering problem with less initially labeled data.</p>
</sec>
<sec id="s3-7">
<title>3.7 Analysis of the contribution of each part of the proposed SSCAC model</title>
<p>In this section, we discuss the contribution of each part of the proposed model. The SSCAC model described by Eq. <xref ref-type="disp-formula" rid="e9">9</xref> consists of two parts: <inline-formula id="inf17">
<mml:math id="m43">
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mi mathvariant="bold-italic">L</mml:mi>
<mml:mi mathvariant="bold-italic">F</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula> and tr(<bold>
<italic>F</italic>
</bold> &#x2212;<bold>
<italic>&#x3bc;Y</italic>
</bold>)<sup>
<italic>T</italic>
</sup>
<bold>
<italic>U</italic>
</bold>(<bold>
<italic>F</italic>
</bold> &#x2212; <bold>
<italic>&#x3bc;Y</italic>
</bold>), which together make the model have high clustering accuracy. <inline-formula id="inf18">
<mml:math id="m44">
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mi mathvariant="bold-italic">L</mml:mi>
<mml:mi mathvariant="bold-italic">F</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula> is the manifold smoothness term of the objective function, the LRRADP low-rank representation matrix <bold>
<italic>Z</italic>
</bold> adopted in SSCAC can effectively enhance the sparsity of the similarity matrix <bold>
<italic>W</italic>
</bold> and improve the discriminative property of gene expression data, which can then improve the clustering accuracy through the graph Laplacian matrix <bold>
<italic>L</italic>
</bold>. To illustrate the advantage of the LRRADP low-rank representation, visualization of the original data matrix and the low-rank representation matrixs of LRR and LRRADP are plotted on the Gal dataset, as shown in <xref ref-type="fig" rid="F5">Figure 5</xref>, and the data in each subplot are sorted according to their cluster labels in an ascending order. As seen from <xref ref-type="fig" rid="F5">Figure 5</xref>, the low-rank representation matrix <bold>
<italic>Z</italic>
</bold> in both <xref ref-type="fig" rid="F5">Figures 5B, C</xref> has a block-diagonal structure, i.e., the four high pixel rectangles along the diagonal of <bold>
<italic>Z</italic>
</bold> correspond to the four Gal gene clusters, respectively. It is obvious that compared with the original data matrix <bold>
<italic>X</italic>
</bold>, the low-rank representation matrix <bold>
<italic>Z</italic>
</bold> can better reveal the subspace structure of gene expression data, i.e., the block-diagonal structure. Comparing <xref ref-type="fig" rid="F5">Figures 5B, C</xref>, it can be seen that since LRRADP considers the locality of gene expression data while focusing on the global low-rank constraint, the resulting low-rank representation matrix <bold>
<italic>Z</italic>
</bold> is more sparse and the diagonal-block structure is more obvious, and thus can provide more discriminative information for SSCAC.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Visualization of original gene expression data and low-rank representation matrices on Gal. <bold>(A)</bold> Image of original data matrix <bold>
<italic>X</italic>
</bold>; <bold>(B)</bold> Image of LRR low-rank representation matrix <bold>
<italic>Z</italic>
</bold>; <bold>(C)</bold> Image of LRRADP low-rank representation matrix <bold>
<italic>Z</italic>
</bold>.</p>
</caption>
<graphic xlink:href="fgene-14-1132370-g005.tif"/>
</fig>
<p>On the other hand, the second term of the SSCAC model, tr(<bold>
<italic>F</italic>
</bold> &#x2212; <bold>
<italic>&#x3bc;Y</italic>
</bold>)<sup>
<italic>T</italic>
</sup>
<bold>
<italic>U</italic>
</bold>(<bold>
<italic>F</italic>
</bold> &#x2212; <bold>
<italic>&#x3bc;Y</italic>
</bold>), incorporates the label confidence <bold>
<italic>&#x3bc;</italic>
</bold> with the aim to reduce the label confidences of mislabeled data through the proposed adaptive adjustment strategy of label confidence, and mitigate their negative impact during the self-training iterations. In order to verify the effectiveness of the adaptive adjustment strategy of label confidence in the SSCAC model, we focus on the newly labeled data, as well as their real labels, predicted labels, and label confidences during the self-training process. In our experiments, all algorithms are run 10 times with randomly selected initial labeled data. Thus, the newly labeled data selected during the iteration of SSCAC are different for different initial labeled data. Here, we take one case of random selection of initial labeled data on GAL as an example, where SSCAC achieves convergence in nine iterations. The newly labeled data selected in the last three iterations and their label confidences are reported in <xref ref-type="table" rid="T4">Table 4</xref>, and similar results can be obtained for other iterations.</p>
<p>As seen from <xref ref-type="table" rid="T4">Table 4</xref>, the adaptive adjustment strategy proposed in this paper can effectively reduce the label confidences of the mislabeled data, such as <bold>
<italic>x</italic>
</bold>
<sub>92</sub> and <bold>
<italic>x</italic>
</bold>
<sub>108</sub> in the 7-th iteration, and assign large confidences to the correctly labeled data. From <xref ref-type="table" rid="T4">Table 4</xref>, we can also observe that the label confidence of the correctly labeled datum <bold>
<italic>x</italic>
</bold>
<sub>23</sub> is rather small. As pointed out in the literature (<xref ref-type="bibr" rid="B3">Chen et al., 2011</xref>), even though some datum has correct label, it may have less impact on supervised learning due to its low partition uncertainty. Therefore, it is reasonable to assign a lower confidence to such correctly labeled datum. Compared with the existing self-training methods that do not consider the label confidence of newly labeled data, SSCAC can adaptively adjust the strength of supervisory guidance for different newly labeled data in the self-training process and effectively mitigate the negative impact of mislabeled data, which helps to significantly improve the clustering accuracy on gene expression data.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s4">
<title>4 Conclusion</title>
<p>To deal with the widely existing problem of mislabeling in self-training learning tasks, a novel self-training subspace clustering algorithm for gene clustering is proposed in this paper. In particular, label confidences are integrated into the self-training clustering model, and the corresponding determination strategy of label confidences is proposed to adaptively adjust the supervision strength of newly labeled data according to their semi-supervised learning values. Moreover, the low-rank representation with distance penalty is adopted to improve discriminative property of gene expression data. Compared with other state-of-the-art unsupervised and semi-supervised learning algorithms, the proposed SSCAC algorithm can effectively mitigate the negative impact of mislabeling and improve the stability and accuracy of gene clustering. In our future work, we will consider biological knowledge such as Gene Ontology annotation information, and extend the proposed model to multi-view clustering framework to further improve clustering performance on gene expression data.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://archive.ics.uci.edu/ml/datasets/Yeast">https://archive.ics.uci.edu/ml/datasets/Yeast</ext-link>, <ext-link ext-link-type="uri" xlink:href="http://genomebiology.com/2003/4/5/R34">http://genomebiology.com/2003/4/5/R34</ext-link>.</p>
</sec>
<sec id="s6">
<title>Author contributions</title>
<p>DL: conceptualization, methodology, software. HL: data curation, writing- original draft preparation. PQ: visualization, investigation. JW: writing- reviewing and editing.</p>
</sec>
<sec sec-type="COI-statement" id="s7">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s8">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s9">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2023.1132370/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2023.1132370/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.PDF" id="SM1" mimetype="application/PDF" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Basri</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Jacobs</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Lambertian reflectance and linear subspaces</article-title>. <source>IEEE Trans. Pattern Analysis Mach. Intell.</source> <volume>25</volume>, <fpage>218</fpage>&#x2013;<lpage>233</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2003.1177153</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chapelle</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Scholkopf</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zien</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2006</year>). <source>Semi-supervised learning</source>. <publisher-loc>Cambridge, Massachusettes</publisher-loc>: <publisher-name>The MIT Press View Article</publisher-name>.</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Multi-class image classification based on active learning and semi-supervised learning</article-title>. <source>Acta Autom. Sin.</source> <volume>37</volume>, <fpage>954</fpage>&#x2013;<lpage>962</lpage>.</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Qu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Weighted co-expression network analysis identifies rnf181 as a causal gene of coronary artery disease</article-title>. <source>Front. Genet.</source> <volume>12</volume>, <fpage>818813</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2021.818813</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Diniz</surname>
<given-names>W. J. S.</given-names>
</name>
<name>
<surname>Mazzoni</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Coutinho</surname>
<given-names>L. L.</given-names>
</name>
<name>
<surname>Banerjee</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Geistlinger</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Cesar</surname>
<given-names>A. S. M.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Detection of co-expressed pathway modules associated with mineral concentration and meat quality in nelore cattle</article-title>. <source>Front. Genet.</source> <volume>10</volume>, <fpage>210</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2019.00210</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fei</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Low rank representation with adaptive distance penalty for semi-supervised subspace classification</article-title>. <source>Pattern Recognit.</source> <volume>67</volume>, <fpage>252</fpage>&#x2013;<lpage>262</lpage>. <pub-id pub-id-type="doi">10.1016/j.patcog.2017.02.017</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Feng</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Davila</surname>
<given-names>J. I.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Semi-supervised topological analysis for elucidating hidden structures in high-dimensional transcriptome datasets</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinforma.</source> <volume>18</volume>, <fpage>1620</fpage>&#x2013;<lpage>1631</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2019.2950657</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Sang</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Tong</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Dan</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Using clustering analysis to improve semi-supervised classification</article-title>. <source>Neurocomputing</source> <volume>101</volume>, <fpage>290</fpage>&#x2013;<lpage>298</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2012.08.020</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Gene classification using parameter-free semi-supervised manifold learning</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinforma.</source> <volume>9</volume>, <fpage>818</fpage>&#x2013;<lpage>827</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2011.152</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A multiview clustering method with low-rank and sparsity constraints for cancer subtyping</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinforma.</source> <volume>19</volume>, <fpage>1</fpage>&#x2013;<lpage>3223</lpage>. <pub-id pub-id-type="doi">10.1109/tcbb.2021.3122917</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ideker</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Thorsson</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Ranish</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Christmas</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Buhler</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Eng</surname>
<given-names>J. K.</given-names>
</name>
<etal/>
</person-group> (<year>2001</year>). <article-title>Integrated genomic and proteomic analyses of a systematically perturbed metabolic network</article-title>. <source>Science</source> <volume>292</volume>, <fpage>929</fpage>&#x2013;<lpage>934</lpage>. <pub-id pub-id-type="doi">10.1126/science.292.5518.929</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kumar</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Edukondalu</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Strategic bidding using fuzzy adaptive gravitational search algorithm in a pool based electricity market</article-title>. <source>Appl. Soft Comput.</source> <volume>13</volume>, <fpage>2445</fpage>&#x2013;<lpage>2455</lpage>. <pub-id pub-id-type="doi">10.1016/j.asoc.2012.12.003</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A boosting self-training framework based on instance generation with natural neighbors for k nearest neighbor</article-title>. <source>Appl. Intell.</source> <volume>50</volume>, <fpage>3535</fpage>&#x2013;<lpage>3553</lpage>. <pub-id pub-id-type="doi">10.1007/s10489-020-01732-1</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A self-training method based on density peaks and an extended parameter-free local noise filter for k nearest neighbor</article-title>. <source>Knowledge-Based Syst.</source> <volume>184</volume>, <fpage>104895</fpage>. <pub-id pub-id-type="doi">10.1016/j.knosys.2019.104895</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Underlying mechanisms and candidate drugs for Covid-19 based on the connectivity map database</article-title>. <source>Front. Genet.</source> <volume>11</volume>, <fpage>558557</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2020.558557</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Robust recovery of subspace structures by low-rank representation</article-title>. <source>IEEE Trans. Pattern Analysis Mach. Intell.</source> <volume>35</volume>, <fpage>171</fpage>&#x2013;<lpage>184</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2012.88</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Identifying protein complexes with clear module structure using pairwise constraints in protein interaction networks</article-title>. <source>Front. Genet.</source> <volume>12</volume>, <fpage>664786</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2021.664786</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Verbi&#x10d;</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A new dynamic security assessment framework based on semi-supervised learning and data editing</article-title>. <source>Electr. Power Syst. Res.</source> <volume>172</volume>, <fpage>221</fpage>&#x2013;<lpage>229</lpage>. <pub-id pub-id-type="doi">10.1016/j.epsr.2019.03.009</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Kong</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Non-negative symmetric low-rank representation graph regularized method for cancer clustering based on score function</article-title>. <source>Front. Genet.</source> <volume>10</volume>, <fpage>1353</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2019.01353</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma&#xe2;touk</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Ayadi</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Bouziri</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Duval</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Evolutionary biclustering algorithms: An experimental study on microarray data</article-title>. <source>Soft Comput.</source> <volume>23</volume>, <fpage>7671</fpage>&#x2013;<lpage>7697</lpage>. <pub-id pub-id-type="doi">10.1007/s00500-018-3394-4</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mahendran</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Durai Raj Vincent</surname>
<given-names>P. M.</given-names>
</name>
<name>
<surname>Srinivasan</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>C.-Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Machine learning based computational gene selection models: A survey, performance evaluation, open issues, and future research directions</article-title>. <source>Front. Genet.</source> <volume>11</volume>, <fpage>603808</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2020.603808</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mellor</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Boukir</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Haywood</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Exploring issues of training data imbalance and mislabelling on random forest performance for large area land cover classification using the ensemble margin</article-title>. <source>ISPRS J. Photogrammetry Remote Sens.</source> <volume>105</volume>, <fpage>155</fpage>&#x2013;<lpage>168</lpage>. <pub-id pub-id-type="doi">10.1016/j.isprsjprs.2015.03.014</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mirjalili</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hashim</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sardroudi</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Training feedforward neural networks using hybrid particle swarm optimization and gravitational search algorithm</article-title>. <source>Appl. Math. Comput.</source> <volume>218</volume>, <fpage>11125</fpage>&#x2013;<lpage>11137</lpage>. <pub-id pub-id-type="doi">10.1016/j.amc.2012.04.069</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nie</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Initialization independent clustering with actively self-training method</article-title>. <source>IEEE Trans. Syst. Man, Cybern. Part B</source> <volume>42</volume>, <fpage>17</fpage>&#x2013;<lpage>27</lpage>. <pub-id pub-id-type="doi">10.1109/TSMCB.2011.2161607</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nisar</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Paracha</surname>
<given-names>R. Z.</given-names>
</name>
<name>
<surname>Arshad</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Adil</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zeb</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hanif</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Integrated analysis of microarray and rna-seq data for the identification of hub genes and networks involved in the pancreatic cancer</article-title>. <source>Front. Genet.</source> <volume>12</volume>, <fpage>663787</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2021.663787</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Semi-supervised self-training for aspect extraction</article-title>. <source>CAAI Trans. Intelligent Syst.</source> <volume>14</volume>, <fpage>635</fpage>&#x2013;<lpage>641</lpage>.</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rashedi</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Nezamabadi-pour</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Saryazdi</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Gsa: A gravitational search algorithm</article-title>. <source>Inf. Sci.</source> <volume>179</volume>, <fpage>2232</fpage>&#x2013;<lpage>2248</lpage>. <pub-id pub-id-type="doi">10.1016/j.ins.2009.03.004</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rodriguez</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Laio</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Machine learning. Clustering by fast search and find of density peaks</article-title>. <source>Science</source> <volume>344</volume>, <fpage>1492</fpage>&#x2013;<lpage>1496</lpage>. <pub-id pub-id-type="doi">10.1126/science.1242072</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Multi-view subspace clustering analysis for aggregating multiple heterogeneous omics data</article-title>. <source>Front. Genet.</source> <volume>10</volume>, <fpage>744</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2019.00744</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Summers</surname>
<given-names>K. M.</given-names>
</name>
<name>
<surname>Bush</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>A. I.</given-names>
</name>
<name>
<surname>Muriuki</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Clark</surname>
<given-names>E. L.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Functional annotation of the transcriptome of the pig, sus scrofa, based upon network analysis of an rnaseq transcriptional atlas</article-title>. <source>Front. Genet.</source> <volume>10</volume>, <fpage>1355</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2019.01355</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ou-Yang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>D.-Q.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Wmlrr: A weighted multi-view low rank representation to identify cancer subtypes from multiple types of omics data</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinforma.</source> <volume>18</volume>, <fpage>2891</fpage>&#x2013;<lpage>2897</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2021.3063284</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kong</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Laplacian regularized low-rank representation for cancer samples clustering</article-title>. <source>Comput. Biol. Chem.</source> <volume>78</volume>, <fpage>504</fpage>&#x2013;<lpage>509</lpage>. <pub-id pub-id-type="doi">10.1016/j.compbiolchem.2018.11.003</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Semi-supervised multi-label image classification based on nearest neighbor editing</article-title>. <source>Neurocomputing</source> <volume>119</volume>, <fpage>462</fpage>&#x2013;<lpage>468</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2013.03.011</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Shang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Self-training semi-supervised classification based on density peaks of data</article-title>. <source>Neurocomputing</source> <volume>275</volume>, <fpage>180</fpage>&#x2013;<lpage>191</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2017.05.072</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xia</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>A self-training subspace clustering algorithm under low-rank representation for cancer classification on gene expression data</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinforma.</source> <volume>15</volume>, <fpage>1315</fpage>&#x2013;<lpage>1324</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2017.2712607</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Fast subspace clustering by learning projective block diagonal representation</article-title>. <source>Pattern Recognit.</source> <volume>135</volume>, <fpage>109152</fpage>. <pub-id pub-id-type="doi">10.1016/j.patcog.2022.109152</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>You</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>H.-S.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Double selection based semi-supervised clustering ensemble for tumor clustering from gene expression profiles</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinforma.</source> <volume>11</volume>, <fpage>727</fpage>&#x2013;<lpage>740</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2014.2315996</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>You</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>H.-S.</given-names>
</name>
<name>
<surname>Leung</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Incremental semi-supervised clustering ensemble for high dimensional data clustering</article-title>. <source>IEEE Trans. Knowl. Data Eng.</source> <volume>28</volume>, <fpage>701</fpage>&#x2013;<lpage>714</lpage>. <pub-id pub-id-type="doi">10.1109/TKDE.2015.2499200</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Graph neural networks and their current applications in bioinformatics</article-title>. <source>Front. Genet.</source> <volume>12</volume>, <fpage>690049</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2021.690049</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>F.-X.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Sinnlrr: A robust subspace clustering method for cell type detection by non-negative and low-rank representation</article-title>. <source>Bioinformatics</source> <volume>35</volume>, <fpage>3642</fpage>&#x2013;<lpage>3650</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz139</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Ghahramani</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Lafferty</surname>
<given-names>J. D.</given-names>
</name>
</person-group> (<year>2003</year>). &#x201c;<article-title>Semi-supervised learning using Gaussian fields and harmonic functions</article-title>,&#x201d; in <conf-name>Proceedings of the 20th International conference on Machine learning</conf-name> (<publisher-loc>Washington, DC, USA</publisher-loc>: <publisher-name>International Conference on Machine Learning</publisher-name>), <fpage>912</fpage>&#x2013;<lpage>919</lpage>.</citation>
</ref>
</ref-list>
</back>
</article>