<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1068075</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2022.1068075</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>LFSC: A linear fast semi-supervised clustering algorithm that integrates reference-bulk and single-cell transcriptomes</article-title>
<alt-title alt-title-type="left-running-head">Liu et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2022.1068075">10.3389/fgene.2022.1068075</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Qiaoming</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2048154/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liang</surname>
<given-names>Yingjian</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1011363/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Dong</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2030554/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Li</surname>
<given-names>Jie</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/771180/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Computer Science and Technology</institution>, <institution>Harbin Institute of Technology</institution>, <addr-line>Harbin</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of General Surgery</institution>, <institution>The First Affiliated Hospital of Harbin Medical University</institution>, <addr-line>Harbin</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Key Laboratory of Hepatosplenic Surgery</institution>, <institution>Ministry of Education</institution>, <institution>The First Affiliated Hospital of Harbin Medical University</institution>, <addr-line>Harbin</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/531759/overview">Quan Zou</ext-link>, University of Electronic Science and Technology of China, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/351544/overview">Ran Su</ext-link>, Tianjin University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1660118/overview">Yijie Ding</ext-link>, University of Electronic Science and Technology of China, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Jie Li, <email>jieli@hit.edu.cn</email>
</corresp>
<fn fn-type="other">
<p>This article was submitted to Computational Genomics, a section of the journal Frontiers in Genetics</p>
</fn>
<fn fn-type="equal" id="fn1">
<label>
<sup>&#x2020;</sup>
</label>
<p>These authors have contributed equally to this work</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>01</day>
<month>12</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>1068075</elocation-id>
<history>
<date date-type="received">
<day>12</day>
<month>10</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>09</day>
<month>11</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Liu, Liang, Wang and Li.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Liu, Liang, Wang and Li</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>The identification of cell types in complex tissues is an important step in research into cellular heterogeneity in disease. We present a linear fast semi-supervised clustering (LFSC) algorithm that utilizes reference samples generated from bulk RNA sequencing data to identify cell types from single-cell transcriptomes. An anchor graph is constructed to depict the relationship between reference samples and cells. By applying a connectivity constraint to the learned graph, LFSC enables the preservation of the underlying cluster structure. Moreover, the overall complexity of LFSC is linear to the size of the data, which greatly improves effectiveness and efficiency. By applying LFSC to real single-cell RNA sequencing datasets, we discovered that it has superior performance over existing baseline methods in clustering accuracy and robustness. An application using infiltrating T cells in liver cancer demonstrates that LFSC can successfully find new cell types, discover differently expressed genes, and explore new cancer-associated biomarkers.</p>
</abstract>
<kwd-group>
<kwd>single-cell RNA-seq</kwd>
<kwd>bulk RNA-seq</kwd>
<kwd>anchor graph</kwd>
<kwd>data integration</kwd>
<kwd>clustering</kwd>
</kwd-group>
<contract-num rid="cn001">62072095</contract-num>
<contract-sponsor id="cn001">National Natural Science Foundation of China<named-content content-type="fundref-id">10.13039/501100001809</named-content>
</contract-sponsor>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Bulk RNA sequencing (RNA-seq) technologies have been widely used to investigate gene expression patterns at the tissue level in recent decades (<xref ref-type="bibr" rid="B6">Conesa et al., 2016</xref>). However, they measure average global gene expression, which obscures true signals between heterogeneous cell types in the tissues. This technical limitation catalyzed the birth of single-cell RNA sequencing technology (scRNA-seq), which investigates RNA biology at the single-cell level. The transcriptome processes of humans and animals are highly heterogeneous; hence, it is more comprehensive and effective to study gene expression patterns using scRNA-seq (<xref ref-type="bibr" rid="B11">Li and Wang, 2021</xref>) in applications such as tumor heterogeneity (<xref ref-type="bibr" rid="B2">Bartoschek et al., 2018</xref>; <xref ref-type="bibr" rid="B25">Zhang et al.,. 2021a</xref>), disease diagnosis (<xref ref-type="bibr" rid="B7">Gate et al., 2020</xref>; <xref ref-type="bibr" rid="B22">Zakharov et al., 2020</xref>), and therapeutic treatment optimization (<xref ref-type="bibr" rid="B24">Zhang et al., 2020</xref>). The application of scRNA-seq data involves cell type identification (<xref ref-type="bibr" rid="B1">Aran et al., 2019</xref>; <xref ref-type="bibr" rid="B21">Wei and Zhang 2021</xref>), selection of differentially expressed genes (<xref ref-type="bibr" rid="B18">Sokolowski et al., 2021</xref>), cell-development trajectory construction (<xref ref-type="bibr" rid="B12">Liu et al., 2022</xref>), and cell&#x2013;cell communication inferencing (<xref ref-type="bibr" rid="B26">Zhang et al., 2021b</xref>).</p>
<p>Among these applications, cell type identification is the most fundamental and essential. Traditional cell type identification methods consist of two steps: clustering cells using unsupervised learning algorithms and labeling cells based on specifically expressed marker genes in each cluster (<xref ref-type="bibr" rid="B3">Butler et al., 2018</xref>; <xref ref-type="bibr" rid="B20">Wang et al., 2018</xref>). While practical, these methods depend heavily on clustering performance and on prior knowledge of marker gene signatures, and they have high time complexity due to the calculated amounts of the cells&#x2019; similarity measurement stage. Based on labeled scRNA-seq data, researchers have proposed two types of cell type identification methods based on supervised learning. The first group of methods involves training a robust classifier on pre-labeled cells and then annotating other unlabeled cells with the trained classifier (<xref ref-type="bibr" rid="B17">Shao et al., 2021</xref>; <xref ref-type="bibr" rid="B8">Heydari et al., 2022</xref>). Another group of methods consists of two steps: embedding unlabeled cells into the subspace of labeled cells and then assigning the unlabeled cells according to the nearest neighbor-labeled cells (<xref ref-type="bibr" rid="B15">Pliner, Shendure, and Trapnell, 2019</xref>; <xref ref-type="bibr" rid="B13">Lotfollahi et al., 2021</xref>). Given the limitations to labeled scRNA-seq datasets, supervised methods cannot be widely used, especially for discovering rare cell populations (<xref ref-type="bibr" rid="B16">Qi et al., 2020</xref>).</p>
<p>Compared with scRNA-seq datasets, many bulk RNA-seq datasets have been archived in recent decades. Hence, some cell type identification methods integrating bulk RNA-seq datasets with known cell types have also been proposed recently. These methods attempt to use information from bulk RNA-seq data to annotate single-cell data. Specifically, they often identify cell types by correlating single-cell transcriptomes with reference datasets of pure cell types sequenced by RNA-seq, then iteratively improve the label inferences. SingleR (<xref ref-type="bibr" rid="B1">Aran et al., 2019</xref>) and RCA (<xref ref-type="bibr" rid="B10">Li et al., 2017</xref>) are the only two known methods of identifying cell types based on reference bulk RNA-seq data. However, it is difficult to detect subtle differences between cells using information from an external reference, since information from one sample in the bulk data comes from one tissue, while the information from the scRNA-seq data comes from one&#xa0;cell (<xref ref-type="bibr" rid="B11">Li and Wang, 2021</xref>).</p>
<p>To address these issues, we present a linear fast semi-supervised clustering (LFSC) algorithm that integrates reference-bulk and single-cell transcriptome data using an anchor graph to improve the effectiveness and efficiency of clustering. The overview of LFSC is shown in <xref ref-type="fig" rid="F1">Figure 1</xref>.<list list-type="simple">
<list-item>
<p>&#x2022; Unlike SingleR and RCA, LFSC generates a dictionary matrix with <italic>m</italic> reference samples from bulk RNA-seq data or labeled scRNA-seq datasets by averaging gene expression profiles in the same cell type. Then, LFSC learns the relationship between cells in scRNA-seq data and the reference samples, generating an anchor graph with <italic>k</italic>-connected components, where <italic>k</italic> denotes the number of clusters.</p>
</list-item>
<list-item>
<p>&#x2022; The advantages of LFSC are that 1) its affinity matrix, based on an anchor graph, preserves the underlying cluster structure of the data, which also reduces memory costs, and 2) its overall complexity is linear to the size of the data, which greatly improves effectiveness and efficiency.</p>
</list-item>
<list-item>
<p>&#x2022; Through benchmark evaluations with 21 real scRNA-seq datasets and application to infiltrating T cells in liver cancer, we demonstrate that LFSC is superior to existing baseline methods.</p>
</list-item>
</list>
</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>orkflow of the LFSC.</p>
</caption>
<graphic xlink:href="fgene-13-1068075-g001.tif"/>
</fig>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Methods and materials</title>
<p>We consider LFSC to be different from supervised learning, in which the goal is to minimize one specific loss function, given the labels of samples. LFSC is also different from unsupervised learning because it is designed under weak supervision: a small set of bulk RNA-seq data, called reference samples, can represent the neighborhood structure of cells in scRNA-seq data. LFSC is regarded as a semi-supervised method since prior knowledge of referenced cell types, generated from bulk RNA-seq data, is combined in the unsupervised clustering process. The details of related studies and the LFSC method are provided in the following paragraphs.</p>
<sec id="s2-1">
<title>2.1 Subspace clustering and anchor graph</title>
<p>Given a set of data <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="script">R</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, where <italic>n</italic> and <italic>d</italic> denote the number of samples and the number of features, respectively, subspace clustering assumes that data samples can be represented by a linear combination of samples underlying the same subspace. This means that <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>X</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, where the linear combination matrix <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="script">R</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> can be modeled as the similarity graph among samples. To find the optimal solution of <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, the estimating process is formulated as<disp-formula id="e1">
<mml:math id="m5">
<mml:mrow>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:munder>
<mml:mi>min</mml:mi>
<mml:mi>S</mml:mi>
</mml:munder>
<mml:msup>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>X</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>S</mml:mi>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>S</mml:mi>
<mml:mn mathvariant="bold">1</mml:mn>
<mml:mo>&#x3d;</mml:mo>
<mml:mn mathvariant="bold">1</mml:mn>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <inline-formula id="inf5">
<mml:math id="m6">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
<mml:mo>&#x3e;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> is a hyperparameter that balances the reconstruction error (first term) and the regularize function (second term <inline-formula id="inf6">
<mml:math id="m7">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mo>&#x2219;</mml:mo>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>). <inline-formula id="inf7">
<mml:math id="m8">
<mml:mrow>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> denotes a column vector with all elements being one. The time complexity in solving <xref ref-type="disp-formula" rid="e1">Eq. 1</xref> is 0 (<italic>n</italic>
<sup>3</sup>), which is costly in terms of running time and storage for large-scale data.</p>
<p>Anchor points, a small set of data samples, were selected to represent the landmarks of the data and preserve the underlying neighbor structure (see <xref ref-type="sec" rid="s10">Supplementary Section S7</xref>). To reduce the computational sources, the anchor graph <inline-formula id="inf8">
<mml:math id="m9">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="script">R</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> between anchor points and other data points was used in subspace clustering (<xref ref-type="bibr" rid="B4">Chen and Deng, 2011</xref>) as the anchor graph <inline-formula id="inf9">
<mml:math id="m10">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is smaller than the similarity graph <inline-formula id="inf10">
<mml:math id="m11">
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. The estimating process is reformulated as<disp-formula id="e2">
<mml:math id="m12">
<mml:mrow>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:munder>
<mml:mi>min</mml:mi>
<mml:mi>A</mml:mi>
</mml:munder>
<mml:msup>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>Z</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>A</mml:mi>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>A</mml:mi>
<mml:mn mathvariant="bold">1</mml:mn>
<mml:mo>&#x3d;</mml:mo>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <inline-formula id="inf11">
<mml:math id="m13">
<mml:mrow>
<mml:mi>Z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the dictionary matrix. Typically, the anchor points are selected by implementing the K-means algorithm on the dataset (<xref ref-type="bibr" rid="B4">Chen and Deng, 2011</xref>), and the centroid points in K-means are updated by calculating the average signals of samples in the same cluster. Hence, the average characteristic of the anchor points is naturally similar to that of the referenced bulk RNA-seq samples, which measure the average expression levels of specific genes in one tissue. We believe the anchor graph to be a potential tool for integrating reference bulk RNA-seq and scRNA-seq data.</p>
</sec>
<sec id="s2-2">
<title>2.2 Data preprocessing</title>
<p>In LFSC, the data preprocessing procedure includes two steps: quality control and normalization. First, data quality control is utilized to filter the low-expressed genes. If a gene has less than 5% or more than 95% of non-zero elements across all cells, it is filtered out. For data normalization, we utilize log-transform normalization, in which each element (<inline-formula id="inf12">
<mml:math id="m14">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) of the expression profile <italic>M</italic> is transformed as follows:<disp-formula id="e3">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>m</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>10000</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
</sec>
<sec id="s2-3">
<title>2.3 Selecting highly variable overlap genes</title>
<p>To reduce redundant features, we first identified the set of genes that were most variable in the expression profile, using the function <italic>FindVariableFeatures</italic> in the package <italic>Seurat</italic> (<xref ref-type="bibr" rid="B3">Butler et al., 2018</xref>). The details of selecting highly variable overlapping genes are provided in <xref ref-type="sec" rid="s10">Supplementary Section S8</xref>. After selecting the highly variable genes, we then selected the genes that overlapped between the remaining genes in scRNA-seq data and the genes in bulk RNA-seq data.</p>
</sec>
<sec id="s2-4">
<title>2.4 Structured anchor graph learning</title>
<p>Constructing an affinity matrix among cells is the key step to identifying cell types in most computational approaches. In LFSC, we integrated reference bulk RNA-seq data and scRNA-seq data into the anchor graph with <italic>k</italic>-connected constraint, which not only improves the clustering performance but also reduces the computational sources. Structured anchor graph learning consists of two steps: generating reference samples with bulk RNA-seq data and constructing a structured anchor graph with reference samples. We used the scRNA-seq data matrix <inline-formula id="inf13">
<mml:math id="m16">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="script">R</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, with <inline-formula id="inf14">
<mml:math id="m17">
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> highly variable genes and <inline-formula id="inf15">
<mml:math id="m18">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> cells, and the bulk RNA-seq data matrix <inline-formula id="inf16">
<mml:math id="m19">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="script">R</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, with <inline-formula id="inf17">
<mml:math id="m20">
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> highly variable genes and <inline-formula id="inf18">
<mml:math id="m21">
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> samples. Details of the structured anchor graph learning are provided in the following paragraphs.</p>
<sec id="s2-4-1">
<title>2.4.1 Generating reference samples from RNA-seq data</title>
<p>In LFSC, we must generate the reference samples which are regarded as anchor points in the anchor graph. Typically, the reference samples are generated from the bulk RNA-seq data. We calculated Pearson&#x2019;s correlation coefficients <inline-formula id="inf19">
<mml:math id="m22">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> between cell <inline-formula id="inf20">
<mml:math id="m23">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and bulk sample <inline-formula id="inf21">
<mml:math id="m24">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. We also generated the remaining sample set <inline-formula id="inf22">
<mml:math id="m25">
<mml:mrow>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:msup>
<mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, the element having the greatest correlation, with at least one&#xa0;cell compared to other bulk samples:<disp-formula id="e4">
<mml:math id="m26">
<mml:mrow>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x7c;</mml:mo>
<mml:munder>
<mml:mrow>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">g</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo>&#x2203;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>With the emergence of the labeled scRNA-seq data, LFSC also provides the option of generating reference samples from the scRNA-seq data. We generated one reference sample for each cell type by measuring the average expression profile for highly variable genes from cells in the same cell type.</p>
</sec>
<sec id="s2-4-2">
<title>2.4.2 Constructing the structured anchor graph with reference samples</title>
<p>Given the scRNA-seq data <inline-formula id="inf23">
<mml:math id="m27">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="script">R</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and reference sample <inline-formula id="inf24">
<mml:math id="m28">
<mml:mrow>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="script">R</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, we utilized the bipartite graph <inline-formula id="inf25">
<mml:math id="m29">
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mi mathvariant="double-struck">A</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msup>
<mml:mi mathvariant="double-struck">A</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
</mml:mtd>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="script">R</mml:mi>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> to represent the anchor graph <inline-formula id="inf26">
<mml:math id="m30">
<mml:mrow>
<mml:mi mathvariant="double-struck">A</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="script">R</mml:mi>
<mml:mrow>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. The normalized Laplacian <inline-formula id="inf27">
<mml:math id="m31">
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is defined as<disp-formula id="e5">
<mml:math id="m32">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>I</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">D</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
<mml:mi>B</mml:mi>
<mml:msup>
<mml:mi mathvariant="double-struck">D</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>where <inline-formula id="inf28">
<mml:math id="m33">
<mml:mrow>
<mml:mi mathvariant="double-struck">D</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="script">R</mml:mi>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>)</mml:mo>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is a diagonal matrix, the <italic>i</italic>th diagonal element of which is calculated as <inline-formula id="inf29">
<mml:math id="m34">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="double-struck">d</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. <xref ref-type="bibr" rid="B5">Chung and Graham (1997)</xref> have demonstrated that the normalized Laplacian <inline-formula id="inf30">
<mml:math id="m35">
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> associated with non-negative matrix <inline-formula id="inf31">
<mml:math id="m36">
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> has the following property.</p>
<p>
<statement content-type="theorem" id="Theorem_1">
<label>Theorem 1</label>
<p>The number of connected components in the bipartite graph <inline-formula id="inf32">
<mml:math id="m37">
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is equal to the multiplicity <italic>k</italic> of the eigenvalue zero of the normalized Laplacian <inline-formula id="inf33">
<mml:math id="m38">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
<p>
<xref ref-type="statement" rid="Theorem_1">Theorem 1</xref> indicates that if <inline-formula id="inf34">
<mml:math id="m39">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, the bipartite graph <inline-formula id="inf35">
<mml:math id="m40">
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> with <inline-formula id="inf36">
<mml:math id="m41">
<mml:mrow>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> reference samples and <inline-formula id="inf37">
<mml:math id="m42">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> cells can be clustered into <italic>k</italic> groups. Motivated by <xref ref-type="statement" rid="Theorem_1">Theorem 1</xref>, we added a constraint to the clustering model, which is formulated as<disp-formula id="e6">
<mml:math id="m43">
<mml:mrow>
<mml:munder>
<mml:mi>min</mml:mi>
<mml:mi>A</mml:mi>
</mml:munder>
<mml:msup>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mi mathvariant="double-struck">A</mml:mi>
</mml:mrow>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mi mathvariant="double-struck">A</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mo>.</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="double-struck">A</mml:mi>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="double-struck">A</mml:mi>
<mml:mn mathvariant="bold">1</mml:mn>
<mml:mo>&#x3d;</mml:mo>
<mml:mn mathvariant="bold">1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
</p>
<p>As the rank constraint is hard to solve, we borrowed the idea from the related literature (<xref ref-type="bibr" rid="B14">Nie et al., 2019</xref>) to relax <xref ref-type="disp-formula" rid="e6">Eq. 6</xref> as<disp-formula id="e7">
<mml:math id="m44">
<mml:mrow>
<mml:munder>
<mml:mi>min</mml:mi>
<mml:mi mathvariant="double-struck">A</mml:mi>
</mml:munder>
<mml:msup>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mi mathvariant="double-struck">A</mml:mi>
</mml:mrow>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mi mathvariant="double-struck">A</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="double-struck">F</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mi>L</mml:mi>
<mml:mi mathvariant="double-struck">F</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mo>.</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="double-struck">A</mml:mi>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="double-struck">A</mml:mi>
<mml:mn mathvariant="bold">1</mml:mn>
<mml:mo>&#x3d;</mml:mo>
<mml:mn mathvariant="bold">1</mml:mn>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">F</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mi mathvariant="double-struck">F</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>I</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>where <inline-formula id="inf38">
<mml:math id="m45">
<mml:mrow>
<mml:mi mathvariant="double-struck">F</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="script">R</mml:mi>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>)</mml:mo>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> In LFSC, the problem (<xref ref-type="disp-formula" rid="e7">Eq. 7</xref>) can be solved by an alternating optimization method; more precisely, we solved <inline-formula id="inf39">
<mml:math id="m46">
<mml:mrow>
<mml:mi mathvariant="double-struck">A</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf40">
<mml:math id="m47">
<mml:mrow>
<mml:mi mathvariant="double-struck">F</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> by fixing one solution and then updating the other one iteratively (see details in <xref ref-type="sec" rid="s10">Supplementary Section S1</xref>).</p>
</statement>
</p>
</sec>
<sec id="s2-4-3">
<title>2.4.3 Estimating the cluster number <italic>k</italic>
</title>
<p>Before the implementation of LFSC, we automatically estimated the cluster number <italic>k</italic> using the <italic>R</italic> package <italic>clustree</italic> (<xref ref-type="bibr" rid="B23">Zappia and Oshlack, 2018</xref>) with the default parameters. The details for estimating the number of clusters are provided in <xref ref-type="sec" rid="s10">Supplementary Section S9</xref>. Finally, the clustering results were achieved by performing the K-means algorithm with the estimated cluster number <italic>k</italic>. The pseudo-code for LFSC is summarized in <xref ref-type="statement" rid="Algorithm_1">Algorithm 1</xref>.</p>
</sec>
<sec id="s2-4-4">
<title>2.4.4 Identifying cell types <italic>via</italic> labeled transcriptomics data</title>
<p>In LFSC, we annotated cell types in scRNA-seq by calculating the Pearson coefficient between cell types from labeled transcriptomics data and cluster annotations of unknown scRNA-seq data. Based on the overlapping HVGs, the cell type of scRNA-seq data is annotated as<disp-formula id="equ1">
<mml:math id="m48">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>s</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>.</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>;</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="e8">
<mml:math id="m49">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x7c;</mml:mo>
<mml:munder>
<mml:mrow>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">g</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3e;</mml:mo>
<mml:mn>0.6</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2203;</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>U</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>k</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>w</mml:mi>
<mml:mi>n</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>0.6</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2200;</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mo>;</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>
<disp-formula id="equ2">
<mml:math id="m50">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf41">
<mml:math id="m51">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo>&#x22ef;</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>K</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is denoted as the <inline-formula id="inf42">
<mml:math id="m52">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> clustering results, and <inline-formula id="inf43">
<mml:math id="m53">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf44">
<mml:math id="m54">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the clustering index and the reference cell of cluster <italic>i</italic>, respectively. We defined <inline-formula id="inf45">
<mml:math id="m55">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> as the cell type of the cluster or reference sample <inline-formula id="inf46">
<mml:math id="m56">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. The correlation analysis of overlapping variable genes between reference transcriptomics data and unlabeled scRNA-seq data was implemented to distinguish closely related cell types. The pseudo-code for LFSC is summarized in <xref ref-type="statement" rid="Algorithm_1">Algorithm 1</xref>.</p>
</sec>
</sec>
<sec id="s2-5">
<title>2.5 Complexity analysis</title>
<p>In LFSC, we utilized an anchor graph to integrate the reference samples from bulk RNA-seq data and unlabeled cells from the scRNA-seq data, so that the complexity would reduce significantly. More precisely, we defined the number of iterations as <inline-formula id="inf47">
<mml:math id="m57">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. In the alternating optimization method, we applied SVD on <inline-formula id="inf48">
<mml:math id="m58">
<mml:mrow>
<mml:mi mathvariant="double-struck">A</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x226a;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> to calculate the matrices <inline-formula id="inf49">
<mml:math id="m59">
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> (taking <inline-formula id="inf50">
<mml:math id="m60">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mo>(</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>(</mml:mo>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mo>&#x2033;</mml:mo>
</mml:msup>
</mml:mrow>
<mml:mn>3</mml:mn>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mi>n</mml:mi>
<mml:mo>)</mml:mo>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>) and <inline-formula id="inf51">
<mml:math id="m61">
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> (taking <inline-formula id="inf52">
<mml:math id="m62">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>). Using <xref ref-type="sec" rid="s10">Supplementary Eq. 4</xref>, the problem can be efficiently solved in parallel using the MATLAB function <italic>quadprog</italic>, costing <inline-formula id="inf53">
<mml:math id="m63">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
<mml:mn>3</mml:mn>
</mml:msup>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. In addition, it costs <inline-formula id="inf54">
<mml:math id="m64">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>t</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> in applying K-means on <inline-formula id="inf55">
<mml:math id="m65">
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to obtain the clustering results, where <inline-formula id="inf56">
<mml:math id="m66">
<mml:mrow>
<mml:msup>
<mml:mi>t</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> denotes the number of iterations in K-means. Hence, the overall time complexity of the LFSC is linear to the number of cells <inline-formula id="inf57">
<mml:math id="m67">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. For space complexity, in addition to commonly used sources like storing the scRNA-seq data <inline-formula id="inf58">
<mml:math id="m68">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and bulk RNA-seq data <inline-formula id="inf59">
<mml:math id="m69">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, we need the storage sources <inline-formula id="inf60">
<mml:math id="m70">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> for <inline-formula id="inf61">
<mml:math id="m71">
<mml:mrow>
<mml:mi mathvariant="double-struck">A</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, while that of the original graph <inline-formula id="inf62">
<mml:math id="m72">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is <inline-formula id="inf63">
<mml:math id="m73">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>n</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. In the alternating optimization method, the matrices <inline-formula id="inf64">
<mml:math id="m74">
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf65">
<mml:math id="m75">
<mml:mrow>
<mml:mi mathvariant="double-struck">D</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> are stored as the sparse matrix, given their specific structures, while the space complexities of <inline-formula id="inf66">
<mml:math id="m76">
<mml:mrow>
<mml:mi>U</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf67">
<mml:math id="m77">
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> are <inline-formula id="inf68">
<mml:math id="m78">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf69">
<mml:math id="m79">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Hence, the complete space complexity of LFSC also reduces significantly.</p>
</sec>
<sec id="s2-6">
<title>2.6 Evaluation metrics, test datasets, and baseline methods</title>
<p>We used the adjusted Rand index (ARI), accuracy (ACC), normalized mutual information (NMI), purity, and silhouette coefficient as our evaluation metrics (see details in <xref ref-type="sec" rid="s10">Supplementary Section S2</xref>). We downloaded 21 public scRNA-seq datasets generated by four sequencing protocols (see details in <xref ref-type="table" rid="T1">Table 1</xref>) as the test datasets. We also selected six state-of-the-art methods (see details in <xref ref-type="table" rid="T2">Table 2</xref> and <xref ref-type="sec" rid="s10">Supplementary Section S3</xref>) as the compared baseline methods. In addition, we analyzed infiltrating T cells in liver cancer to examine LFSC&#x2019;s application value in finding new cell types, discovering differently expressed genes, and exploring new cancer-associated biomarkers.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Summary of the 21 real single-cell RNA-seq datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Dataset</th>
<th align="left">Cells</th>
<th align="left">Genes</th>
<th align="left">Types</th>
<th align="left">Protocol</th>
<th align="left">Resource</th>
<th align="left">Usage</th>
<th align="left">Confidence</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Treutlin</td>
<td align="char" char=".">80</td>
<td align="char" char=".">23271</td>
<td align="char" char=".">5</td>
<td align="left">SMARTer</td>
<td align="left">Human lung epithelium</td>
<td align="left">Clustering</td>
<td align="left">Silver standard</td>
</tr>
<tr>
<td align="left">Yan</td>
<td align="char" char=".">90</td>
<td align="char" char=".">20214</td>
<td align="char" char=".">7</td>
<td align="left">Tang</td>
<td align="left">Human preimplantation</td>
<td align="left">Clustering and parameter analysis; visualization</td>
<td align="left">Gold standard</td>
</tr>
<tr>
<td align="left">Ting</td>
<td align="char" char=".">114</td>
<td align="char" char=".">14405</td>
<td align="char" char=".">5</td>
<td align="left">Drop-seq</td>
<td align="left">Human circulating tumor</td>
<td align="left">Clustering</td>
<td align="left">Silver standard</td>
</tr>
<tr>
<td align="left">mECS</td>
<td align="char" char=".">182</td>
<td align="char" char=".">8989</td>
<td align="char" char=".">3</td>
<td align="left">HiSeq</td>
<td align="left">Mouse embryonic stem cells</td>
<td align="left">Clustering</td>
<td align="left">Silver standard</td>
</tr>
<tr>
<td align="left">Buettner</td>
<td align="char" char=".">189</td>
<td align="char" char=".">8989</td>
<td align="char" char=".">3</td>
<td align="left">Drop-seq</td>
<td align="left">Mouse T cells</td>
<td align="left">Clustering</td>
<td align="left">Silver standard</td>
</tr>
<tr>
<td align="left">Goolam</td>
<td align="char" char=".">214</td>
<td align="char" char=".">41480</td>
<td align="char" char=".">5</td>
<td align="left">Smart-seq</td>
<td align="left">Mouse embryonic cells</td>
<td align="left">Clustering and parameter analysis; visualization</td>
<td align="left">Gold standard</td>
</tr>
<tr>
<td align="left">Ginhoux</td>
<td align="char" char=".">251</td>
<td align="char" char=".">11834</td>
<td align="char" char=".">3</td>
<td align="left">Smart-seq</td>
<td align="left">Mouse conventional dendritic cells</td>
<td align="left">Clustering</td>
<td align="left">Silver standard</td>
</tr>
<tr>
<td align="left">Deng</td>
<td align="char" char=".">268</td>
<td align="char" char=".">22431</td>
<td align="char" char=".">7</td>
<td align="left">Smart-seq</td>
<td align="left">Mouse embryo cell</td>
<td align="left">Clustering and parameter analysis; visualization</td>
<td align="left">Gold standard</td>
</tr>
<tr>
<td align="left">Pollen</td>
<td align="char" char=".">301</td>
<td align="char" char=".">23730</td>
<td align="char" char=".">11</td>
<td align="left">Smart-seq</td>
<td align="left">Human cerebral cortex</td>
<td align="left">Clustering and parameter analysis; visualization</td>
<td align="left">Gold standard</td>
</tr>
<tr>
<td align="left">Patel</td>
<td align="char" char=".">430</td>
<td align="char" char=".">5848</td>
<td align="char" char=".">5</td>
<td align="left">Smart-seq</td>
<td align="left">Human glioblastomas</td>
<td align="left">Clustering</td>
<td align="left">Silver standard</td>
</tr>
<tr>
<td align="left">Usoskin</td>
<td align="char" char=".">622</td>
<td align="char" char=".">17772</td>
<td align="char" char=".">11</td>
<td align="left">Drop-seq</td>
<td align="left">Mouse lumbar cells</td>
<td align="left">Clustering</td>
<td align="left">Silver standard</td>
</tr>
<tr>
<td align="left">Kolod</td>
<td align="char" char=".">704</td>
<td align="char" char=".">13473</td>
<td align="char" char=".">3</td>
<td align="left">SMARTer</td>
<td align="left">Embryonic stem cells</td>
<td align="left">Clustering</td>
<td align="left">Silver standard</td>
</tr>
<tr>
<td align="left">Seger</td>
<td align="char" char=".">1099</td>
<td align="char" char=".">25525</td>
<td align="char" char=".">9</td>
<td align="left">Smart-seq</td>
<td align="left">Pancreatic islet</td>
<td align="left">Clustering</td>
<td align="left">Silver standard</td>
</tr>
<tr>
<td align="left">Tasic</td>
<td align="char" char=".">1679</td>
<td align="char" char=".">24150</td>
<td align="char" char=".">49</td>
<td align="left">SMARTer</td>
<td align="left">Mouse cortical cells</td>
<td align="left">Clustering</td>
<td align="left">Silver standard</td>
</tr>
<tr>
<td align="left">Grun</td>
<td align="char" char=".">1915</td>
<td align="char" char=".">23536</td>
<td align="char" char=".">3</td>
<td align="left">CEL-seq</td>
<td align="left">Hematopoietic stem cells</td>
<td align="left">Clustering</td>
<td align="left">Silver standard</td>
</tr>
<tr>
<td align="left">Baron</td>
<td align="char" char=".">1937</td>
<td align="char" char=".">20125</td>
<td align="char" char=".">14</td>
<td align="left">InDrop</td>
<td align="left">Pancreatic islet</td>
<td align="left">Clustering</td>
<td align="left">Silver standard</td>
</tr>
<tr>
<td align="left">Zeisel</td>
<td align="char" char=".">3005</td>
<td align="char" char=".">19972</td>
<td align="char" char=".">47</td>
<td align="left">STRT-seq</td>
<td align="left">Mouse cortex cells</td>
<td align="left">Clustering</td>
<td align="left">Silver standard</td>
</tr>
<tr>
<td align="left">Marques</td>
<td align="char" char=".">5053</td>
<td align="char" char=".">23556</td>
<td align="char" char=".">13</td>
<td align="left">C1</td>
<td align="left">Mouse neuronal cells</td>
<td align="left">Clustering</td>
<td align="left">Silver standard</td>
</tr>
<tr>
<td align="left">Macosko</td>
<td align="char" char=".">6418</td>
<td align="char" char=".">23288</td>
<td align="char" char=".">39</td>
<td align="left">Drop-seq</td>
<td align="left">Mouse retina cells</td>
<td align="left">Clustering and running time analysis</td>
<td align="left">Silver standard</td>
</tr>
<tr>
<td align="left">Chen</td>
<td align="char" char=".">14437</td>
<td align="char" char=".">23284</td>
<td align="char" char=".">45</td>
<td align="left">Drop-seq</td>
<td align="left">Hypothalamic cells</td>
<td align="left">Clustering and running time analysis</td>
<td align="left">Silver standard</td>
</tr>
<tr>
<td align="left">Campbell</td>
<td align="char" char=".">21086</td>
<td align="char" char=".">26774</td>
<td align="char" char=".">35</td>
<td align="left">Drop-seq</td>
<td align="left">Hypothalamic cells</td>
<td align="left">Clustering and running time analysis</td>
<td align="left">Silver standard</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Summary of the compared baseline methods.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Algorithm</th>
<th align="left">Language</th>
<th align="left">Theory</th>
<th align="left">Link</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">SingleR</td>
<td align="left">R</td>
<td align="left">Semi-supervised</td>
<td align="left">
<ext-link ext-link-type="uri" xlink:href="https://github.com/dviraran/SingleR">https://github.com/dviraran/SingleR</ext-link>
</td>
</tr>
<tr>
<td align="left">RCA</td>
<td align="left">R</td>
<td align="left">Semi-supervised</td>
<td align="left">
<ext-link ext-link-type="uri" xlink:href="https://github.com/GIS-SP-Group/RCA">https://github.com/GIS-SP-Group/RCA</ext-link>
</td>
</tr>
<tr>
<td align="left">Garnett</td>
<td align="left">R</td>
<td align="left">Semi-supervised</td>
<td align="left">
<ext-link ext-link-type="uri" xlink:href="https://github.com/cole-trapnell-lab/garnett">https://github.com/cole-trapnell-lab/garnett</ext-link>
</td>
</tr>
<tr>
<td align="left">SC3</td>
<td align="left">R</td>
<td align="left">Unsupervised</td>
<td align="left">
<ext-link ext-link-type="uri" xlink:href="https://github.com/hemberg-lab/SC3">https://github.com/hemberg-lab/SC3</ext-link>
</td>
</tr>
<tr>
<td align="left">Seurat</td>
<td align="left">R</td>
<td align="left">Unsupervised</td>
<td align="left">
<ext-link ext-link-type="uri" xlink:href="https://github.com/satijalab/seurat">https://github.com/satijalab/seurat</ext-link>
</td>
</tr>
<tr>
<td align="left">SIMLR</td>
<td align="left">R</td>
<td align="left">Unsupervised</td>
<td align="left">
<ext-link ext-link-type="uri" xlink:href="https://github.com/BatzoglouLabSU/SIMLR">https://github.com/BatzoglouLabSU/SIMLR</ext-link>
</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<sec id="s3-1">
<title>3.1 LFSC outperforms six baseline methods for clustering single-cell transcriptomes</title>
<p>To investigate the clustering performance of LFSC, we applied LFSC and six baseline methods on 21 real scRNA-seq datasets. The parameter settings of the six baseline methods and LFSC are provided in <xref ref-type="sec" rid="s10">Supplementary Table S1</xref>. We also used the ARI, NMI, ACC, and purity metrics to evaluate the clustering results. For the scRNA-seq datasets, we generated reference samples by summing up the gene expression profiles in the same cell types, then averaging them with the number of cells (see details in Methods). The clustering results of LFSC and six baseline methods are presented in <xref ref-type="fig" rid="F2">Figure 2</xref> and <xref ref-type="table" rid="T3">Table 3</xref>. LFSC clearly improved clustering performance for 21 real scRNA-seq datasets. For example, LFSC obtained the optimal ARI solution for 13 out of 21 datasets, followed by SingleR (11 datasets) and SIMLR (one dataset). More precisely, LFSC obtained completely correct labels (ARI value equal to 1) on five datasets, followed by SingleR (four datasets). For NMI values, LFSC obtained the optimal solutions on 14 scRNA-seq datasets and the second-best solutions on seven datasets. For the other two clustering evaluation metrics, LFSC also had better clustering performance. In addition, LFSC statistically improved clustering performance on scRNA-seq datasets. As shown in <xref ref-type="table" rid="T3">Table 3</xref>, we applied statistically significant comparisons with the paired Wilcoxon signed-rank test. The symbol &#x2248; means that there was no significant difference between LFSC and the compared method; the symbol &#x2212; means LFSC was worse than the compared method, and the symbol &#x2b; denotes the opposite. The <italic>p</italic>-value was set as 0.05. The results demonstrate that LFSC is superior to the six baseline methods for four clustering evaluation metrics. The average ARI values in <xref ref-type="table" rid="T3">Table 3</xref> show that LSFC (0.844) increased by about 1.5%, 21.9%, 24.3%, 29.8%, 38.6%, and 30.7% compared to SingleR (0.831), RCA (0.659), Garnett (0.679), SC3 (0.592), Seurat (0.518), and SIMLR (0.585). For NMI values, LFSC (0.856) was superior to SingleR (0.838), RCA (0.681), Garnett (0.685), SC3 (0.633), Seurat (0.570), and SIMLR (0.679). Similar conclusions can be drawn from the results of ACC and purity values. Furthermore, the semi-supervised clustering methods performed better than the unsupervised clustering methods. More precisely, the average ARI values of the semi-supervised clustering methods (0.753) were significantly better than those of the unsupervised clustering methods (0.565). To avoid the basis of comparing with average measurement, we also calculated the mean rank of four clustering evaluation metrics on the real scRNA-seq dataset (see <xref ref-type="table" rid="T3">Table 3</xref>). For ARI values, the mean rank of LFSC was 5.857, which was better than other baseline methods (SingleR: 5.524, RCA: 4.905, Garnett: 4.238, SC3: 3.381, Seurat: 1.667, and SIMLR: 2.429). For NMI values, LFSC produced the optimal mean rank value (5.476), followed by SingleR (5.333), RCA (4.476), and Garnett (4.524). For the other two clustering evaluation metrics, the mean rank values of LFSC were also better than those of the baseline methods. Based on the aforementioned discussion, we believe that LFSC significantly improves clustering.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>lustering evaluation (ARI, NMI, ACC, and purity) heatmap of LFSC and six baseline methods on 21 real scRNA-seq datasets.</p>
</caption>
<graphic xlink:href="fgene-13-1068075-g002.tif"/>
</fig>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Clustering results of LFSC and baseline methods on real scRNA-seq datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th colspan="2" align="left">Metric</th>
<th align="left">Ours</th>
<th align="left">SingleR</th>
<th align="left">RCA</th>
<th align="left">Garnett</th>
<th align="left">SC3</th>
<th align="left">Seurat</th>
<th align="left">SIMLR</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="3" align="left">ARI</td>
<td align="left">Average</td>
<td align="left">0.844</td>
<td align="left">0.831</td>
<td align="left">0.659</td>
<td align="left">0.679</td>
<td align="left">0.592</td>
<td align="left">0.518</td>
<td align="left">0.585</td>
</tr>
<tr>
<td align="left">Mean rank</td>
<td align="left">5.857</td>
<td align="left">5.524</td>
<td align="left">4.905</td>
<td align="left">4.238</td>
<td align="left">3.381</td>
<td align="left">1.667</td>
<td align="left">2.429</td>
</tr>
<tr>
<td align="left">&#x2b;/&#x2212;/&#x2248;</td>
<td align="left">N/A</td>
<td align="left">11/4/7</td>
<td align="left">21/0/1</td>
<td align="left">22/0/0</td>
<td align="left">19/2/1</td>
<td align="left">22/0/0</td>
<td align="left">20/1/1</td>
</tr>
<tr>
<td rowspan="3" align="left">NMI</td>
<td align="left">Average</td>
<td align="left">0.856</td>
<td align="left">0.838</td>
<td align="left">0.681</td>
<td align="left">0.685</td>
<td align="left">0.633</td>
<td align="left">0.570</td>
<td align="left">0.679</td>
</tr>
<tr>
<td align="left">Mean rank</td>
<td align="left">5.476</td>
<td align="left">5.333</td>
<td align="left">4.476</td>
<td align="left">4.524</td>
<td align="left">4.000</td>
<td align="left">1.857</td>
<td align="left">2.333</td>
</tr>
<tr>
<td align="left">&#x2b;/&#x2212;/&#x2248;</td>
<td align="left">N/A</td>
<td align="left">12/4/6</td>
<td align="left">22/0/0</td>
<td align="left">22/0/0</td>
<td align="left">20/2/0</td>
<td align="left">22/0/0</td>
<td align="left">20/1/1</td>
</tr>
<tr>
<td rowspan="3" align="left">ACC</td>
<td align="left">Average</td>
<td align="left">0.902</td>
<td align="left">0.886</td>
<td align="left">0.683</td>
<td align="left">0.674</td>
<td align="left">0.604</td>
<td align="left">0.591</td>
<td align="left">0.645</td>
</tr>
<tr>
<td align="left">Mean rank</td>
<td align="left">5.714</td>
<td align="left">4.810</td>
<td align="left">4.667</td>
<td align="left">4.762</td>
<td align="left">4.048</td>
<td align="left">1.714</td>
<td align="left">2.286</td>
</tr>
<tr>
<td align="left">&#x2b;/&#x2212;/&#x2248;</td>
<td align="left">N/A</td>
<td align="left">10/4/8</td>
<td align="left">22/0/0</td>
<td align="left">22/0/0</td>
<td align="left">20/2/0</td>
<td align="left">22/0/0</td>
<td align="left">20/1/1</td>
</tr>
<tr>
<td rowspan="3" align="left">Purity</td>
<td align="left">Average</td>
<td align="left">0.916</td>
<td align="left">0.902</td>
<td align="left">0.778</td>
<td align="left">0.742</td>
<td align="left">0.669</td>
<td align="left">0.622</td>
<td align="left">0.724</td>
</tr>
<tr>
<td align="left">Mean rank</td>
<td align="left">5.571</td>
<td align="left">5.048</td>
<td align="left">4.952</td>
<td align="left">4.714</td>
<td align="left">3.333</td>
<td align="left">2.095</td>
<td align="left">2.286</td>
</tr>
<tr>
<td align="left">&#x2b;/&#x2212;/&#x2248;</td>
<td align="left">N/A</td>
<td align="left">11/4/7</td>
<td align="left">20/0/2</td>
<td align="left">22/0/0</td>
<td align="left">20/2/0</td>
<td align="left">22/0/0</td>
<td align="left">20/1/1</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-2">
<title>3.2 Robustness analysis of highly confident datasets</title>
<p>To investigate the effects of generated reference samples on clustering performance, we introduced the downsampling strategy to generate reference samples with different sampling ratios (see details in <xref ref-type="sec" rid="s10">Supplementary Section S4</xref>). Since the cell types of some real scRNA-seq datasets were generated by computational approaches (<xref ref-type="bibr" rid="B9">Kiselev et al., 2017</xref>), the highly confident scRNA-seq datasets, called gold standard datasets, were selected as the test datasets, including Yan, Goolam, Deng, and Pollen (see details in <xref ref-type="table" rid="T1">Table 1</xref>). The downsampling ratios were set as 0.05, 0.1, 0.2, 0.4, and 0.6. For different ratios, we implemented LFSC with randomly generated reference samples 30 times and calculated four clustering evaluation metrics. <xref ref-type="fig" rid="F3">Figure 3</xref> shows how the clustering performance of LFSC varied with different downsampling ratios. We found that the clustering performance of LFSC gradually improved by increasing the ratio from 0.05 to 0.6. This demonstrates that more selected samples may generate more representative reference samples. In addition, to investigate the effects of hyperparameter (alpha: <italic>&#x3b1;</italic> and beta: <italic>&#x3b2;</italic>) settings on clustering performance, we implemented LFSC with different combinations of hyperparameters on highly confident scRNA-seq datasets. Alpha (<italic>&#x3b1;</italic>) was selected as 0.1, 1, 10, 50, and 100. Beta (<italic>&#x3b2;</italic>) was selected as 0.0001, 0.001, 0.01, 0.1, 1, 10, 50, and 100. <xref ref-type="fig" rid="F4">Figure 4</xref> shows how the clustering performance of LFSC varied with different combinations of <italic>&#x3b1;</italic> and <italic>&#x3b2;</italic>. We found that the performance of LFSC is stable for a large range of <italic>&#x3b1;</italic> and <italic>&#x3b2;</italic> values. In practice, we recommend optimizing <italic>&#x3b1;</italic> and <italic>&#x3b2;</italic> by fixing <italic>&#x3b1;</italic> and tuning <italic>&#x3b2;</italic>.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>iolin plots of clustering evaluations (ARI, NMI, ACC, and purity) of LFSC with different downsampling ratios for four highly confident scRNA-seq datasets.</p>
</caption>
<graphic xlink:href="fgene-13-1068075-g003.tif"/>
</fig>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>lustering evaluation (ARI, NMI, ACC, and purity) results of LFSC with different hyperparameters on four highly confident scRNA-seq datasets.</p>
</caption>
<graphic xlink:href="fgene-13-1068075-g004.tif"/>
</fig>
</sec>
<sec id="s3-3">
<title>3.3 LFSC speed improves significantly for large scRNA-seq datasets</title>
<p>Since the number of cells sequenced with advanced sequencing protocol is growing, it is important to have a satisfactory running time when analyzing scRNA-seq data. To demonstrate that LFSC is efficiently implemented in practice, we compared the running time between LFSC and six baseline methods on four relatively large real scRNA-seq datasets using a six-core and 32&#xa0;GB memory computer. The Macosko, Chen, Campbell, and Pbmc68K datasets contain 6418, 14437, 21086, and 68579 cells, respectively. <xref ref-type="fig" rid="F5">Figure 5</xref> shows that Seurat (Macosko: 8.491s, Chen: 13.035s, Campbell: 24.946s, Pbmc68K: 140.576s) was the fastest of the seven methods, followed by LFSC (Macosko: 18.944s, Chen: 27.017s, Campbell: 35.792s, Pbmc68K: 181.331s) and SingleR (Macosko: 26.135s, Chen: 34.714s, Campbell: 38.91s, Pbmc68K: 275.74s). Furthermore, LFSC was comparable with RCA (Macosko:153.06s, Chen: 302.94s, Campbell: 451.45s, Pbmc68K: 1489.14s) and Garnett (Macosko: 27.66s, Chen: 66.98s, Campbell: 72.89s, Pbmc68K: 285.75s). SC3 and SIMLR cost significantly more time than the other five methods. Although Seurat is superior in running time to LFSC, SingleR, RCA, and Garnett, it is the closest method, especially when Seurat has applied a parallelization operator that is lacking in LFSC.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Running time of LFSC and six baseline methods on datasets Macosko, Chen, Campbell, and Pbmc68K.</p>
</caption>
<graphic xlink:href="fgene-13-1068075-g005.tif"/>
</fig>
</sec>
<sec id="s3-4">
<title>3.4 The visualization ability of LFSC</title>
<p>We investigated LFSC&#x2019;s ability to embed cells into the two-dimensional space using <italic>t</italic>-distributed stochastic neighbor embedding (t-SNE). In the experiment, we first chose four real gold standard scRNA-seq datasets (shown in <xref ref-type="table" rid="T2">Table 2</xref>). Then, we selected Seurat and SIMLR as compared baseline methods for LFSC since they are graph-based models. In particular, SIMLR and LFSC are variants of spectral clustering, which generates the affinity and decomposition matrixes in the clustering process. Thus, we implemented t-SNE on the generated matrixes for SIMLR and LFSC (see <xref ref-type="fig" rid="F6">Figure 6</xref>). The silhouette coefficient (Sil, see <xref ref-type="sec" rid="s10">Supplementary Section S2</xref>) is a widely used cluster metric that compares inter- and intra-distances among data points based on the clustering partition. To measure the quality of the visualization, we used the silhouette coefficient to analyze whether cells of the same types were closer, while those from different cell types were more separated in the t-SNE space. <xref ref-type="fig" rid="F6">Figure 6</xref> shows the visualization plots of LFSC, Seurat, and SIMLR for the four highly confident scRNA-seq datasets. For the dataset Yan, we found that LFSC(U) and LFSC(A) produced the best (Sil: 0.9231) and the second-best (Sil: 0.86) performances, respectively, which were significantly better than Seurat, SIMLR(S), and SIMLR(F). Similar conclusions were obtained for the dataset Goolam, where the Sil values of LFSC(U), LFSC(A), Seurat, SIMLR(S), and SIMLR(F) were 0.564, 0.4236, 0.2723, 0.2784, and 0.3491. It is noted that only LFSC identified the correct number of cell types (five clusters, see <xref ref-type="fig" rid="F6">Figure 6</xref>), while Seurat and SIMLR clustered cells into six and seven clusters, respectively. In the dataset Deng, LFSC was not only superior to compared methods in Sil values and clustering but also had the best performance in visualization since LFSC separated cells from different cell types and combined cells from the same cell types. In the dataset Pollen, only LFSC detected the correct number of clusters (see <xref ref-type="fig" rid="F6">Figure 6</xref>), and the learned embedding space of LFSC was well separated, while cells in the same cell type were more compact. In conclusion, the aforementioned analysis indicates that LFSC has better visualization ability than do the compared baseline methods.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>t-SNE plots of LFSC and baseline methods with corresponding silhouette coefficients and clustering results on four highly confident scRNA-seq datasets.</p>
</caption>
<graphic xlink:href="fgene-13-1068075-g006.tif"/>
</fig>
</sec>
<sec id="s3-5">
<title>3.5 Ablation study of LFSC</title>
<p>We completed the ablation study to investigate the importance of each component in LFSC. In particular, the ablation experiment was designed as follows. 1) Without HVG selection, all overlap genes between the scRNA-seq data and the reference transcriptomics data were selected. 2) Without reference transcriptomics data, we generated the reference sample by directly implementing the K-mean algorithm on the scRNA-seq data. <xref ref-type="sec" rid="s10">Supplementary Table S2</xref> summarizes the NMI, ARI, ACC, and purity values for the 21 scRNA-seq datasets. In the first step, the HVG selection had a positive impact on clustering performance, which not only reduced redundant features but also sped up the convergence process. Second, without the reference transcriptomics data, the clustering performance significantly deteriorated for the 21 scRNA-seq datasets, which demonstrates that the reference transcriptomics data are important for LFSC to improve robustness and efficiency. In conclusion, the aforementioned analysis indicates that all components in LFSC were designed effectively and reasonably.</p>
</sec>
<sec id="s3-6">
<title>3.6 LFSC detected two new function-specific subtypes of tumor-infiltrating lymphocytes</title>
<p>Exploring subtypes of the tumor-infiltrating lymphocytes is a benefit for the investigation of immunotherapies and associated clinical responses in cancers. To investigate the exploring ability in biological analysis, we downloaded a GEO dataset (GSE98638) containing 5,063 single T cells isolated from peripheral blood, tumor, and adjacent normal tissues from six hepatocellular carcinoma patients (<xref ref-type="bibr" rid="B27">Zheng et al., 2017</xref>). We first implemented <italic>clustree</italic> (<xref ref-type="bibr" rid="B23">Zappia and Oshlack 2018</xref>) to explore the correct number of clusters. In a cluster tree plot, one node represents a cluster, and a larger node means the cluster has more data points. Since the tree has no branches and the leaf nodes have similar sizes (see <xref ref-type="sec" rid="s10">Supplementary Figure S1</xref>), when the number of clusters equaled 26, the stability and robustness of clustering were best. Then, we used the Immunological Genome Project (ImmGen) database (<xref ref-type="bibr" rid="B1">Aran et al., 2019</xref>) as the reference bulk RNA-seq data and applied LFSC with the number of clusters equal to 26. The heatmap of the Pearson coefficient between the 26 clusters and 11&#xa0;T-cell subtypes is shown in <xref ref-type="fig" rid="F7">Figure 7A</xref>. Cluster 5 and cluster 14 were significantly unassociated with the 11&#xa0;T-cell subtypes. We annotated the other 24 clusters with the 11&#xa0;T-cell subtypes by maximizing the associated coefficient values (see <xref ref-type="fig" rid="F7">Figure 7B</xref>). Cluster 5 and cluster 14 were well separated from other clusters, which indicates that they are different from the six annotated T-cell subtypes in function or biological process.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Single-cell transcriptional profiling of the tumor-infiltrating lymphocytes. <bold>(A)</bold> Heatmap of Pearson coefficient values between 26 clusters and 11&#xa0;T-cell subtypes; <bold>(B)</bold> t-SNE plots of tumor-infiltrating lymphocytes annotated by LFSC using the ImmGen database as the reference; <bold>(C)</bold> volcano plot showing differentially expressed genes in cluster 5; <bold>(D)</bold> volcano plot showing differentially expressed genes in cluster 14; <bold>(E)</bold> Venn diagram showing the overlap of DEGs between cluster 5 and cluster 14; <bold>(F)</bold> heatmap of the enriched term across DEGs on cluster 5 and cluster 14.</p>
</caption>
<graphic xlink:href="fgene-13-1068075-g007.tif"/>
</fig>
<p>To determine the biological differences between cluster 5 and cluster 14, we applied Seurat to identify the differentially expressed genes (DEGs) of cluster 5 and cluster 14. The DEG genes were selected under the criteria 1) absolute log2-fold change larger than 1.5, and 2) adjusted <italic>p</italic>-value of F test &#x3c;0.05. There were 48 DEGs (11 upregulated genes and 37 downregulated genes for cluster 5 (see <xref ref-type="fig" rid="F7">Figure 7C</xref>) and 28 DEGs (7 upregulated genes and 21 downregulated genes; see <xref ref-type="fig" rid="F7">Figure 7D</xref>) for cluster 14. <xref ref-type="sec" rid="s10">Supplementary Figure S2</xref> shows the t-SNE projection of the tumor-infiltrating lymphocytes to be colored by DEGs of cluster 5 (<italic>RTKN2</italic>, <italic>IL2RA</italic>, <italic>SELL</italic>, <italic>LMNA</italic>, <italic>TFRC</italic>, and <italic>CCR8</italic>) and cluster 14 (<italic>GZMA</italic>, <italic>GZMK</italic>, <italic>PTGDR</italic>, <italic>TNF</italic>, <italic>CCR2</italic>, and <italic>IL18RAP</italic>). Some genes are associated with immunological diseases. For example, <italic>CCR2</italic> and <italic>CCR8</italic> are protein-coding genes associated with diseases including human immunodeficiency virus type 1 and molluscum contagiosum. <italic>GZMA</italic> and <italic>GZMK</italic> are well-known marker genes regarded as T-cell- and natural killer cell-specific serine proteases. Some studies have demonstrated <italic>IL2RA</italic> and <italic>IL18RAP</italic> to be associated with the same cytokine signaling pathway in the immune system.</p>
<p>To further demonstrate that cluster 5 and cluster 14 have specific functions, we completed functional enrichment analysis on the DEGs with the analysis tool <italic>Metascape</italic> (<xref ref-type="bibr" rid="B28">Zhou et al., 2019</xref>). As seen in <xref ref-type="fig" rid="F7">Figure 7F</xref>, we found that the most enriched functions for cluster 5 and cluster 14 were enriched for different biological terms. For example, cluster 5 enriched for the term GO: 0019835, which is related to the rupture of cell membranes and the loss of cytoplasm, and GO: 0002520, which is associated with immune system development. Although cluster 14 is also enriched in the same terms, like GO: 0002520, GO:0050865 (regulation of cell activation), and GO: 0001775 (cell activation), DEGs of cluster 14 were significantly enriched in the biological process of leukocyte adhesion to vascular endothelial cells (GO:0061756) and the development of regulatory T lymphocytes (R-HSA-8877330). This proves that the newly detected cluster 5 and cluster 14 have different functions in biological processes.</p>
</sec>
<sec id="s3-7">
<title>3.7 LFSC detects DEGs associated with biomarkers of liver cancer</title>
<p>To investigate the clinical research values of selected DEGs of newly found subtypes (cluster 5 and cluster 9), liver hepatocellular carcinoma (LIHC) samples from The Cancer Genome Atlas Program (TCGA) dataset (<xref ref-type="bibr" rid="B19">Tomczak et al., 2015</xref>) were used to test the correlations between selected genes and patient survival. The analysis details are provided in <xref ref-type="sec" rid="s10">Supplementary Section S5</xref>. <xref ref-type="fig" rid="F8">Figure 8</xref> shows the survival curves for the DEGs of cluster 5 and cluster 14. Three hundred and seventy LIHC tumor samples were divided into two groups based on the expression profiles of six gene sets composed of different combinations of DEGs. Significantly, the intersection set of the DEGs on cluster 5 and cluster 14 (<italic>p</italic> &#x3d; 0.16, paired Wilcoxon test, <xref ref-type="fig" rid="F8">Figure 8A</xref>), the combined set of these (<italic>p</italic> &#x3d; 0.14, paired Wilcoxon test, <xref ref-type="fig" rid="F8">Figure 8D</xref>), and the DEGs of cluster 14 (<italic>p</italic> &#x3d; 0.21, paired Wilcoxon test, <xref ref-type="fig" rid="F8">Figure 8E</xref>) were statistically unassociated with poor prognosis. Meanwhile, the DEGs of cluster 5 (<italic>p</italic> &#x3d; 0.034, paired Wilcoxon test, <xref ref-type="fig" rid="F8">Figure 8B</xref>) and cluster-specific DEGs (paired Wilcoxon test, <italic>p</italic> &#x3c; 0.0001, <xref ref-type="fig" rid="F8">Figure 8C</xref> and <italic>p</italic> &#x3d; 0.024, <xref ref-type="fig" rid="F8">Figure 8D</xref>) correlated with good prognosis in TCGA cohort. Thus, our results provide evidence that the DEGs of newly found clusters are biomarkers in the tumor microenvironment of LIHC.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Results of survival analysis on selected DEGs across TCGA LIHC clinical data. Survival curves on the intersection of DEGs between cluster 5 and cluster 14 <bold>(A)</bold>; on DEGs of cluster 5 <bold>(B)</bold>; on DEGs belonging to cluster 5 but not to cluster 14 <bold>(C)</bold>; on the combination of DEGs between cluster 5 and cluster 14 <bold>(D)</bold>; on DEGs of cluster 14 <bold>(E)</bold>; on DEGs belonging to cluster 14 but not to cluster 5 <bold>(F)</bold>.</p>
</caption>
<graphic xlink:href="fgene-13-1068075-g008.tif"/>
</fig>
<p>
<statement content-type="algorithm" id="Algorithm_1">
<label>Algorithm 1</label>
<p>Framework of LFSC<list list-type="simple">
<list-item>
<p>
<italic>Input</italic>:</p>
</list-item>
<list-item>
<p>&#x2003;scRNA-seq dataset: <inline-formula id="inf70">
<mml:math id="m80">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>;</p>
</list-item>
<list-item>
<p>&#x2003;Reference dataset: <inline-formula id="inf71">
<mml:math id="m81">
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;Hyperparameters: <inline-formula id="inf72">
<mml:math id="m82">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf73">
<mml:math id="m83">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<italic>Begin</italic>
</p>
</list-item>
<list-item>
<p>1.&#x2003;<bold>
<italic>Step 1</italic>:</bold> Implement quality control and Normalize data <inline-formula id="inf74">
<mml:math id="m84">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> using (3);</p>
</list-item>
<list-item>
<p>2.&#x2003;<bold>
<italic>Step 2</italic>:</bold> Identify highly variable overlap genes with the package <italic>Seurat</italic> on <inline-formula id="inf75">
<mml:math id="m85">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf76">
<mml:math id="m86">
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>;</p>
</list-item>
<list-item>
<p>3.&#x2003;<bold>
<italic>Step 3</italic>:</bold> Estimate the cluster number, <italic>k</italic>, using the package <italic>clustree</italic> on <inline-formula id="inf77">
<mml:math id="m87">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>;</p>
</list-item>
<list-item>
<p>4.&#x2003;<bold>
<italic>Step 4</italic>:</bold> Generate the reference samples <inline-formula id="inf78">
<mml:math id="m88">
<mml:mrow>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> with reference dataset <inline-formula id="inf79">
<mml:math id="m89">
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> using (4);</p>
</list-item>
<list-item>
<p>5.&#x2003;<bold>
<italic>Step 5</italic>:</bold> Construct the structured anchor graph <inline-formula id="inf80">
<mml:math id="m90">
<mml:mrow>
<mml:mi mathvariant="double-struck">A</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> with reference samples <inline-formula id="inf81">
<mml:math id="m91">
<mml:mrow>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>;</p>
</list-item>
<list-item>
<p>6.&#xa0;&#xa0;<bold>
<italic>Step 5.1</italic>:</bold> Initialize the matrix <inline-formula id="inf82">
<mml:math id="m92">
<mml:mrow>
<mml:mi mathvariant="double-struck">F</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>;</p>
</list-item>
<list-item>
<p>7.&#xa0;&#xa0;<bold>While</bold> convergence condition does not meet <bold>do</bold>
</p>
</list-item>
<list-item>
<p>8.&#xa0;&#xa0;<bold>
<italic>Step 5.2</italic>:</bold> Update <inline-formula id="inf83">
<mml:math id="m93">
<mml:mrow>
<mml:mi mathvariant="double-struck">A</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> in <xref ref-type="sec" rid="s10">Supplementary Eq. S4</xref> using convex quadratic programming;</p>
</list-item>
<list-item>
<p>9.&#xa0;&#xa0;<bold>
<italic>Step 5.3</italic>:</bold> Update <inline-formula id="inf84">
<mml:math id="m94">
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> in <xref ref-type="sec" rid="s10">Supplementary Eq. S7</xref> by <xref ref-type="sec" rid="s10">Supplementary Lemma S1</xref>;</p>
</list-item>
<list-item>
<p>10.&#xa0;&#xa0;<italic>end while</italic>
</p>
</list-item>
<list-item>
<p>11.&#x2003;<bold>
<italic>Step 6</italic>:</bold> Run K-means on with the cluster number <italic>k</italic>;</p>
</list-item>
<list-item>
<p>
<italic>End</italic>
</p>
</list-item>
<list-item>
<p>
<italic>Output:</italic>
</p>
</list-item>
<list-item>
<p>&#x2003;Clustering result <inline-formula id="inf85">
<mml:math id="m95">
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
</list>
</p>
</statement>
</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s4">
<title>4 Conclusion</title>
<p>We presented a linear fast semi-supervised clustering method, based on bulk and single-cell transcriptomes, that has the following characteristics: 1) LFSC generates reference samples with bulk-RNA-seq or labeled single-cell RNA-seq data, which implicitly provides the label information to the graph construction process; 2) LFSC introduces anchor graph theory to measure the similarities between unlabeled cells and a small number of reference samples, which significantly reduces the size of the graph; and 3) the <italic>K</italic>-connectivity constraint is added to the cell-reference anchor graph to preserve the underlying clustering structure of the data. In general, the proposed mechanisms not only improve the clustering accuracy of the model but also make its overall complexity linearly related to data size, and they reduce the memory overhead of the model.</p>
<p>The experiments on several scRNA-seq datasets demonstrate the following conclusions: 1) LFSC is superior to state-of-the-art methods in clustering accuracy and robustness; 2) the visualization analysis proves that the anchor graph in LFSC can retain the correct clustering structure of the data, and the learning embedding space has good separation, which has a better visualization effect compared with the benchmark methods; and 3) the results of ablation analysis show that all components of LFSC are effective and reasonable. In addition, the case study of infiltrating T cells in liver cancer demonstrated that LFSC shows promising application potential in discovering new cell types, identifying differentially expressed genes, and exploring new cancer-related biomarkers.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s10">Supplementary Material</xref>, and further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s6">
<title>Author contributions</title>
<p>QL designed the algorithm and completed the experiments. YL prepared and edited the manuscript. DW checked and proofread the manuscript. JL supervised and guided the research process.</p>
</sec>
<sec id="s7">
<title>Funding</title>
<p>The study was supported by the National Natural Science Foundation of China (Grant nos 62072095 and 61771165) and the National Key R&#x26;D Program of China (grant no. 2021YFC2100100).</p>
</sec>
<ack>
<p>The authors would like to thank the reviewers for their reading and constructive comments.</p>
</ack>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors, and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s10">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2022.1068075/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2022.1068075/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table1.DOCX" id="SM1" mimetype="application/DOCX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Aran</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Looney</surname>
<given-names>A. P.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Fong</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Hsu</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Reference-based analysis of lung single-cell sequencing reveals a transitional profibrotic macrophage</article-title>. <source>Nat. Immunol.</source> <volume>20</volume> (<issue>2</issue>), <fpage>163</fpage>&#x2013;<lpage>172</lpage>. <pub-id pub-id-type="doi">10.1038/s41590-018-0276-y</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bartoschek</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Oskolkov</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Bocci</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lovrot</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Larsson</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Sommarin</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Spatially and functionally distinct subclasses of breast cancer-associated fibroblasts revealed by single cell RNA sequencing</article-title>. <source>Nat. Commun.</source> <volume>9</volume> (<issue>1</issue>), <fpage>5150</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-018-07582-3</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Butler</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hoffman</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Smibert</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Papalexi</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Satija</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Integrating single-cell transcriptomic data across different conditions, technologies, and species</article-title>. <source>Nat. Biotechnol.</source> <volume>36</volume> (<issue>5</issue>), <fpage>411</fpage>&#x2013;<lpage>420</lpage>. <pub-id pub-id-type="doi">10.1038/nbt.4096</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Large scale spectral clustering with landmark-based representation</article-title>. <source>Twenty-fifth AAAI Conf. Artif. Intell.</source> <volume>45</volume>, <fpage>1669</fpage>&#x2013;<lpage>1680</lpage>. <pub-id pub-id-type="doi">10.1109/TCYB.2014.2358564</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chung</surname>
<given-names>F. R. K.</given-names>
</name>
<name>
<surname>Graham</surname>
<given-names>F. C.</given-names>
</name>
</person-group> (<year>1997</year>). <source>Spectral graph theory</source>. <publisher-loc>Rhode Island, United States</publisher-loc>: <publisher-name>American Mathematical Soc</publisher-name>.</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Conesa</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Madrigal</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Tarazona</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gomez-Cabrero</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Cervera</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>McPherson</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>A survey of best practices for RNA-seq data analysis</article-title>. <source>Genome Biol.</source> <volume>17</volume>, <fpage>13</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-016-0881-8</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gate</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Saligrama</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Leventhal</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>A. C.</given-names>
</name>
<name>
<surname>Unger</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Middeldorp</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Clonally expanded CD8 T cells patrol the cerebrospinal fluid in Alzheimer&#x2019;s disease</article-title>. <source>Nature</source> <volume>577</volume> (<issue>7790</issue>), <fpage>399</fpage>&#x2013;<lpage>404</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-019-1895-7</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Heydari</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Davalos</surname>
<given-names>O. A.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Hoyer</surname>
<given-names>K. K.</given-names>
</name>
<name>
<surname>Sindi</surname>
<given-names>S. S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Activa: Realistic single-cell RNA-seq generation with automatic cell-type identification using introspective variational autoencoders</article-title>. <source>Bioinformatics</source> <volume>38</volume>, <fpage>2194</fpage>&#x2013;<lpage>2201</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btac095</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kiselev</surname>
<given-names>V. Y.</given-names>
</name>
<name>
<surname>Kirschner</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Schaub</surname>
<given-names>M. T.</given-names>
</name>
<name>
<surname>Andrews</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yiu</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chandra</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>SC3: Consensus clustering of single-cell RNA-seq data</article-title>. <source>Nat. Methods</source> <volume>14</volume> (<issue>5</issue>), <fpage>483</fpage>&#x2013;<lpage>486</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.4236</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Courtois</surname>
<given-names>E. T.</given-names>
</name>
<name>
<surname>Sengupta</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>K. H.</given-names>
</name>
<name>
<surname>Goh</surname>
<given-names>J. J. L.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Reference component analysis of single-cell transcriptomes elucidates cellular heterogeneity in human colorectal tumors</article-title>. <source>Nat. Genet.</source> <volume>49</volume> (<issue>5</issue>), <fpage>708</fpage>&#x2013;<lpage>718</lpage>. <pub-id pub-id-type="doi">10.1038/ng.3818</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C. Y.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>From bulk, single-cell to spatial RNA sequencing</article-title>. <source>Int. J. Oral Sci.</source> <volume>13</volume> (<issue>1</issue>), <fpage>36</fpage>. <pub-id pub-id-type="doi">10.1038/s41368-021-00146-0</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Ccpe: Cell cycle pseudotime estimation for single cell RNA-seq data</article-title>. <source>Nucleic Acids Res.</source> <volume>50</volume> (<issue>2</issue>), <fpage>704</fpage>&#x2013;<lpage>716</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkab1236</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lotfollahi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Naghipourfar</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Luecken</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>Khajavi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Buttner</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wagenstetter</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Mapping single-cell data to reference atlases by transfer learning</article-title>. <source>Nat. Biotechnol.</source> <volume>40</volume>, <fpage>121</fpage>&#x2013;<lpage>130</lpage>. <pub-id pub-id-type="doi">10.1038/s41587-021-01001-7</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Nie</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.-L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>K-Multiple-Means</article-title>,&#x201d; in <conf-name>Proceedings of the 25th ACM SIGKDD International Conference on Knowledge Discovery &#x26; Data Mining</conf-name>, <conf-loc>Anchorage, AK</conf-loc>, <conf-date>August 4&#x2013;8, 2019</conf-date>.</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pliner</surname>
<given-names>H. A.</given-names>
</name>
<name>
<surname>Shendure</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Trapnell</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Supervised classification enables rapid annotation of cell atlases</article-title>. <source>Nat. Methods</source> <volume>16</volume> (<issue>10</issue>), <fpage>983</fpage>&#x2013;<lpage>986</lpage>. <pub-id pub-id-type="doi">10.1038/s41592-019-0535-3</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Clustering and classification methods for single-cell RNA-sequencing data</article-title>. <source>Brief. Bioinform.</source> <volume>21</volume> (<issue>4</issue>), <fpage>1196</fpage>&#x2013;<lpage>1208</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbz062</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhuang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>scDeepSort: a pre-trained cell-type annotation method for single-cell transcriptomics using deep learning with a weighted graph neural network</article-title>. <source>Nucleic Acids Res.</source> <volume>49</volume>, <fpage>e122</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkab775</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sokolowski</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Faykoo-Martinez</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Erdman</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Hou</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Single-cell mapper (scMappR): Using scRNA-seq to infer the cell-type specificities of differentially expressed genes</article-title>. <source>Nar. Genom. Bioinform.</source> <volume>3</volume> (<issue>1</issue>), <fpage>lqab011</fpage>. <pub-id pub-id-type="doi">10.1093/nargab/lqab011</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tomczak</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Czerwi&#x144;ska</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Wiznerowicz</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>The cancer Genome Atlas (TCGA): An immeasurable source of knowledge</article-title>. <source>Contemp. Oncol.</source> <volume>2015</volume> (<issue>1</issue>), <fpage>68</fpage>&#x2013;<lpage>77</lpage>. <pub-id pub-id-type="doi">10.5114/wo.2014.47136</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Ramazzotti</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>De Sano</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Pierson</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Batzoglou</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Simlr: A tool for large-scale Genomic Analyses by multi-kernel learning</article-title>. <source>Proteomics</source> <volume>18</volume> (<issue>2</issue>), <fpage>1700232</fpage>. <pub-id pub-id-type="doi">10.1002/pmic.201700232</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Callr: A semi-supervised cell-type annotation method for single-cell RNA sequencing data</article-title>. <source>Bioinformatics</source> <volume>37</volume> (<issue>1</issue>), <fpage>i51</fpage>&#x2013;<lpage>i58</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btab286</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zakharov</surname>
<given-names>P. N.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Unanue</surname>
<given-names>E. R.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Single-cell RNA sequencing of murine islets shows high cellular complexity at all stages of autoimmune diabetes</article-title>. <source>J. Exp. Med.</source> <volume>217</volume> (<issue>6</issue>), <fpage>e20192362</fpage>. <pub-id pub-id-type="doi">10.1084/jem.20192362</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zappia</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Oshlack</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Clustering trees: A visualization for evaluating clusterings at multiple resolutions</article-title>. <source>Gigascience</source> <volume>7</volume> (<issue>7</issue>). <pub-id pub-id-type="doi">10.1093/gigascience/giy083</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>J.-Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.-M.</given-names>
</name>
<name>
<surname>Xing</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>J.-W.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Single-cell landscape of immunological responses in patients with COVID-19</article-title>. <source>Nat. Immunol.</source> <volume>21</volume> (<issue>9</issue>), <fpage>1107</fpage>&#x2013;<lpage>1118</lpage>. <pub-id pub-id-type="doi">10.1038/s41590-020-0762-x</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2021a</year>). <article-title>CellCall: Integrating paired ligand-receptor and transcription factor activities for cell-cell communication</article-title>. <source>Nucleic Acids Res.</source> <volume>49</volume>, <fpage>8520</fpage>&#x2013;<lpage>8534</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkab638</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ouyang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xiong</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2021b</year>). <article-title>Single-cell RNA sequencing in cancer research</article-title>. <source>J. Exp. Clin. Cancer Res.</source> <volume>40</volume> (<issue>1</issue>), <fpage>81</fpage>&#x2013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.1186/s13046-021-01874-1</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yoo</surname>
<given-names>J. K.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Landscape of infiltrating T cells in liver cancer revealed by single-cell sequencing</article-title>. <source>Cell</source> <volume>169</volume> (<issue>7</issue>), <fpage>1342</fpage>&#x2013;<lpage>1356</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2017.05.035</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Pache</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Khodabakhshi</surname>
<given-names>A. H.</given-names>
</name>
<name>
<surname>Tanaseichuk</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Metascape provides a biologist-oriented resource for the analysis of systems-level datasets</article-title>. <source>Nat. Commun.</source> <volume>10</volume> (<issue>1</issue>), <fpage>1523</fpage>&#x2013;<lpage>1610</lpage>. <pub-id pub-id-type="doi">10.1038/s41467-019-09234-6</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>