<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">847112</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2022.847112</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>NISC: Neural Network-Imputation for Single-Cell RNA Sequencing and Cell Type Clustering</article-title>
<alt-title alt-title-type="left-running-head">Zhang et al.</alt-title>
<alt-title alt-title-type="right-running-head">NISC Imputation</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Xiang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1621229/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Zhuo</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1762415/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Bhadani</surname>
<given-names>Rahul</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Cao</surname>
<given-names>Siyang</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lu</surname>
<given-names>Meng</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lytal</surname>
<given-names>Nicholas</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/803255/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Yin</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1027751/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>An</surname>
<given-names>Lingling</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/501662/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Interdisciplinary Program in Statistics and Data Science</institution>, <institution>University of Arizona</institution>, <addr-line>Tucson</addr-line>, <addr-line>AZ</addr-line>, <country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Biosystems Engineering</institution>, <institution>University of Arizona</institution>, <addr-line>Tucson</addr-line>, <addr-line>AZ</addr-line>, <country>United States</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Electrical and Computer Engineering</institution>, <institution>University of Arizona</institution>, <addr-line>Tucson</addr-line>, <addr-line>AZ</addr-line>, <country>United States</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Department of Mathematics and Statistics</institution>, <institution>California State University at Chico</institution>, <addr-line>Chico</addr-line>, <addr-line>CA</addr-line>, <country>United States</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>College of Pharmacy</institution>, <institution>University of Arizona</institution>, <addr-line>Tucson</addr-line>, <addr-line>AZ</addr-line>, <country>United States</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Department of Biostatistics and Epidemiology</institution>, <institution>University of Arizona</institution>, <addr-line>Tucson</addr-line>, <addr-line>AZ</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/117988/overview">Robert Friedman</ext-link>, Retired from University of South Carolina, United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1491704/overview">Andrea Tangherloni</ext-link>, University of Bergamo, Italy</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/749021/overview">Lu Zhang</ext-link>, Hong Kong Baptist University, Hong Kong SAR, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Lingling An, <email>anling@email.arizona.edu</email>
</corresp>
<fn fn-type="other">
<p>This article was submitted to Computational Genomics, a section of the journal Frontiers in Genetics</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>03</day>
<month>05</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>847112</elocation-id>
<history>
<date date-type="received">
<day>01</day>
<month>01</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>04</day>
<month>04</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Zhang, Chen, Bhadani, Cao, Lu, Lytal, Chen and An.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Zhang, Chen, Bhadani, Cao, Lu, Lytal, Chen and An</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Single-cell RNA sequencing (scRNA-seq) reveals the transcriptome diversity in heterogeneous cell populations as it allows researchers to study gene expression at single-cell resolution. The latest advances in scRNA-seq technology have made it possible to profile tens of thousands of individual cells simultaneously. However, the technology also increases the number of missing values, i. e, dropouts, from technical constraints, such as amplification failure during the reverse transcription step. The resulting sparsity of scRNA-seq count data can be very high, with greater than 90% of data entries being zeros, which becomes an obstacle for clustering cell types. Current imputation methods are not robust in the case of high sparsity. In this study, we develop a Neural Network-based Imputation for scRNA-seq count data, NISC. It uses autoencoder, coupled with a weighted loss function and regularization, to correct the dropouts in scRNA-seq count data. A systematic evaluation shows that NISC is an effective imputation approach for handling sparse scRNA-seq count data, and its performance surpasses existing imputation methods in cell type identification.</p>
</abstract>
<kwd-group>
<kwd>imputation</kwd>
<kwd>deep learning</kwd>
<kwd>single cell RNA-seq</kwd>
<kwd>dropout</kwd>
<kwd>autoencoder</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Single-cell RNA sequencing (scRNA-seq) is designed to profile gene expression at the single-cell level, making it possible to study the heterogeneity among individual cells (<xref ref-type="bibr" rid="B33">Pierson and Yau, 2015</xref>). However, one important characteristic of scRNA-seq data is a phenomenon called &#x201c;dropout&#x201d;, which causes challenges in data analysis. These dropout events occur because of the low amounts of genetic material in individual cells and inefficient mRNA capture, as well as the stochasticity of mRNA expression (<xref ref-type="bibr" rid="B24">Lin et al., 2017</xref>). Specifically, a large number of dropouts is due to transcripts lost in the RNA reverse transcription procedure during library preparation (<xref ref-type="bibr" rid="B15">Gordon et al., 2015</xref>). In other words, many zero counts in the gene expression data are not &#x201c;true&#x201d; values. Consequently, the scRNA-seq data may be incredibly sparse due to the high dropout rate, e.g., more than 90% of the expression counts have values of zero. Imputation has become an essential preprocessing step for downstream analysis of scRNA-seq data (<xref ref-type="bibr" rid="B47">Tracy et al., 2019</xref>). Recent studies have shown that some imputation methods improve downstream analysis and have already been implemented in scRNA-seq analysis pipelines (<xref ref-type="bibr" rid="B56">Zhang and Zhang, 2018</xref>). Meanwhile, with the increasing size of scRNA-seq data sets, appropriate imputation methods are necessary to compensate for these dropouts to reduce the impacts of missing values (<xref ref-type="bibr" rid="B2">Angerer et al., 2017</xref>).</p>
<p>Many methods have recently been developed for modeling and processing scRNA-seq count data, including scVI (<xref ref-type="bibr" rid="B26">Lopez et al., 2018</xref>), VASC (<xref ref-type="bibr" rid="B51">Wang and Gu, 2018</xref>), scSVA (<xref ref-type="bibr" rid="B43">Sun et al., 2019</xref>), scVAE (<xref ref-type="bibr" rid="B16">Gronbech et al., 2020</xref>), and scAEspy (<xref ref-type="bibr" rid="B45">Tangherloni et al., 2021</xref>), which used neural networks to reduce the noisy dimension to increase the accuracy of downstream analysis. There also exists quite a number of methods to impute the missing values in scRNA-seq data, including scImpute, MAGIC (<xref ref-type="bibr" rid="B49">Van Dijk et al., 2018</xref>), SAVER (<xref ref-type="bibr" rid="B18">Huang et al., 2018</xref>), DrImpute (<xref ref-type="bibr" rid="B14">Gong et al., 2018</xref>), VIPER (<xref ref-type="bibr" rid="B10">Chen and Zhou, 2018</xref>), ALRA (<xref ref-type="bibr" rid="B25">Linderman et al., 2018</xref>), EnImpute (<xref ref-type="bibr" rid="B57">Zhang et al., 2019</xref>) and scDoc (<xref ref-type="bibr" rid="B34">Ran et al., 2020</xref>). In ScImpute, separated Gamma-Normal mixture models are constructed for different cell subgroups to calculate the probabilities of drop-out. It leverages information of cell similarity in terms of genes with a lower dropout probability and then imputes the values of genes with higher dropout probability. MAGIC is a method that shares information across similar cells <italic>via</italic> data diffusion to predict the true gene expression level. SAVER is a Bayesian-based imputation method that imputes dropout values and generates a substitution for each gene. DrImpute is a clustering-based method that generates estimations using cluster priors and distance matrices. ALRA is an adaptively-thresholded low-rank approximation method that rescales the scRNA-seq expression matrix using randomized singular value decomposition. VIPER is a statistical method that fits a linear model for each cell by cell-cell interaction.</p>
<p>Basically, these methods impute dropouts by leveraging information on similarities between cells/genes using the correlation structure of the scRNA-seq data. For example, current imputation approaches, including scImpute and DrImpute, identify similar cells/genes based on clustering and then impute the missing data by averaging the gene expression values for each detected cluster. The accuracy of these imputation methods highly relies on clustering analysis. EnImpute combines the imputation results obtained from eight different imputation methods and calculates the expected values. scDoc imputes dropout events by leveraging information for the same gene from highly similar cells. However, current methods may fail to capture the nonlinearity and the count structure of the scRNA-seq data. Moreover, it becomes more challenging for the traditional imputation methods to handle datasets with increasing size (<xref ref-type="bibr" rid="B12">Eraslan et al., 2019</xref>).</p>
<p>Recently, some deep learning-based imputation methods have been developed for efficiently handling the higher dimensional scRNA-seq data, such as DCA (<xref ref-type="bibr" rid="B12">Eraslan et al., 2019</xref>), DeepImpute (<xref ref-type="bibr" rid="B4">Arisdakessian et al., 2019</xref>), AutoImpute (<xref ref-type="bibr" rid="B44">Talwar et al., 2018</xref>), LATE (<xref ref-type="bibr" rid="B5">Badsha et al., 2020</xref>), scIGAN (<xref ref-type="bibr" rid="B54">Xu et al., 2020</xref>), and scGNN (<xref ref-type="bibr" rid="B52">Wang et al., 2021</xref>). DCA is a neural network-based denoising method for scRNA-seq count data. This method assumes that the scRNA-seq count data follow a negative binomial distribution and then are denoised by maximizing a likelihood function. DeepImpute is a deep learning-based method that splits the genes into several subsets of neural networks. However, these imputation methods lack accuracy and power in handling highly sparse data. AutoImpute uses autoencoder with one hidden layer to impute missing values in scRNA-seq data by minimizing the Euclidean cost function. LATE uses autoencoder to train on nonzero data by minimizing the loss function, therefore imputing the missing values based on information of dependence between genes and cells. scIGAN uses generative adversarial networks for scRNA-seq imputation. scGNN uses a graph neural network for scRNA-seq analysis.</p>
<p>In this study, we develop a novel imputation method, Neural Network-based Imputation for scRNA-seq data (NISC) to improve cell type clustering. It is based on neural networks with a novel weighted loss function, coupled with regularizations. Through a series of simulation studies and real data analysis, NISC is compared with the other imputation methods, including AutoImpute, DCA, DeepImpute, LATE, SAVER, MAGIC, ScImpute, DrImpute, EnImpute, ALRA, VIPER, scDoc, scIGAN, and scGNN. The results show that NISC outperforms the existing imputation methods as it can recover the gene expression more correctly and distinguish the cell types more precisely, particularly for scRNA-seq data with high sparsity/noise.</p>
</sec>
<sec id="s2">
<title>2 Methods</title>
<sec id="s2-1">
<title>2.1 Neural Network Architecture</title>
<p>It is evident that the process of imputing the dropouts for scRNA-seq data is similar to the process of outlining a noisy image, so autoencoder is utilized to impute the sparse scRNA-seq data (<xref ref-type="bibr" rid="B38">Shao et al., 2013</xref>). Autoencoder is an unsupervised learning technique that has been used in image denoising (<xref ref-type="bibr" rid="B50">Vincent et al., 2010</xref>). The autoencoder technique allows nonlinear data vectors to be stacked, making the technique more powerful and able to learn complicated relations between layers (<xref ref-type="bibr" rid="B28">Mao et al., 2016</xref>). An autoencoder model consists of an encoder and a decoder. An encoder stage compresses the input data into a low-dimensional code, and then a similar decoder stage reconstructs the output data from the code (<xref ref-type="bibr" rid="B17">Hinton and Salakhutdinov, 2006</xref>). <xref ref-type="fig" rid="F1">Figure 1</xref> shows the neural network architecture of NISC. The number of neurons for the hidden layer in the middle is usually much smaller than the number of neurons for the input/output layers to reduce the redundant information in data. In our method NISC, the number of neurons in the neural network architecture is set to be proportional to the number of genes.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Neural network architecture of NISC. This network is mainly composed of three hidden layers (i.e., dots with three different colors, purple, red, and green). The first hidden layer has neurons equal to twice the number of genes of the input data. It is followed by the second hidden layer in the middle with neurons equal to around half the number of genes of the input data. The third layer has neurons equal to the number of neurons of the first layer. This neural network is trained using an optimization process with a loss function to calculate the model error.</p>
</caption>
<graphic xlink:href="fgene-13-847112-g001.tif"/>
</fig>
</sec>
<sec id="s2-2">
<title>2.2 Loss Function and Regularizations</title>
<p>It has been found that the main reason for dropouts in scRNA-seq data is due to failure of the reverse transcription of mRNA (<xref ref-type="bibr" rid="B7">Bengtsson et al., 2005</xref>; <xref ref-type="bibr" rid="B35">Reiter et al., 2011</xref>). Reverse transcription is an enzyme reaction; therefore, the Michaelis-Menten function can be used to model the relationship between dropout probability and gene expression for full-transcripts scRNA-seq data (<xref ref-type="bibr" rid="B1">Andrews and Hemberg, 2019</xref>). The following equation shows the dropout probability <italic>P</italic>
<sub>
<italic>ij</italic>
</sub> for the gene <italic>i</italic> in cell <italic>j</italic> using Michaelis-Menten kinetics (MMK) (<xref ref-type="bibr" rid="B9">Brennecke et al., 2013</xref>),<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>K</mml:mi>
<mml:mi>M</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <italic>S</italic>
<sub>
<italic>ij</italic>
</sub> is the observed gene expression level of gene <italic>i</italic> in cell <italic>j</italic>, and <italic>K</italic>
<sub>
<italic>M</italic>
</sub> is the Michaelis constant (<xref ref-type="bibr" rid="B20">Johnson and Goody, 2011</xref>). We use this probability to describe the dropout event, which will then be involved in the calculating the network&#x2019;s denoised output.</p>
<p>We propose a novel loss function with the mean square error weighted by the dropout probability estimated through Michaelis-Menten kinetics.<disp-formula id="e2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:mstyle>
<mml:mo>&#x22c5;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>log</mml:mtext>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>log</mml:mtext>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
</p>
<p>The loss function will be minimized through the autoencoder learning process. Note: the function &#x201c;log&#x201d; is the natural logarithm. The intuition behind this is that the estimated dropout probability <inline-formula id="inf1">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> affects the loss function adversely. In this manner, the imputed gene expression <inline-formula id="inf2">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> will be close to the observed gene expression <inline-formula id="inf3">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> when the estimated dropout probability <inline-formula id="inf4">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is low. When we train an autoencoder network, a challenging problem is how to avoid overfitting. Overfitting refers to a neural network model that fits the training data too well to predict the pattern of new data. Overfitting is caused by noise in the training data, and the neural network includes this noise during the learning process. To avoid overfitting, we need to reduce the complexity of the network; therefore, we applied <italic>L</italic>
<sub>
<italic>2</italic>
</sub> regularization (ridge regression) and dropout regularization to reduce the complexity of the autoencoder network (note: this is different from the term &#x201c;dropout&#x201d; event in scRNA-seq data). It is the first time that these two regularization techniques have been combined with an autoencoder network for imputation of scRNA-seq data. We define the regularization term <inline-formula id="inf5">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> as the <italic>L</italic>
<sub>
<italic>2</italic>
</sub> norm of the weight matrix, that is, the sum of all squared weight values of the matrix (i.e., the first term in the above loss function). <italic>&#x3b1;</italic> is defined as the value of the regularization rate, which determines how powerful the effect of the regularization term will be. The regularization term <inline-formula id="inf6">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is weighted by the scalar <italic>&#x3b1;</italic> and the regularization term will be excluded if <italic>&#x3b1;</italic> is zero. If <italic>&#x3b1;</italic> is too large, the neural network model will be less sensitive therefore increase the risk of underfitting. Conversely, if <italic>&#x3b1;</italic> is too small, the complexity of the model will be increased, so the risk of overfitting will be high. An appropriate value of <italic>&#x3b1;</italic> can be determined through cross-validation suggested by <xref ref-type="bibr" rid="B32">Ng et al.(2004)</xref>.</p>
<p>In addition to <italic>L</italic>
<sub>
<italic>2</italic>
</sub> regularization, dropout regularization is also used in NISC as it is a strategy to turn off neurons of the neural network with certain probability during training, which then further reduces the model&#x2019;s complexity (<xref ref-type="bibr" rid="B41">Srivastava et al., 2014</xref>). Furthermore, to mitigate the effect of reaching the local optimization peak by the neural network, the Adaptive Moment estimation algorithm is used to perform stochastic optimization (<xref ref-type="bibr" rid="B13">Eweda and Macchi, 1984</xref>).</p>
</sec>
<sec id="s2-3">
<title>2.3 Performance Evaluation</title>
<p>The proposed method is compared with the existing imputation methods through a series of simulated datasets and three real datasets. First, we visualize cell type sub-populations using 2-dimensional PCA (principal component analysis) plots or t-SNE (t-distributed stochastic neighbor embedding) plots (<xref ref-type="bibr" rid="B21">Kin et al., 2002</xref>; <xref ref-type="bibr" rid="B22">Kobak and Berens, 2019</xref>) depending on the data property (<xref ref-type="bibr" rid="B3">Anowar et al., 2021</xref>). UMAP (uniform manifold approximation) plots are also drawn (<xref ref-type="bibr" rid="B6">Becht et al., 2019</xref>). The commonly used unsupervised clustering algorithms, k-means (<xref ref-type="bibr" rid="B30">Na et al., 2010</xref>) and hierarchical clustering algorithms (<xref ref-type="bibr" rid="B29">Murtagh and Contreras, 2017</xref>), and Leiden algorithm(<xref ref-type="bibr" rid="B46">Traag et al., 2019</xref>), are used to group the cells on the reduced dimension of visualization results, which can then be used for calculating the performance measurements of each imputation method.</p>
<p>Four evaluation metrics are calculated to evaluate the accuracy of the cell type clusters in the visualization plots, including Adjusted Mutual Information (AMI) (<xref ref-type="bibr" rid="B36">Romano et al., 2014</xref>), Adjusted Rand Index (ARI) (<xref ref-type="bibr" rid="B42">Steinley, 2004</xref>), Fowlkes-Mallows Index (FMI) (<xref ref-type="bibr" rid="B31">Nemec and Brinkhurst, 1988</xref>), and Silhouette Score (SS) (<xref ref-type="bibr" rid="B37">Rousseeuw, 1987</xref>). Since we know the truth for the simulated data, the RMSE (Root Mean Square Error) is also calculated between the imputed values and the truth to assess the performance of imputation methods (<xref ref-type="bibr" rid="B8">Blondel et al., 2008</xref>; <xref ref-type="bibr" rid="B39">Skinnider et al., 2019</xref>). Additionally, the heatmap of gene expression in the simulated studies is also drawn to demonstrate the direct comparison of the methods in detail.</p>
</sec>
</sec>
<sec id="s3">
<title>3 Results</title>
<sec id="s3-1">
<title>3.1 NISC Enhances Cell Type Visualization in Simulated scRNA-Seq Data</title>
<p>To evaluate the performance of our imputation method, we compare it with existing methods on simulated scRNA-seq count data, which are generated by the widely used simulator, Splatter (<xref ref-type="bibr" rid="B55">Zappia et al., 2017</xref>). Both raw count data with dropouts/noise and its corresponding true data are available through simulations. The raw count data is the input data of the learning framework, and the ground truth data can be used to assess the performance of imputation. The count data are represented as an expression matrix, where each row is a gene, and each column is a cell. We consider three scenarios:<list list-type="simple">
<list-item>
<p>(1) Two cell types for 800 genes and 1,000 cells.</p>
</list-item>
<list-item>
<p>(2) Four cell types for 800 genes and 1,000 cells</p>
</list-item>
<list-item>
<p>(3) Four cell types for 2,000 genes and 10,000 cells</p>
</list-item>
</list>
</p>
<p>For each scenario, two sparsity levels are examined, i.e., approximately 80 vs 90%. In the Splatter simulation setting, the differential rate of 0.2 is used, indicating that 20% of the total genes are marker genes. As substantial noise is added to input data to mask cell type identities through simulation, our purpose is to predict the imputed values for the dropouts accurately and therefore identify cell types.</p>
<p>Our deep learning framework in NISC consists of three hidden layers with 1600, 400, and 1600 neurons, respectively, for the simulation data of 800 genes. For the case of 2,000 genes, the number of neurons for three hidden layers are 4,000, 1,000, and 4,000, respectively. A widely used active function, rectified linear unit (<xref ref-type="bibr" rid="B53">Xing et al., 2016</xref>), is employed to train each cell to capture the nonlinearity of the data. The number of neurons for the encoder/decoder layers is twice the number of genes, while the number of neurons for the hidden layer in the middle of the architecture is half of the number of genes. We compare NISC to other existing imputation methods in simulation data for various scenarios. The figures below are for the scenario (2). Some representative results for scenario 1) and 3) are included in the <xref ref-type="sec" rid="s10">Supplementary File</xref>.</p>
<p>
<xref ref-type="fig" rid="F2">Figure 2</xref> shows the t-SNE plots derived from the ground truth of cells, the raw input data, and the imputed data by NISC and other existing methods. The ground truth contains 4&#xa0;cell types while the types are mixed in the raw data. This is due to the high sparsity (i.e., high noise, 90% data are zeros) in the raw input, which distorts the topology of the ground truth. NISC can accurately recover the dropouts, and the cells are clearly located in four groups/clusters, followed by scDoc and DeepImpute. However, it is challenging for other imputation approaches to distinguish the 4&#xa0;cell types.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>NISC significantly improves the performance of t-SNE in visualizing simulated scRNA-seq count data. Plots of the first two components are calculated from the simulated ground truth data, raw data, and imputed data using various imputation methods. The dataset contains 800 genes and 1,000 cells in 4&#xa0;cell types, with 90% sparsity. Cells are colored by cell types as indicated.</p>
</caption>
<graphic xlink:href="fgene-13-847112-g002.tif"/>
</fig>
<p>Four evaluation metrics, including AMI, ARI, FMI, and SS are calculated on the visualization result for the simulated data in <xref ref-type="fig" rid="F2">Figure 2</xref>. To consider the data uncertainty (even with the same parameter settings) in the simulation, we generated ten replicates of datasets under each setting. <xref ref-type="fig" rid="F3">Figure 3</xref> shows boxplots for four evaluation measures based on K-means clustering result of the t-SNE visualization. The boxplots of Leiden method are shown in <xref ref-type="sec" rid="s10">Supplementary Figure S2</xref>. Higher values in measures indicate higher accuracy in cluster results. It is obvious that the performance of NISC surpasses all the existing imputation methods in clustering accuracy in this simulation study.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Boxplots of four evaluation measures, including Adjusted Mutual Information (AMI), Adjusted Rand Index (ARI), Fowlkes-Mallows Index (FMI), and Silhouette Score (SS), are calculated for comparing NISC and other imputation methods. Each dataset contains 800 genes and 1,000 cells in 4&#xa0;cell types, with 90% sparsity, and is replicated 10 times. Detailed information about these measurements can be found in the supplementary materials.</p>
</caption>
<graphic xlink:href="fgene-13-847112-g003.tif"/>
</fig>
<p>High accuracy in cell type visualization does not necessarily mean the imputed values are close to the true values. We calculated RMSE (root mean square error, the detailed definition can be found in the supplementary materials) between the ground truth value and the corresponding imputed value by each method. <xref ref-type="fig" rid="F4">Figure 4</xref> shows boxplots of RMSE for 10 replicates of simulations. Compared with other imputation methods, the accuracy of NISC is highest, followed by DeepImpute, which is a neural network-based imputation method as well. Note: three imputation methods, DCA, AutoImpute and scGNN, are excluded from the RMSE plot as only highly variable genes are selected in these methods to perform imputation.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>RMSE (root mean square error) boxplots for the raw input and imputed data by each method. The RMSE is calculated between the ground truth and either the raw or imputed values. The raw dataset contains 800 genes and 1,000 cells in 4&#xa0;cell types, with 90% sparsity, and is replicated 10 times. Detailed information about the RMSE can be found in the supplementary materials.</p>
</caption>
<graphic xlink:href="fgene-13-847112-g004.tif"/>
</fig>
<p>A direct comparison in gene expression values among the ground truth, raw data, and imputed data can be found in the heatmap plot (<xref ref-type="fig" rid="F5">Figure 5</xref>). It shows that NISC imputed values are closest to the ground truth and therefore this method shows great capability in correcting the dropout values, which confirms the promising result in data visualization in <xref ref-type="fig" rid="F2">Figure 2</xref>. Again, three imputation methods DCA, AutoImpute, and scGNN, are excluded from the heatmap plot as only highly variable genes are selected in these methods to perform imputation.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Heatmaps of ground truth, raw simulated data, and imputed data by various methods. The simulated raw dataset contains 800 genes and 1,000 cells in 4&#xa0;cell types, with 90% sparsity. Each row in the heatmap represents a gene, while each column represents a cell. The color bar shows the magnitude of the logarithm of gene expression values.</p>
</caption>
<graphic xlink:href="fgene-13-847112-g005.tif"/>
</fig>
<p>A consistent conclusion can be obtained from UMAP plot (<xref ref-type="sec" rid="s10">Supplementary Figure S3</xref>) for this dataset. We also examine the impact of a different sparsity level (80%) on the imputation for the simulated data with 4&#xa0;cell types and 2&#xa0;cell types, respectively. When the sparsity of the simulated data with 4&#xa0;cell types is about 80%, the cell populations can be revealed clearly in several imputation methods (<xref ref-type="sec" rid="s10">Supplementary Figure S4</xref>), and NISC is one of them. Then, we observe that the performance of all methods significantly decreases when dropout noise increases (<xref ref-type="sec" rid="s10">Supplementary Figure S4</xref> vs <xref ref-type="fig" rid="F2">Figure 2</xref>). A consistent conclusion can be obtained for the 2&#xa0;cell types. <xref ref-type="sec" rid="s10">Supplementary Figure S5</xref> shows an example of the t-SNE plot of 1,000 cells (in 2&#xa0;cell types) and 800 genes with 80% sparsity. The cells are clearly separated into two groups/clusters by NISC, DeepImpute, DrImpute, EnImpute and scDoc, followed by scImpute, SAVER, and scIGAN.</p>
<p>For the case of 4&#xa0;cell types with 10,000 cells, we only compared the deep-learning-based methods (<xref ref-type="sec" rid="s10">Supplementary Figure S6</xref>). We noticed that the performances of three methods, NISC, DCA, and DeepImpute, are improved when the number of cells increases from 1,000 (<xref ref-type="fig" rid="F2">Figure 2</xref>) to 10,000 (<xref ref-type="sec" rid="s10">Supplementary Figure S6</xref>). The t-SNE plot in <xref ref-type="sec" rid="s10">Supplementary Figure S6</xref> still shows that NISC surpasses other deep-learning-based methods, followed by DCA and DeepImpute.</p>
<p>
<bold>Computational time</bold>: Among the deep-learning-based methods, LATE is the fastest, and scIGAN is the slowest. Specifically, the order of the computational time for seven deep learning-based methods is: LATE &#x3c; DeepImpute &#x3c; DCA &#x3c; NISC &#x3c; AutoImpute &#x3c; scGNN &#x3c; scIGAN. We used High Performance Computer systems with 2894&#xa0;MHz CPU, 5 cores, and 36&#xa0;GB memory on each core. For a simulation dataset with 2,000 genes and 10,000 cells, it took about 10&#xa0;min for LATE, 12&#xa0;h scIGAN, and 50&#xa0;min for NISC.</p>
</sec>
<sec id="s3-2">
<title>3.2 NISC Improves Visualization Clarity and Clustering Accuracy in Real scRNA-Seq Data</title>
<sec id="s3-2-1">
<title>3.2.1 Mouse Lung scRNA-Seq Data</title>
<p>We apply NISC and the compared methods on mouse lung scRNA-seq data (GSE52583) with 201 cells (<xref ref-type="bibr" rid="B48">Treutlein et al., 2014</xref>). <xref ref-type="fig" rid="F6">Figure 6</xref> shows PCA plots for NISC and other imputation methods. The denoised data by imputation of scGNN, AutoImpute, ALRA, SAVER, scImpute, DrImpute, scDoc and EnImpute show E14.5 and E16.5 are not separated well, although cell type AT2 and E18.5 can be identified. In addition, with imputation of MAGIC, E16.5 is successfully identified, but E18.5, E14.5, and AT2 are mixed. By DCA, the 4&#xa0;cell types (E14.5, E16.5, E18.5, and AT2) are grouped into two clusters, with two types in each. For DeepImpute, scIGAN and VIPER, the 4&#xa0;cell types are mixed together. It seems that NISC can assign the 4&#xa0;cell types into four clusters more accurately.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>NISC recovers the cell types (E14.5, E16.5, E18.5, and AT2) in mouse lung data. PCA plots of the raw data and imputed data by various imputation methods. The sparsity of the data is 72.6%. Cells are colored by cell types, which are reported in the original publication.</p>
</caption>
<graphic xlink:href="fgene-13-847112-g006.tif"/>
</fig>
<p>The evaluation matrices on the clustering for this dataset are also calculated (<xref ref-type="fig" rid="F7">Figure 7</xref>). Though NISC result does not provide the tightest clusters (from Silhouette score), among all the imputation methods, it scores the highest consistently across three measures of clustering accuracy, which confirms the separation pattern in the visualization in <xref ref-type="fig" rid="F6">Figure 6</xref>.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Evaluation of clustering accuracy on the mouse lung data. Four measurements, AMI, ARI, FMI, and SS, are calculated for the imputed and raw data. The definitions of the measurements can be found in the supplementary materials.</p>
</caption>
<graphic xlink:href="fgene-13-847112-g007.tif"/>
</fig>
</sec>
<sec id="s3-2-2">
<title>3.2.2 Mouse Embryonic Data</title>
<p>We also apply NISC and the compared methods on scRNA-seq data of 92 mouse embryonic cells and 22,936 genes (GSE29087). The sparsity of the data is 83.04%. The cell types of this data set are reported in the original publication (<xref ref-type="bibr" rid="B19">Islam et al., 2011</xref>). We visualize the clustering result with t-SNE plots (<xref ref-type="sec" rid="s10">Supplementary Figure S7</xref>), illustrating that, through NISC imputation, the 2&#xa0;cell types, 48 mouse embryonic stem cells (ES) and 44 mouse embryonic fibroblasts (MEF), are separated, followed by DrImpute, DCA and scGNN. Through imputation of scIGAN, AutoImpute, LATE, ALRA, SAVER, MAGIC, EnImpute, SAVER, VIPER, and scDoc, the 2&#xa0;cell types in this data are not separated well. With imputation of scGNN, DCA, and DrImpute, the 2&#xa0;cell types are only somewhat separated. With scImpute, the cells are isolated into many tighter subclusters. In other words, some cells which should belong to the same cell type are scattered. The accuracy of clustering is assessed by four evaluation measures. Though NISC result does not provide the tightest clusters (from Silhouette score), among all the methods compared here, NISC is superior to others in terms of cluster accuracy ARI, AMI, and FMI. It improves the cluster results on original raw data.</p>
</sec>
<sec id="s3-2-3">
<title>3.2.3 Human Lung Adenocarcinoma Data</title>
<p>The above real scRNA-seq datasets do not have ground truth, since usually it is challenging to obtain the ground truth for real scRNA-seq data. Alternatively, it will be convincing to evaluate the performance of the imputation approaches if we use a real scRNA-seq dataset with low sparsity and distinct cell types and set it to be the ground truth data for evaluations. For this purpose, we apply the imputation methods on lung adenocarcinoma data (GSE69405) that profiles the gene expression of single cancer cells with TPM (normalization by transcripts per million) measurements (<xref ref-type="bibr" rid="B40">Soneson and Robinson, 2018</xref>). These cancer cells are originally from lung adenocarcinoma patient-derived xenograft (PDX) tumors, including four types, H358 human lung cancer cells (H358), cancer cells in PDX from primary tumors (LC-PT-45), an additional batch of PDX cells (LC-Pt-45-Re), and PDX cells for another lung cancer case (LC-MBT-15). This data set contains 176 cells, and the sparsity of the data is relatively low (46%). The cell types in this data can be clearly identified in the original data without imputation (<xref ref-type="fig" rid="F8">Figure 8A</xref>). Therefore, we set the original data to be the ground truth. Following the method in (<xref ref-type="bibr" rid="B4">Arisdakessian et al., 2019</xref>) to generate noisy data, similarly, we mask the low-noise data by randomly changing some non-zeros to zeros so that the sparsity of the data is increased to 80% and the synthesized dataset here is termed as raw data.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>NISC recovers the cell types in lung adenocarcinoma data (GSE69405) <bold>(A)</bold> plots of t-SNE components 1 and 2 derived from raw data, imputed data using NISC and other imputation methods. With additional zeros the sparsity of the data is 80%. Cells are colored by cell types, which are reported in the original publication <bold>(B)</bold> Bar plots of evaluation of cluster accuracy on the raw and imputed data. Four measurements, AMI, ARI, FMI, and SS, are calculated for the imputed and raw data. The definitions of the measurements can be found in the supplementary.</p>
</caption>
<graphic xlink:href="fgene-13-847112-g008.tif"/>
</fig>
<p>T-SNE plots (<xref ref-type="fig" rid="F8">Figure 8A</xref>) of the synthesized data show that NISC successfully recovers the cell types of the original data through imputing the sparse raw data. However, other imputation methods result in either one big cluster (i.e., all cells are mixed together) or several tight clusters, but each with two or more different cell types. A consistent conclusion can be obtained in evaluation plots (<xref ref-type="fig" rid="F8">Figure 8B</xref>). Though the cells are not separately into tight clusters in NISC data, this method results in the highest cluster accuracy, considering the actual cell type status.</p>
</sec>
</sec>
</sec>
<sec id="s4">
<title>4 Discussion</title>
<p>NISC is a data-driven method and does not require any prior knowledge. Real data and simulated data show that NISC can impute the dropouts in the scRNA-seq data, improving the accuracy of cell type clustering. Four performance measures were calculated to evaluate the clustering accuracy for the imputed data by various imputation methods. RMSE, which measures the distance between true (if available) and imputed values, was also calculated. Generally, compared with other existing estimation methods, NISC has a lower RMSE and a higher score in the evaluation measures of clustering accuracy.</p>
<p>NISC is an unsupervised neural network-based imputation method with autoencoder techniques implemented. Compared with other neural network-based methods, we investigated how different loss functions affect the imputation results. We developed a novel loss function weighted by Michaelis-Menten kinetics (MMK) and investigated its difference and standard mean square error (MSE) loss. Fig. S1 shows that the MMK loss can achieve more effective imputation under the sparse simulation setting, while by regular MSE the loss function is less effective. In addition, we add L2 regularization and dropout regularization to the model (<xref ref-type="bibr" rid="B11">Cortes et al., 2012</xref>) to avoid overfitting when denoising the input data. This is the first time the two regularizations are implemented simultaneously in the autoencoder model to impute scRNA-seq data.</p>
<p>An effective neural network for imputation requires sufficient neurons in the network. Due to many genes in scRNA-seq studies, GPUs are recommended for NISC to speed up the training process of the autoencoder network. NISC imputation is not suitable for some types of data which lose Michaelis-Menten kinetics, such as 10x Genomics data (<xref ref-type="bibr" rid="B1">Andrews and Hemberg, 2019</xref>), and some normalized data, for example, RPKM (Reads per kilo base per million mapped reads) or FPKM (Fragments Per Kilobase Million) (<xref ref-type="bibr" rid="B27">Lytal et al., 2020</xref>). However, TPM normalization is applicable as it maintains the data structure of the original gene expressions (<xref ref-type="bibr" rid="B23">Li and Li, 2018</xref>).</p>
</sec>
</body>
<back>
<sec id="s5">
<title>Data Availability Statement</title>
<p>The original contributions presented in the study are publicly available. This data can be found here: <ext-link ext-link-type="uri" xlink:href="http://github.com/anlingUA/NISC">github.com/anlingUA/NISC</ext-link>.</p>
</sec>
<sec id="s6">
<title>Author Contributions</title>
<p>LA and XZ conceived the study. XZ and SC designed the methods and algorithms. XZ, ZC, RB, ML and NL performed the simulation studies. XZ, ZC, RB, YC and LA contributed to the real data analyses. XZ and LA drafted the manuscript. All authors revised, proofread, and approved the submitted manuscript.</p>
</sec>
<sec id="s7">
<title>Funding</title>
<p>This work has been partially supported by the National Institute of Health (1R01GM139829-01; 1P01AI148104-01A1; U19AG065169; 5P01AG052359-05) and the United States Department of Agriculture (ARZT-1361620-H22-149) to LA and by the National Institute of Health (R01AI149754 and R01ES027013) to YC.</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s10">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2022.847112/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2022.847112/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.PDF" id="SM1" mimetype="application/PDF" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Andrews</surname>
<given-names>T. S.</given-names>
</name>
<name>
<surname>Hemberg</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>M3Drop: Dropout-Based Feature Selection for scRNASeq</article-title>. <source>Bioinformatics</source> <volume>35</volume> (<issue>16</issue>), <fpage>2865</fpage>&#x2013;<lpage>2867</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty1044</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Angerer</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Simon</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Tritschler</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wolf</surname>
<given-names>F. A.</given-names>
</name>
<name>
<surname>Fischer</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Theis</surname>
<given-names>F. J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Single Cells Make Big Data: New Challenges and Opportunities in Transcriptomics</article-title>. <source>Curr. Opin. Syst. Biol.</source> <volume>4</volume>, <fpage>85</fpage>&#x2013;<lpage>91</lpage>. <pub-id pub-id-type="doi">10.1016/j.coisb.2017.07.004</pub-id> </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Anowar</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Sadaoui</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Selim</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Conceptual and empirical comparison of dimensionality reduction algorithms (pca, kpca, lda, mds, svd, lle, isomap, le, ica, t-sne)</article-title>. <source>Comp. Sci. Rev.</source> <volume>40</volume>, <fpage>100378</fpage>. <pub-id pub-id-type="doi">10.1016/j.cosrev.2021.100378</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arisdakessian</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Poirion</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Yunits</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Garmire</surname>
<given-names>L. X.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>DeepImpute: an Accurate, Fast, and Scalable Deep Neural Network Method to Impute Single-Cell RNA-Seq Data</article-title>. <source>Genome Biol.</source> <volume>20</volume> (<issue>1</issue>), <fpage>211</fpage>&#x2013;<lpage>214</lpage>. <pub-id pub-id-type="doi">10.1186/s13059-019-1837-6</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Badsha</surname>
<given-names>M. B.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y. I.</given-names>
</name>
<name>
<surname>Xian</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Banovich</surname>
<given-names>N. E.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Imputation of Single-Cell Gene Expression with an Autoencoder Neural Network</article-title>. <source>Quant Biol.</source> <volume>8</volume> (<issue>1</issue>), <fpage>78</fpage>&#x2013;<lpage>94</lpage>. <pub-id pub-id-type="doi">10.1007/s40484-019-0192-7</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Becht</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>McInnes</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Healy</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Dutertre</surname>
<given-names>C.-A.</given-names>
</name>
<name>
<surname>Kwok</surname>
<given-names>I. W. H.</given-names>
</name>
<name>
<surname>Ng</surname>
<given-names>L. G.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Dimensionality Reduction for Visualizing Single-Cell Data Using UMAP</article-title>. <source>Nat. Biotechnol.</source> <volume>37</volume> (<issue>1</issue>), <fpage>38</fpage>&#x2013;<lpage>44</lpage>. <pub-id pub-id-type="doi">10.1038/nbt.4314</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bengtsson</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>St&#xe5;hlberg</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Rorsman</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Kubista</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Gene Expression Profiling in Single Cells from the Pancreatic Islets of Langerhans Reveals Lognormal Distribution of mRNA Levels</article-title>. <source>Genome Res.</source> <volume>15</volume> (<issue>10</issue>), <fpage>1388</fpage>&#x2013;<lpage>1392</lpage>. <pub-id pub-id-type="doi">10.1101/gr.3820805</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Blondel</surname>
<given-names>V. D.</given-names>
</name>
<name>
<surname>Guillaume</surname>
<given-names>J.-L.</given-names>
</name>
<name>
<surname>Lambiotte</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Lefebvre</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Fast Unfolding of Communities in Large Networks</article-title>. <source>J. Stat. Mech.</source> <volume>2008</volume> (<issue>10</issue>), <fpage>P10008</fpage>. <pub-id pub-id-type="doi">10.1088/1742-5468/2008/10/p10008</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brennecke</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Anders</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>J. K.</given-names>
</name>
<name>
<surname>Ko&#x142;odziejczyk</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Proserpio</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Accounting for Technical Noise in Single-Cell RNA-Seq Experiments</article-title>. <source>Nat. Methods</source> <volume>10</volume> (<issue>11</issue>), <fpage>1093</fpage>&#x2013;<lpage>1095</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.2645</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>VIPER: Variability-Preserving Imputation for Accurate Gene Expression Recovery in Single-Cell RNA Sequencing Studies</article-title>. <source>Genome Biol.</source> <volume>19</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>15</lpage>. <pub-id pub-id-type="doi">10.1186/s13059-018-1575-1</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Cortes</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Mohri</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Rostamizadeh</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2012</year>). <source>L2 Regularization for Learning Kernels</source>. <comment>arXiv preprint arXiv:1205.2653</comment>. </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Eraslan</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Simon</surname>
<given-names>L. M.</given-names>
</name>
<name>
<surname>Mircea</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mueller</surname>
<given-names>N. S.</given-names>
</name>
<name>
<surname>Theis</surname>
<given-names>F. J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Single-cell RNA-Seq Denoising Using a Deep Count Autoencoder</article-title>. <source>Nat. Commun.</source> <volume>10</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.1038/s41467-018-07931-2</pub-id> </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Eweda</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Macchi</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>1984</year>). <article-title>Convergence of an Adaptive Linear Estimation Algorithm</article-title>. <source>IEEE Trans. Automat. Contr.</source> <volume>29</volume> (<issue>2</issue>), <fpage>119</fpage>&#x2013;<lpage>127</lpage>. <pub-id pub-id-type="doi">10.1109/tac.1984.1103463</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gong</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Kwak</surname>
<given-names>I. Y.</given-names>
</name>
<name>
<surname>Pota</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Koyano-Nakagawa</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Garry</surname>
<given-names>D. J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>DrImpute: Imputing Dropout Events in Single Cell RNA Sequencing Data</article-title>. <source>BMC bioinformatics</source> <volume>19</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-018-2226-y</pub-id> </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gordon</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Satory</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Halliday</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Herman</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Lost in Transcription: Transient Errors in Information Transfer</article-title>. <source>Curr. Opin. Microbiol.</source> <volume>24</volume>, <fpage>80</fpage>&#x2013;<lpage>87</lpage>. <pub-id pub-id-type="doi">10.1016/j.mib.2015.01.010</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gr&#xf8;nbech</surname>
<given-names>C. H.</given-names>
</name>
<name>
<surname>Vording</surname>
<given-names>M. F.</given-names>
</name>
<name>
<surname>Timshel</surname>
<given-names>P. N.</given-names>
</name>
<name>
<surname>S&#xf8;nderby</surname>
<given-names>C. K.</given-names>
</name>
<name>
<surname>Pers</surname>
<given-names>T. H.</given-names>
</name>
<name>
<surname>Winther</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>scVAE: Variational Auto-Encoders for Single-Cell Gene Expression Data</article-title>. <source>Bioinformatics</source> <volume>36</volume> (<issue>16</issue>), <fpage>4415</fpage>&#x2013;<lpage>4422</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa293</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hinton</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Salakhutdinov</surname>
<given-names>R. R.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Reducing the Dimensionality of Data with Neural Networks</article-title>. <source>Science</source> <volume>313</volume> (<issue>5786</issue>), <fpage>504</fpage>&#x2013;<lpage>507</lpage>. <pub-id pub-id-type="doi">10.1126/science.1127647</pub-id> </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Torre</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Dueck</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Shaffer</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bonasio</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>SAVER: Gene Expression Recovery for Single-Cell RNA Sequencing</article-title>. <source>Nat. Methods</source> <volume>15</volume> (<issue>7</issue>), <fpage>539</fpage>&#x2013;<lpage>542</lpage>. <pub-id pub-id-type="doi">10.1038/s41592-018-0033-z</pub-id> </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Islam</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kj&#xe4;llquist</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Moliner</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zajac</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>J.-B.</given-names>
</name>
<name>
<surname>L&#xf6;nnerberg</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Characterization of the Single-Cell Transcriptional Landscape by Highly Multiplex RNA-Seq</article-title>. <source>Genome Res.</source> <volume>21</volume> (<issue>7</issue>), <fpage>1160</fpage>&#x2013;<lpage>1167</lpage>. <pub-id pub-id-type="doi">10.1101/gr.110882.110</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Johnson</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Goody</surname>
<given-names>R. S.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>The Original Michaelis Constant: Translation of the 1913 Michaelis-Menten Paper</article-title>. <source>Biochemistry</source> <volume>50</volume> (<issue>39</issue>), <fpage>8264</fpage>&#x2013;<lpage>8269</lpage>. <pub-id pub-id-type="doi">10.1021/bi201284u</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kin</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Tsuda</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Asai</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Marginalized Kernels for RNA Sequence Data Analysis</article-title>. <source>Genome Inform.</source> <volume>13</volume>, <fpage>112</fpage>&#x2013;<lpage>122</lpage>. </citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kobak</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Berens</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>The Art of Using T-SNE for Single-Cell Transcriptomics</article-title>. <source>Nat. Commun.</source> <volume>10</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.1038/s41467-019-13056-x</pub-id> </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>W. V.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J. J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>An Accurate and Robust Imputation Method scImpute for Single-Cell RNA-Seq Data</article-title>. <source>Nat. Commun.</source> <volume>9</volume> (<issue>1</issue>), <fpage>997</fpage>&#x2013;<lpage>999</lpage>. <pub-id pub-id-type="doi">10.1038/s41467-018-03405-7</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Troup</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ho</surname>
<given-names>J. W.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>CIDR: Ultrafast and Accurate Clustering through Imputation for Single-Cell RNA-Seq Data</article-title>. <source>Genome Biol.</source> <volume>18</volume> (<issue>1</issue>), <fpage>59</fpage>&#x2013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1186/s13059-017-1188-0</pub-id> </citation>
</ref>
<ref id="B25">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Linderman</surname>
<given-names>G. C.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kluger</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2018</year>). <source>Zero-preserving Imputation of scRNA-Seq Data Using Low-Rank Approximation</source>. <comment>BioRxiv</comment>, <fpage>397588</fpage>. </citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lopez</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Regier</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cole</surname>
<given-names>M. B.</given-names>
</name>
<name>
<surname>Jordan</surname>
<given-names>M. I.</given-names>
</name>
<name>
<surname>Yosef</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Deep Generative Modeling for Single-Cell Transcriptomics</article-title>. <source>Nat. Methods</source> <volume>15</volume> (<issue>12</issue>), <fpage>1053</fpage>&#x2013;<lpage>1058</lpage>. <pub-id pub-id-type="doi">10.1038/s41592-018-0229-2</pub-id> </citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lytal</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Ran</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>An</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Normalization Methods on Single-Cell RNA-Seq Data: an Empirical Survey</article-title>. <source>Front. Genet.</source> <volume>11</volume>, <fpage>41</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2020.00041</pub-id> </citation>
</ref>
<ref id="B28">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Mao</surname>
<given-names>X. J.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y. B.</given-names>
</name>
</person-group> (<year>2016</year>). <source>Image Restoration Using Convolutional Auto-Encoders with Symmetric Skip Connections</source>. <comment>arXiv preprint arXiv:1606.08921</comment>. </citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Murtagh</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Contreras</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Algorithms for Hierarchical Clustering: an Overview, II</article-title>. <source>Wiley Interdiscip. Rev. Data Mining Knowledge Discov.</source> <volume>7</volume> (<issue>6</issue>), <fpage>e1219</fpage>. <pub-id pub-id-type="doi">10.1002/widm.1219</pub-id> </citation>
</ref>
<ref id="B30">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Na</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xumin</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yong</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2010</year>). &#x201c;<article-title>Research on K-Means Clustering Algorithm: An Improved K-Means Clustering Algorithm</article-title>,&#x201d; in <conf-name>Proceeding of the 2010 Third International Symposium on intelligent information technology and security informatics</conf-name>, <conf-loc>Jian, China</conf-loc>, <conf-date>April 2010</conf-date> (<publisher-name>IEEE</publisher-name>), <fpage>63</fpage>&#x2013;<lpage>67</lpage>. <pub-id pub-id-type="doi">10.1109/iitsi.2010.74</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nemec</surname>
<given-names>A. F. L.</given-names>
</name>
<name>
<surname>Brinkhurst</surname>
<given-names>R. O.</given-names>
</name>
</person-group> (<year>1988</year>). <article-title>The Fowlkes-Mallows Statistic and the Comparison of Two Independently Determined Dendrograms</article-title>. <source>Can. J. Fish. Aquat. Sci.</source> <volume>45</volume> (<issue>6</issue>), <fpage>971</fpage>&#x2013;<lpage>975</lpage>. <pub-id pub-id-type="doi">10.1139/f88-119</pub-id> </citation>
</ref>
<ref id="B32">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ng</surname>
<given-names>A. Y.</given-names>
</name>
</person-group> (<year>2004</year>). &#x201c;<article-title>Feature Selection, L 1 vs. L 2 Regularization, and Rotational Invariance</article-title>,&#x201d; in <conf-name>Proceedings of the twenty-first international conference on Machine learning</conf-name>, <conf-date>July 2004</conf-date>, <fpage>78</fpage>. </citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pierson</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Yau</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>ZIFA: Dimensionality Reduction for Zero-Inflated Single-Cell Gene Expression Analysis</article-title>. <source>Genome Biol.</source> <volume>16</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1186/s13059-015-0805-z</pub-id> </citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ran</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lytal</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>An</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>scDoc: Correcting Drop-Out Events in Single-Cell RNA-Seq Data</article-title>. <source>Bioinformatics</source> <volume>36</volume> (<issue>15</issue>), <fpage>4233</fpage>&#x2013;<lpage>4239</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa283</pub-id> </citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reiter</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kirchner</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Muller</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Holzhauer</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Mann</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Pfaffl</surname>
<given-names>M. W.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Quantification Noise in Single Cell Experiments</article-title>. <source>Nucleic Acids Res.</source> <volume>39</volume> (<issue>18</issue>), <fpage>e124</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkr505</pub-id> </citation>
</ref>
<ref id="B36">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Romano</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bailey</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Verspoor</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>Standardized Mutual Information for Clustering Comparisons: One Step Further in Adjustment for Chance</article-title>,&#x201d; in <conf-name>Proceedings of the International Conference on Machine Learning</conf-name>, <conf-date>Aug 2021</conf-date>, <fpage>1143</fpage>&#x2013;<lpage>1151</lpage>. </citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rousseeuw</surname>
<given-names>P. J.</given-names>
</name>
</person-group> (<year>1987</year>). <article-title>Silhouettes: a Graphical Aid to the Interpretation and Validation of Cluster Analysis</article-title>. <source>J. Comput. Appl. Math.</source> <volume>20</volume>, <fpage>53</fpage>&#x2013;<lpage>65</lpage>. <pub-id pub-id-type="doi">10.1016/0377-0427(87)90125-7</pub-id> </citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>From Heuristic Optimization to Dictionary Learning: A Review and Comprehensive Comparison of Image Denoising Algorithms</article-title>. <source>IEEE Trans. Cybern</source> <volume>44</volume> (<issue>7</issue>), <fpage>1001</fpage>&#x2013;<lpage>1013</lpage>. <pub-id pub-id-type="doi">10.1109/TCYB.2013.2278548</pub-id> </citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Skinnider</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Squair</surname>
<given-names>J. W.</given-names>
</name>
<name>
<surname>Foster</surname>
<given-names>L. J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Evaluating Measures of Association for Single-Cell Transcriptomics</article-title>. <source>Nat. Methods</source> <volume>16</volume> (<issue>5</issue>), <fpage>381</fpage>&#x2013;<lpage>386</lpage>. <pub-id pub-id-type="doi">10.1038/s41592-019-0372-4</pub-id> </citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Soneson</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Robinson</surname>
<given-names>M. D.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Bias, Robustness and Scalability in Single-Cell Differential Expression Analysis</article-title>. <source>Nat. Methods</source> <volume>15</volume> (<issue>4</issue>), <fpage>255</fpage>&#x2013;<lpage>261</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.4612</pub-id> </citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Srivastava</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Hinton</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Krizhevsky</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sutskever</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Salakhutdinov</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Dropout: a Simple Way to Prevent Neural Networks from Overfitting</article-title>. <source>J. machine Learn. Res.</source> <volume>15</volume> (<issue>1</issue>), <fpage>1929</fpage>&#x2013;<lpage>1958</lpage>. <pub-id pub-id-type="doi">10.5555/2627435.2670313</pub-id> </citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Steinley</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Properties of the Hubert-Arable Adjusted Rand Index</article-title>. <source>Psychol. Methods</source> <volume>9</volume> (<issue>3</issue>), <fpage>386</fpage>&#x2013;<lpage>396</lpage>. <pub-id pub-id-type="doi">10.1037/1082-989x.9.3.386</pub-id> </citation>
</ref>
<ref id="B43">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Shang</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Deep Generative Autoencoder for Low-Dimensional Embeding Extraction from Single-Cell RNAseq Data</article-title>,&#x201d; in <conf-name>Proceedings of the 2019 IEEE International Conference on Bioinformatics and Biomedicine (BIBM)</conf-name>, <conf-loc>San Diego, CA, USA</conf-loc>, <conf-date>Nov. 2019</conf-date> (<publisher-name>IEEE</publisher-name>), <fpage>1365</fpage>&#x2013;<lpage>1372</lpage>. <pub-id pub-id-type="doi">10.1109/bibm47256.2019.8983289</pub-id> </citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Talwar</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Mongia</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sengupta</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Majumdar</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>AutoImpute: Autoencoder Based Imputation of Single-Cell RNA-Seq Data</article-title>. <source>Sci. Rep.</source> <volume>8</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1038/s41598-018-34688-x</pub-id> </citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tangherloni</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ricciuti</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Besozzi</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Li&#xf2;</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Cvejic</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Analysis of Single-Cell RNA Sequencing Data Based on Autoencoders</article-title>. <source>BMC bioinformatics</source> <volume>22</volume> (<issue>1</issue>), <fpage>309</fpage>&#x2013;<lpage>327</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-021-04150-3</pub-id> </citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Traag</surname>
<given-names>V. A.</given-names>
</name>
<name>
<surname>Waltman</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Van Eck</surname>
<given-names>N. J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>From Louvain to Leiden: Guaranteeing Well-Connected Communities</article-title>. <source>Sci. Rep.</source> <volume>9</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1038/s41598-019-41695-z</pub-id> </citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tracy</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>G. C.</given-names>
</name>
<name>
<surname>Dries</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>RESCUE: Imputing Dropout Events in Single-Cell RNA-Sequencing Data</article-title>. <source>BMC bioinformatics</source> <volume>20</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-019-2977-0</pub-id> </citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Treutlein</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Brownfield</surname>
<given-names>D. G.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Neff</surname>
<given-names>N. F.</given-names>
</name>
<name>
<surname>Mantalas</surname>
<given-names>G. L.</given-names>
</name>
<name>
<surname>Espinoza</surname>
<given-names>F. H.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Reconstructing Lineage Hierarchies of the Distal Lung Epithelium Using Single-Cell RNA-Seq</article-title>. <source>Nature</source> <volume>509</volume> (<issue>7500</issue>), <fpage>371</fpage>&#x2013;<lpage>375</lpage>. <pub-id pub-id-type="doi">10.1038/nature13173</pub-id> </citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Van Dijk</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Sharma</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Nainys</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yim</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kathail</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Carr</surname>
<given-names>A. J.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Recovering Gene Interactions from Single-Cell Data Using Data Diffusion</article-title>. <source>Cell</source> <volume>174</volume> (<issue>3</issue>), <fpage>716</fpage>&#x2013;<lpage>729</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2018.05.061</pub-id> </citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vincent</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Larochelle</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Lajoie</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Manzagol</surname>
<given-names>P. A.</given-names>
</name>
<name>
<surname>Bottou</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Stacked Denoising Autoencoders: Learning Useful Representations in a Deep Network with a Local Denoising Criterion</article-title>. <source>J. machine Learn. Res.</source> <volume>11</volume> (<issue>12</issue>), <fpage>3371</fpage>&#x2013;<lpage>3408</lpage>. </citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>VASC: Dimension Reduction and Visualization of Single-Cell RNA-Seq Data by Deep Variational Autoencoder</article-title>. <source>Genomics, proteomics &#x26; bioinformatics</source> <volume>16</volume> (<issue>5</issue>), <fpage>320</fpage>&#x2013;<lpage>331</lpage>. <pub-id pub-id-type="doi">10.1016/j.gpb.2018.08.003</pub-id> </citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Gong</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>scGNN Is a Novel Graph Neural Network Framework for Single-Cell RNA-Seq Analyses</article-title>. <source>Nat. Commun.</source> <volume>12</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1038/s41467-021-22197-x</pub-id> </citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xing</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Stacked Denoise Autoencoder Based Feature Extraction and Classification for Hyperspectral Images</article-title>. <source>J. Sensors</source> <volume>2016</volume>, <fpage>1</fpage>&#x2013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1155/2016/3632943</pub-id> </citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>You</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>scIGANs: Single-Cell RNA-Seq Imputation Using Generative Adversarial Networks</article-title>. <source>Nucleic Acids Res.</source> <volume>48</volume> (<issue>15</issue>), <fpage>e85</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkaa506</pub-id> </citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zappia</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Phipson</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Oshlack</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Splatter: Simulation of Single-Cell RNA Sequencing Data</article-title>. <source>Genome Biol.</source> <volume>18</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>15</lpage>. <pub-id pub-id-type="doi">10.1186/s13059-017-1305-0</pub-id> </citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Comparison of Computational Methods for Imputing Single-Cell RNA-Sequencing Data</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinform.</source> <volume>17</volume> (<issue>2</issue>), <fpage>174</fpage>&#x2013;<lpage>389</lpage>. <pub-id pub-id-type="doi">10.1109/tcbb.2018.2848633</pub-id> </citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>X.-F.</given-names>
</name>
<name>
<surname>Ou-Yang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>X.-M.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>EnImpute: Imputing Dropout Events in Single-Cell RNA-Sequencing Data via Ensemble Learning</article-title>. <source>Bioinformatics</source> <volume>35</volume> (<issue>22</issue>), <fpage>4827</fpage>&#x2013;<lpage>4829</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz435</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>