<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Bioinform.</journal-id>
<journal-title>Frontiers in Bioinformatics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Bioinform.</abbrev-journal-title>
<issn pub-type="epub">2673-7647</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1340339</article-id>
<article-id pub-id-type="doi">10.3389/fbinf.2024.1340339</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Bioinformatics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Detecting subtle transcriptomic perturbations induced by lncRNAs knock-down in single-cell CRISPRi screening using a new sparse supervised autoencoder neural network</article-title>
<alt-title alt-title-type="left-running-head">Truchi et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fbinf.2024.1340339">10.3389/fbinf.2024.1340339</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Truchi</surname>
<given-names>Marin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lacoux</surname>
<given-names>Caroline</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Gille</surname>
<given-names>Cyprien</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2642121/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Fassy</surname>
<given-names>Julien</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Magnone</surname>
<given-names>Virginie</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lopes Goncalves</surname>
<given-names>Rafael</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Girard-Riboulleau</surname>
<given-names>C&#xe9;dric</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Manosalva-Pena</surname>
<given-names>Iris</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Gautier-Isola</surname>
<given-names>Marine</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lebrigand</surname>
<given-names>Kevin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Barbry</surname>
<given-names>Pascal</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/262151/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Spicuglia</surname>
<given-names>Salvatore</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/307113/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Vassaux</surname>
<given-names>Georges</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2582484/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Rezzonico</surname>
<given-names>Roger</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Barlaud</surname>
<given-names>Michel</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2582168/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Mari</surname>
<given-names>Bernard</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/24223/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Universit&#xe9; C&#xf4;te d&#x2019;Azur</institution>, <institution>IPMC</institution>, <institution>UMR CNRS 7275 Inserm 1323</institution>, <institution>IHU RespiERA</institution>, <addr-line>Valbonne</addr-line>, <country>France</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Universit&#xe9; C&#xf4;te d&#x2019;Azur</institution>, <institution>I3S</institution>, <institution>CNRS UMR7271</institution>, <addr-line>Nice</addr-line>, <country>France</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Aix-Marseille University</institution>, <institution>Inserm</institution>, <institution>TAGC, UMR1090, Equipe Lab&#x00E9;lis&#x00E9;e Ligue Contre le Cancer</institution>, <addr-line>Marseille</addr-line>, <country>France</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/149793/overview">Ahmed Mahfouz</ext-link>, Leiden University Medical Center (LUMC), Netherlands</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1803995/overview">Martin Hemberg</ext-link>, Harvard Medical School, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1746057/overview">Edgar Gonzalez-Kozlova</ext-link>, Icahn School of Medicine at Mount Sinai, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Michel Barlaud, <email>barlaud@i3s.unice.fr</email>; Bernard Mari, <email>mari@unice.fr</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>04</day>
<month>03</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>4</volume>
<elocation-id>1340339</elocation-id>
<history>
<date date-type="received">
<day>17</day>
<month>11</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>14</day>
<month>02</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Truchi, Lacoux, Gille, Fassy, Magnone, Lopes Goncalves, Girard-Riboulleau, Manosalva-Pena, Gautier-Isola, Lebrigand, Barbry, Spicuglia, Vassaux, Rezzonico, Barlaud and Mari.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Truchi, Lacoux, Gille, Fassy, Magnone, Lopes Goncalves, Girard-Riboulleau, Manosalva-Pena, Gautier-Isola, Lebrigand, Barbry, Spicuglia, Vassaux, Rezzonico, Barlaud and Mari</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Single-cell CRISPR-based transcriptome screens are potent genetic tools for concomitantly assessing the expression profiles of cells targeted by a set of guides RNA (gRNA), and inferring target gene functions from the observed perturbations. However, due to various limitations, this approach lacks sensitivity in detecting weak perturbations and is essentially reliable when studying master regulators such as transcription factors. To overcome the challenge of detecting subtle gRNA induced transcriptomic perturbations and classifying the most responsive cells, we developed a new supervised autoencoder neural network method. Our Sparse supervised autoencoder (SSAE) neural network provides selection of both relevant features (genes) and actual perturbed cells. We applied this method on an in-house single-cell CRISPR-interference-based (CRISPRi) transcriptome screening (CROP-Seq) focusing on a subset of long non-coding RNAs (lncRNAs) regulated by hypoxia, a condition that promote tumor aggressiveness and drug resistance, in the context of lung adenocarcinoma (LUAD). The CROP-seq library of validated gRNA against a subset of lncRNAs and, as positive controls, HIF1A and HIF2A, the 2 main transcription factors of the hypoxic response, was transduced in A549 LUAD cells cultured in normoxia or exposed to hypoxic conditions during 3, 6 or 24&#xa0;h. We first validated the SSAE approach on HIF1A and HIF2 by confirming the specific effect of their knock-down during the temporal switch of the hypoxic response. Next, the SSAE method was able to detect stable short hypoxia-dependent transcriptomic signatures induced by the knock-down of some lncRNAs candidates, outperforming previously published machine learning approaches. This proof of concept demonstrates the relevance of the SSAE approach for deciphering weak perturbations in single-cell transcriptomic data readout as part of CRISPR-based screening.</p>
</abstract>
<kwd-group>
<kwd>single-cell RNA-seq</kwd>
<kwd>lncRNAs</kwd>
<kwd>hypoxia</kwd>
<kwd>CRISPRi</kwd>
<kwd>sparse supervised autoencoder</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Single Cell Bioinformatics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Cancer cells in solid tumors often suffer from hypoxic stress and adapt to this micro-environment via the activation of Hypoxia inducible factor (HIF), a heterodimeric transcription factor composed of either HIF-1<italic>&#x3b1;</italic> or HIF-2<italic>&#x3b1;</italic> (initially identified as endothelial PAS domain protein (EPAS1)) and HIF-1<italic>&#x3b2;</italic>/ARNT subunits <xref ref-type="bibr" rid="B50">Semenza (2012)</xref>; <xref ref-type="bibr" rid="B47">Rankin et al. (2016)</xref>; <xref ref-type="bibr" rid="B46">Rankin and Giaccia (2016)</xref>. In normoxia, HIF<italic>&#x3b1;</italic> is continuously degraded by an ubiquitin&#x2013;dependent mechanism mediated by interaction with to the von Hippel&#x2013;Lindau (VHL) protein. Hydroxylation of proline residues in HIF<italic>&#x3b1;</italic> is necessary for VHL binding and is catalyzed by the <italic>&#x3b1;</italic>-ketoglutarate-dependent dioxygenases prolyl hydroxylases (PHD). During hypoxia, PHDs are inactive, leading to HIF-<italic>&#x3b1;</italic> stabilization, dimerization with HIF-1<italic>&#x3b2;</italic> and finally translocation into the nucleus to bind to E-box-like hypoxia response elements (HREs) within the promoter region of a wide range of genes that control cellular oxygen homeostasis, erythrocyte production, angiogenesis and mitochondrial metabolism <xref ref-type="bibr" rid="B29">Kaelin and Ratcliffe (2008)</xref>. These molecular changes are notably crucial for cells to adapt to stress by lowering oxygen consumption by shifting from oxidative metabolism to glycolysis. While HIF-1 and HIF-2 bind to the same HRE consensus sequence, they are non-redundant and have distinct target genes and mechanisms of regulation. It is generally accepted that the individual HIFs have specific temporal and functional roles during hypoxia, known as the HIF switch, with HIF-1 driving the initial response and HIF-2 directing the chronic response <xref ref-type="bibr" rid="B34">Koh and Powis (2012)</xref>. In most solid tumors, including lung adenocarcinoma (LUAD), the degree of hypoxia is associated with poor clinical outcome. Induction of HIF activity upregulates genes involved in many hallmarks of cancer, including metabolic reprogramming, epithelial-mesenchymal transition (EMT), invasion and metastasis, apoptosis, genetic instability and resistance to therapies. Emerging evidence have highlighted that hypoxia regulates expression of a wide number of non-coding RNAs classes including microRNAs (miRNAs) and long non-coding RNAs (lncRNAs) that in turn are able to influence the HIF-mediated response to hypoxia <xref ref-type="bibr" rid="B11">Bertero et al. (2017)</xref>; <xref ref-type="bibr" rid="B13">Choudhry and Harris (2018)</xref>; <xref ref-type="bibr" rid="B10">Barth et al. (2020)</xref>. LncRNAs constitute a heterogeneous class of transcripts which are more than 200&#xa0;nt long with low or no protein coding potential, such as intergenic and antisense RNAs, transcribed ultraconserved regions (T-UCR) as well as pseudogenes. Recent advances in cancer genomics have highlighted aberrant expression of a wide set of lncRNAs <xref ref-type="bibr" rid="B12">Carlevaro-Fita et al. (2020)</xref>, revealing their roles in regulating the genome at several levels, including genomic imprinting, chromatin state, transcription activation or repression, splicing and translation control <xref ref-type="bibr" rid="B51">Slack and Chinnaiyan (2019)</xref>. LncRNAs can regulate gene expression through different mechanisms, as guide, decoy, scaffold, miRNA sponges or micropeptides. Of note, recent studies demonstrated the role of several lncRNAs in the direct and indirect regulation of HIF expression and pathway through diverse mechanisms <xref ref-type="bibr" rid="B13">Choudhry and Harris (2018)</xref>. Moreover, hypoxia-responsive lncRNAs have been shown to play regulatory functions in pathways associated with the hallmarks of cancer. For instance, the hypoxia-induced Nuclear-Enriched Abundant Transcript 1 (NEAT1) lncRNA has been associated with the formation of nuclear structures called paraspeckles during hypoxia as well as an increased clonogenic survival of breast cancer cells (<xref ref-type="bibr" rid="B13">Choudhry and Harris, 2018</xref>). Another highly studied lncRNA, Metastasis-Associated Lung Adenocarcinoma Transcript 1 (MALAT1, also known as NEAT2) has been found upregulated by hypoxia in LUAD A549 cells and associated with various cellular functions depending on tumor cell types including cell death, proliferation, migration and invasion <xref ref-type="bibr" rid="B27">Hu et al. (2018)</xref>. Starting from an expression screening in LUAD patients samples and cell lines subjected to hypoxia, we have characterized a new nuclear hypoxia-regulated transcript from the Lung Cancer Associated Transcript (LUCAT1) locus associated with patient prognosis and involved in redox signaling with implication for drug resistance <xref ref-type="bibr" rid="B42">Moreno Leon et al. (2019)</xref>. Additional promising lncRNAs candidates regulated by hypoxia and/or associated with bad prognosis have been identified but deciphering the regulatory functions of these poorly annotated transcripts remains a major challenge. Pooled screening approaches using CRISPR-based technology have offered the possibility to evaluate mammalian gene function, including lncRNAs at genome scale levels <xref ref-type="bibr" rid="B37">Liu et al. (2017)</xref>. More recently, they have been applied to cancer cell lines and have confirmed the oncogenic or tumor suppressor roles of some lncRNAs <xref ref-type="bibr" rid="B19">Esposito et al. (2022)</xref>. This strategy is able to test a large number of candidates simultaneously but require well identified phenotypes such as cell proliferation, cell viability, or cell migration. More subtle screens require techniques based on transcriptomic signatures <xref ref-type="bibr" rid="B21">Gapp et al. (2016)</xref> and approaches have been developed to combine CRISPR gene manipulation, including CRISPR interference and single-cell RNA-seq (scRNA-seq) based on droplet isolation, such as Perturb-seq <xref ref-type="bibr" rid="B49">Replogle et al. (2020)</xref>, CROP-seq <xref ref-type="bibr" rid="B15">Datlinger et al. (2017)</xref> and ECCITE-seq <xref ref-type="bibr" rid="B41">Mimitou et al. (2019)</xref>. These methods combine the advantages of screening a large number of genes simultaneously and linking the modifications to the transcriptomic phenotype, all by breaking down the perturbation signal cell by cell <xref ref-type="bibr" rid="B51">Slack and Chinnaiyan (2019)</xref>; <xref ref-type="bibr" rid="B49">Replogle et al. (2020)</xref>.</p>
<p>In single cell omics applications, most of the quantified features are weakly detected, resulting in large, sparse and noisy data which required feature selection to extract biologically relevant signals <xref ref-type="bibr" rid="B55">Townes et al. (2019)</xref>. Moreover, cells are often grouped according to their phenotype and/or their experimental condition in order to compare features quantification between the defined cell classes. However, the intra-classes heterogeneity can mask a signal of interest. This is particularly the case in the context of CRISPRi screens with a single-cell transcriptomic readout where the inhibition level of the target gene varies between each cell and induces a more or less detectable perturbation signature. Classification tools such as Mixscape <xref ref-type="bibr" rid="B43">Papalexi et al. (2021)</xref>, based on Mixture Discriminant Analysis <xref ref-type="bibr" rid="B24">Hastie and Tibshirani (1996)</xref>, has proven efficacy to identify strong CRISPR-induced effects but was unable to detect subtle weak transcriptomic perturbations.</p>
<p>In the present work, we have developed a single-cell CRISPR-interference-based (CRISPRi) transcriptome screening based on the CROP-Seq approach to gain insight on the regulatory functions of hypoxia-regulated lncRNAs. As a proof-of-concept, we generated a CROP-seq library, including validated guide RNAs (gRNA) targeting six previously identified lncRNAs regulated by hypoxia and/or associated with bad prognosis <xref ref-type="bibr" rid="B42">Moreno Leon et al. (2019)</xref> as well as the two master transcription factors of the hypoxic response (HIF1A and HIF2/EPAS1) and negative control guides. To optimize analysis of fine-tuned regulations in this dataset, we have adapted a Sparse supervised autoencoder (SSAE) neural network <xref ref-type="bibr" rid="B8">Barlaud and Guyard (2021a)</xref>, where we relax the parametric distribution assumption of classical VAE. It leverages on the known cell labels, corresponding to the received gRNA, and a classification loss to incite the latent space to fit the true data distribution. We first validated the approach on HIF1 and HIF2/EPAS1 knock-down, showing a good sensitivity to detect the known temporal switch between both regulators. We then applied the SSAE to the cells treated with the different hypoxia-regulated lncRNAs gRNA to identify subtle signatures linked to the knock-down of the lncRNAs.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<sec id="s2-1">
<title>2.1 Lentivirus production</title>
<p>Lentiviruses were produced using a standard Lipofectamine 2000&#x2122;transfection protocol, using one million HEK293 cells seeded in a 25&#xa0;cm<sup>2</sup> flask in DMEM medium supplemented with 10% bovine serum. A mixture of four plasmids (3&#xa0;&#xb5;g pMDLg/pRRE (addgene &#x201d;12,251&#x201d;), 1.4&#xa0;&#xb5;g pRSV-Rev (addgene &#x201d;12,253&#x201d;), 2&#xa0;&#xb5;g pVSV-G (addgene &#x201d;12,259&#x201d;) and 2.5&#xa0;&#xb5;g of the plasmid containing the expression cassette to package the pooled CROP-seq guides) was transfected. Forty-eight&#xa0;h later, the medium was collected, centrifuged for 5&#xa0;min at 3000&#xa0;rpm, and 2.5&#xa0;mL supernatant containing the viral particles was collected and used to infect cells or aliquoted and stored at &#x2212;80&#xb0;C. Large scale preparations of lentiviruses were produced at the Vectorology facility, PVM, Biocampus (CNRS UMS3426), Montpellier, France.</p>
</sec>
<sec id="s2-2">
<title>2.2 Generation of dCas9-expressing A549 cell line</title>
<p>The A549 lung adenocarcinoma cell line was infected with a lentivirus produced from the plasmid lenti-dCas9-KRAB-MeCP2 (a gift from Andrea Califano, addgene 122,205) allowing the expression of a dCas9-KRAB-MeCP2 fusion protein and a gene conferring resistance to blasticidin. Infected cells were then grown in the presence of 10&#xa0;&#x3bc;g/mL of blasticidin (Sigma). Selection of A549-KRAB-MeCP2 cells was complete within 3&#x2013;5&#xa0;days. Bulk blasticidin resistant cells were amplified and cloned for the CRISPRi scRNA-seq experiments. The best clone was selected according to the expression level of dCas9-KRAB-MeCP2 mRNA and to the most effective inhibition of NLUCAT1 using the NLUCAT1 sg3 RNA.</p>
</sec>
<sec id="s2-3">
<title>2.3 Cloning of individual guides in the CROPseq-Guide-Puro plasmid</title>
<p>The CROPseq-Guide-Puro plasmid (<xref ref-type="bibr" rid="B15">Datlinger et al., 2017</xref>) (a gift from C Bock, Addgene plasmid 86,708) was digested using the restriction enzyme BsmBI (NEB R0580) for 2&#xa0;h at 50&#xb0;C. The relevant fragments (around 8&#xa0;kB) were gel-purified using the Qiagen Gel purification kit and stored at &#x2212;20&#xb0;C in 20-fmol aliquots. Guides against the targeted genes (see <xref ref-type="sec" rid="s11">Supplemental Table S1</xref> for selected sequences) were cloned using the Gibson assembly method (NEBuilder HiFi DNA Assembly Master Mix, NEB E2621). Aliquoted, BsmBI-digested plasmid was mixed with 0.55&#xa0;&#xb5;L guide oligonucleotide (200&#xa0;nM) in 10&#xa0;&#xb5;L total volume, combined with 10&#xa0;&#xb5;L 2X NEBuilder HiFi Assembling Master mix and the mixture was incubated at 50&#xb0;C for 20&#xa0;min. Then, 8&#xa0;&#xb5;L of NEBuilder Assembling mixture was incubated with 100&#xa0;&#xb5;L of Stable competent <italic>E coli</italic>. The mixture was heat-shocked at 42&#xb0;C for 45&#xa0;s and transferred to ice for 2&#xa0;min. SOC medium (900&#xa0;&#xb5;L) was added to the Stabl2-NEBuilder mixture and the mix was incubated at 37&#xb0;C for 1&#xa0;h. Transformed bacterial cells (350&#xa0;&#xb5;L) were plated onto LB agarose plates containing ampicillin (100&#xa0;&#x3bc;g/mL) and incubated overnight at 37&#xb0;C. Individual colonies were picked and grown overnight in 5&#xa0;mL of Terrific Broth medium containing 150&#xa0;&#x3bc;g/mL ampicillin and low-endotoxin, small scale preparation of plasmid DNA were performed using the ToxOut EndoFree Plasmid Mini Kit from BioVision (K1326-250). All plasmids were verified by Sanger sequencing with the primer 5&#x2032;-TTG&#x200b;GGC&#x200b;ACT&#x200b;GAC&#x200b;AAT&#x200b;TCC&#x200b;GT-3&#x2019;.</p>
</sec>
<sec id="s2-4">
<title>2.4 Selection of the guides</title>
<p>A549-dCas9-KRAB-MeCP2 cells were infected with lentiviruses obtained from individual CROPseq-Guide-Puro plasmids, encoding individual guides. Infected cells were then grown in the presence of 1&#xa0;&#x3bc;g/mL of puromycin (Sigma). A week later, total RNAs were purified from A549-KRAB-MeCP2 cells infected with guide encoding lentiviruses and RT-qPCR (primers sequences presented in <xref ref-type="sec" rid="s11">Supplemental Table S2</xref>) were performed to measure expression of the targeted genes. RT-qPCR was performed using Fast SYBR Green Master Mix (Thermo Fisher Scientific) and ABI 7900HT real-time PCR machine. A validated guide was defined as a guide providing at least 75% inhibition of targeted gene expression compared to a control guide.</p>
</sec>
<sec id="s2-5">
<title>2.5 Lentiviral transduction with gRNA libraries and cell preparation for chromium scRNA-seq</title>
<p>A549-dCas9-KRAB-MeCP2 cells were transduced with different amounts of the viral stock containing the library of pooled, selected gRNA. After 6&#xa0;h, the virus-containing medium was replaced by fresh complete culture medium. Puromycin selection (1&#xa0;&#x3bc;g/mL) was started at 48&#xa0;h post-transduction, and 2&#xa0;days later, the plate with about 30% surviving cells was selected, corresponding roughly to a MOI &#x3d; 0.3. The cells were then amplified under puromycin selection for 5 days. The cells were then plated and further cultured in normoxia or in hypoxic condition (1% O2) for 3&#xa0;h, 6&#xa0;h or 24&#xa0;h. Cells were trypsinized counted and assessed for cell viability using the Countess 3 FL (Fisher Scientific). Samples were then stained for multiplexing using cell hashing <xref ref-type="bibr" rid="B53">Stoeckius et al. (2018)</xref>, according to the Cell Hashing Total-Seq-ATM protocol (Biolegend), using 4 distinct Hash Tag Oligonucleotides-conjugated mAbs (TotalSeq&#x2122;-B0255, B0256, B0257 and B0258). Briefly, for each condition, 1.106 cells were resuspended in 100&#xa0;&#xb5;L of PBS, 2% BSA, 0.01% Tween and incubated with 10&#xa0;&#xb5;L Fc Blocking reagent for 10&#xa0;min at 4&#xb0;C then stained with 0.5&#xa0;&#xb5;g of cell hashing antibody for 20&#xa0;min at 4&#xb0;C. After washing with PBS, 2% BSA, 0.01% Tween, samples were counted and merged at the same proportion, spun 5&#xa0;min 350 x g at 4&#xb0;C and resuspended in PBS supplemented with 0.04% of bovine serum albumin at final concentration of 500 cells/&#x03BC;L. Samples were then adjusted to the same concentration, mixed in PBS supplemented with 0.04% of bovine serum albumin and pooled sample were immediately loaded onto10X Genomics Chromium device to perform the single cell capture.</p>
</sec>
<sec id="s2-6">
<title>2.6 Generation of CROP-seq librairies and single-cell RNA-seq data processing</title>
<p>After single-cell capture on the 10X Genomics Chromium device (3&#x2032; V3), libraries were prepared as recommended, following the Chromium Next GEM Single Cell 3&#x2019; Reagent Feature Barcoding V3.1 kit (10X Genomics) and a targeted gRNA amplification <xref ref-type="bibr" rid="B25">Hill et al. (2018)</xref> with respectively 6, 8 and 10 PCR cycles. Libraries were then quantified, pooled (80% RNA libraries, 10% gRNA libraries and 10% hashing libraries) and sequenced on an Illumina NextSeq 2,000. Alignment of reads from the single cell RNA-seq library and unique molecular identifiers (UMI) counting, as well as oligonucleotides tags (HTOs) counting, were performed with 10X Genomics Cell Ranger tool (v6.0.2). Reads of the gRNA library were counted with CITE-seq-Count (v1.4.2). Cells without gRNA counts were discared. Counts matrices of total UMI, HTOs, and gRNA were thus integrated on a single object using Seurat R package (v4.1.0), from which the data were processed for analysis. On the total of 19,663 cells, 817 cells without gRNA counts were discared. HTOs and gRNA were demultiplexed with HTODemux() and MULTIseqDemux(autoThresh &#x3d; TRUE) functions respectively, in order to assign treatment and received gRNA for each cell. On the remaining 18,846 cells, only cells identified as &#x201c;Singlet&#x201d; after demultiplexing of HTO counts were conserved (14,276 cells). The repartition of cells assigned as &#x201d;Doublet&#x201d; (high expression of at list 2 different gRNA), &#x201d;Negative&#x201d; (no detected gRNA) and &#x201d;Singlet&#x201d; (a unique detected gRNA) in all conditions is showed in (<xref ref-type="sec" rid="s11">Supplemental Table S3</xref>). Finally, after transforming the data of the subset of &#x201d;Singlet&#x201d; cells using SCTransform(), computing PCA, and performing KNN clustering, 2 clusters of low UMI content and high mitochondrial content cells (3087 cells) were eliminated for the rest of the analysis.</p>
</sec>
<sec id="s2-7">
<title>2.7 Method: a new sparse supervised autoencoder neural network (SSAE)</title>
<sec id="s2-7-1">
<title>2.7.1 State of the art of neural networks methods</title>
<p>Deep neural networks have proven their efficiency for classification and feature selection in many domains (<xref ref-type="bibr" rid="B18">Emmert-Streib et al., 2020</xref>), and have also been applied to omics data analyses (<xref ref-type="bibr" rid="B38">Lopez et al., 2018</xref>; <xref ref-type="bibr" rid="B36">Leclercq et al., 2019</xref>). Among the proposed neural networks architectures, autoencoders are able to learn a representation of the data, typically in a latent space of lower dimension than the input space. As such, they are often used for dimensionality reduction (<xref ref-type="bibr" rid="B26">Hinton and Zemel, 1994</xref>) and have applications in the medical field as data denoisers or relevant feature selectors (<xref ref-type="bibr" rid="B56">Vincent et al., 2010</xref>; <xref ref-type="bibr" rid="B52">Snoek et al., 2012</xref>). A widely used type of autoencoders is the Variational Autoencoder (VAE) (<xref ref-type="bibr" rid="B32">Kingma and Welling, 2014</xref>). This VAE adds the assumption that the encoded data follows a prior gaussian distribution, and thus combines the reconstruction loss with a distance function (between the gaussian prior and the actual learned distribution). For example, VAEs have been applied to scRNA-seq to predict cell response to biological perturbations (<xref ref-type="bibr" rid="B39">Lotfollahi et al., 2019</xref>) and Omics dataset (<xref ref-type="bibr" rid="B17">Eltager et al., 2023</xref>). Recently, (<xref ref-type="bibr" rid="B35">Le et al., 2018</xref>), provided a supervised auto-encoder neural network that jointly predicts targets and inputs (reconstruction). However, neither VAEs (<xref ref-type="bibr" rid="B32">Kingma and Welling, 2014</xref>) nor SAEs (<xref ref-type="bibr" rid="B35">Le et al., 2018</xref>) provide a solution to the problem of relevant features and cells selections needed to increase the sensitivity of CRISPR-based perturbation associated with scRNAseq readout.</p>
</sec>
<sec id="s2-7-2">
<title>2.7.2 SSAE criterion</title>
<p>In this section, we cope with these two issues by providing a sparse supervised autoencoder (SSAE) neural network method for selecting both relevant features (genes) and actual perturbed cells. <xref ref-type="fig" rid="F1">Figure 1</xref> depicts the main constituent blocks of our proposed approach. Note that we added a &#x201d;soft max&#x201d; block to our SSAE to compute the classification score. Let <italic>X</italic> be the concatenated raw counts matrix (<italic>n</italic> &#xd7; <italic>d</italic>) (n is the number of cells and d the number of genes) of control cells (targeted with a negative control gRNA) and gRNA-targeted cells for each target gene in a particular condition (Normoxia, Hypoxia 3&#xa0;h, 6&#xa0;h or 24&#xa0;h). Let <italic>Y</italic> be the vector of labels (<italic>n</italic> &#xd7; 1) which component is 0 for control cells and 1 for the perturbed cell. Those labels, either &#x201d;control&#x201d; or &#x201d;gRNA-targeted&#x201d;, has been previously assigned for each cell according to the quantification of each gRNA of the CROP-seq library. Let Z be the encoded latent matrix (2 &#xd7; 2). The matrix <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> (<italic>n</italic> &#xd7; <italic>d</italic>) is the reconstructed data. <italic>W</italic> is the matrix of the weights of the linear fully connected autoencoder neural network.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Sparse Supervised autoencoder (SSAE) framework: <bold>(A)</bold> SSAE framework overview. <italic>X</italic> is the concatenated raw counts matrix (<italic>n</italic> &#xd7; d) (n is the number of cells and d the number of genes) of control cells (targeted with a negative control gRNA) and gRNA-targeted cells for each target gene in a particular condition. <italic>Y</italic> is the vector of labels (<italic>n</italic> &#xd7;1) which component is 0 for control cells and 1 for the perturbed cell. <italic>Z</italic> is the encoded latent matrix (2&#xd7;2). <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> (<italic>n</italic> &#xd7; d) is the reconstructed data. <bold>(B)</bold> Two-step SSAE classification of perturbed cells among gRNA-targeted cells.</p>
</caption>
<graphic xlink:href="fbinf-04-1340339-g001.tif"/>
</fig>
<p>The goal is to compute the network weights <italic>W</italic> minimizing the total loss which includes both the classification loss and the reconstruction loss. To perform feature selection, as large datasets often present a relatively small number of informative features, we also want to sparsify the network, following the work proposed in <xref ref-type="bibr" rid="B7">Barlaud and Guyard (2020)</xref>. Thus, instead of the classical computationally expensive lagrangian regularization approach (<xref ref-type="bibr" rid="B23">Hastie et al., 2004</xref>), we propose to minimize the following constrained approach (<xref ref-type="bibr" rid="B5">Barlaud et al., 2017</xref>) according to the Eq. (<xref ref-type="disp-formula" rid="e1">1</xref>):<disp-formula id="e1">
<mml:math id="m3">
<mml:mi>L</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="script">H</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>Z</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:mi>&#x3c8;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mtext>&#x2009;s.t.&#x2009;</mml:mtext>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:mi>W</mml:mi>
<mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>&#x3b7;</mml:mi>
<mml:mo>.</mml:mo>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
<p>We use the Cross Entropy (CE) Loss for the classification loss <inline-formula id="inf3">
<mml:math id="m4">
<mml:mi mathvariant="script">H</mml:mi>
</mml:math>
</inline-formula>. We use the robust Smooth <italic>&#x2113;</italic>
<sub>1</sub> (Huber) Loss (<xref ref-type="bibr" rid="B28">Huber, 2011</xref>) more robust than the mean square error (MSE) as the reconstruction loss <italic>&#x3c8;</italic>.</p>
</sec>
<sec id="s2-7-3">
<title>2.7.3 Sparsity and gene selection using structured projections</title>
<p>A classical approach for structured sparsity is the Group LASSO method (<xref ref-type="bibr" rid="B57">Yuan and Lin, 2006</xref>; <xref ref-type="bibr" rid="B31">Kim and Xing, 2010</xref>) which consists of using the <italic>&#x2113;</italic>
<sub>2,1</sub> norm for the constraint on <italic>W</italic>. However, the <italic>&#x2113;</italic>
<sub>2,1</sub> norm does not induce an efficient sparse structured sparsity of the network (<xref ref-type="bibr" rid="B6">Barlaud et al., 2020</xref>), which leads to negative effects on performance.</p>
<p>In our method we achieve structured sparsity (feature selection) using the bilevel <italic>&#x2113;</italic>
<sub>1,1</sub> projection (<xref ref-type="bibr" rid="B7">Barlaud and Guyard, 2020</xref>) of the weights W. We compute this bilevel <italic>&#x2113;</italic>
<sub>1,1</sub> projection using fast <italic>&#x2113;</italic>
<sub>1</sub> algorithms (<xref ref-type="bibr" rid="B14">Condat, 2016</xref>; <xref ref-type="bibr" rid="B44">Perez et al., 2019</xref>). We can also use the new <italic>&#x2113;</italic>
<sub>1,<italic>&#x221e;</italic>
</sub> which provides similar sparsity performances(<xref ref-type="bibr" rid="B45">Perez et al., 2023</xref>). Note that low values of <italic>&#x3b7;</italic> imply high sparsity of the network. Since the curve accuracy as a function of parameter <italic>&#x3b7;</italic> is concave <xref ref-type="bibr" rid="B45">Perez et al. (2023)</xref>, we compute the maximum with the golden section algorithm (or any classical optimization algorithm). We compute feature importance for the sparse supervised autoencoder using the SHAP method, implemented in the captum python package (<xref ref-type="bibr" rid="B40">Lundberg and Lee, 2017</xref>). Those ranked weights give the top discriminating genes between the compared classes, which can be interpreted as the perturbation signature.</p>
<p>The main difference with the criterion proposed for VAEs in <xref ref-type="bibr" rid="B32">Kingma and Welling (2014)</xref>; <xref ref-type="bibr" rid="B17">Eltager et al. (2023)</xref> and the criterion proposed for SAEs in <xref ref-type="bibr" rid="B35">Le et al. (2018)</xref> is the introduction of the constraint on the weights <italic>W</italic> to sparsify the neural network, in order to select relevant genes.</p>
</sec>
<sec id="s2-7-4">
<title>2.7.4 Selecting actual perturbed cells using the softmax classifier</title>
<p>The goal of this section is to estimate the cells actually perturbed. We propose the following procedure thanks to the softmax formula (Eq. (<xref ref-type="disp-formula" rid="e2">2</xref>)). A first SSAE run gives a perturbation score thanks to the softmax layer <xref ref-type="bibr" rid="B9">Barlaud and Guyard (2021b)</xref> for both non-targeted control cells and for cells targeted for a particular gene.<disp-formula id="e2">
<mml:math id="m5">
<mml:mi>s</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>Z</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>p</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mo movablelimits="false" form="prefix">&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mi>e</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>p</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2200;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1,2</mml:mn>
</mml:math>
<label>(2)</label>
</disp-formula>According to this specific score, called perturbation score, cells are separated into 2 subsets: targeted cells with a score <inline-formula id="inf4">
<mml:math id="m6">
<mml:mo>&#x3e;</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:math>
</inline-formula> are classified as &#x201d;perturbed&#x201d; cells, whereas targeted cells with a score <inline-formula id="inf5">
<mml:math id="m7">
<mml:mo>&#x3c;</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:math>
</inline-formula> are classified as &#x201d;non-perturbed&#x201d; cells. A new data matrix and a new label vector is generated, containing only the raw counts and labels of the selected perturbed cells and an equivalent number of randomly sampled non-targeted control cells in order to balance both classes. A second SSAE run provides a new list of the most discriminant features between both classes, ranked by their weight. This procedure is run multiple times with different initialization seeds in order to compute a mean and a standard deviation of the obtained ranks. The standard deviation ranks are used to evaluate the robustness of the perturbation signature. Again, neither VAEs (<xref ref-type="bibr" rid="B32">Kingma and Welling, 2014</xref>) nor SAEs (<xref ref-type="bibr" rid="B35">Le et al., 2018</xref>) provide a solution to the actual perturbed cell selection.</p>
</sec>
<sec id="s2-7-5">
<title>2.7.5 Implementation of the SSAE framework</title>
<p>Following the work by Frankle and Carbin in (<xref ref-type="bibr" rid="B20">Frankle and Carbin, 2019</xref>), and further developed in (<xref ref-type="bibr" rid="B58">Zhou et al., 2019</xref>), we follow a double descent algorithm, originally proposed as follows: after training a network, set all weights smaller than a given threshold to zero, rewind the rest of the weights to their initial configuration, and then retrain the network from this starting configuration while keeping the zero weights frozen. We replace the thresholding by our <italic>&#x2113;</italic>
<sub>1,1</sub> projection. We implemented our SSAE method using the PyTorch framework for the model, optimizer, schedulers and loss functions. We train the network using the classical Adam optimizer (<xref ref-type="bibr" rid="B33">Kingma and Ba, 2015</xref>). We used a symmetric linear fully connected network (<xref ref-type="bibr" rid="B8">Barlaud and Guyard, 2021a</xref>), with the encoder comprised of an input layer of <italic>d</italic> neurons, one hidden layer followed by a ReLU activation function and a latent layer of dimension <italic>k</italic> &#x3d; 2 since we have two classes. The accuracy of the model, the mean and variance of the rank of selected genes was computed for each SSAE run using 4-fold cross-validation (which means that the train-validation split is random every time) and a mean over 3 seeds.</p>
</sec>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<sec id="s3-1">
<title>3.1 Single-cell CRISPRi screening of hypoxia-regulated lncRNA</title>
<p>In order to gain new insights into the molecular functions of 6 hypoxia-regulated lncRNAs in LUAD cells, we performed a single-cell CRISPRi transcriptome screening based on the CROP-Seq approach (<xref ref-type="fig" rid="F2">Figure 2A</xref>). We transduced A549 cells expressing double transcriptionnal repressor dCas9-KRAB-MeCP2 with a mini-library containing 12 validated gRNA targeting CYTOR (also known as LINC00152), LUCAT1, MALAT1, NEAT1, SNHG12 and SNHG21 as well as the two key regulators of the hypoxic response, HIF1A and HIF2/EPAS1 (<xref ref-type="table" rid="T1">Table 1</xref>). Each guide was individually validated by qPCR in A549 cells, showing a 75%&#x2013;95% inhibition of the target compared with control cells (<xref ref-type="fig" rid="F2">Figure 2B</xref>). Two additional guides, with no effect on the genome, were used as negative controls. In order to mimic the hypoxic environment in which tumors develop <italic>in vivo</italic>, we equally divided the transduced dCas9-KRAB-MeCP2 A549 cells in 4 samples that we then cultured in normoxia or in hypoxia during 3, 6 or 24&#xa0;h. Cells from each sample were labeled with a specific barcoded antibody (HTOs), pooled, and simultaneously sequenced using droplet based scRNA-seq (10X Genomics Chromium). The received gRNA and the culture condition were subsequently assigned for each cell by demultiplexing both gRNA and HTOs counts respectively.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Single-cell CRISPRi screening: <bold>(A)</bold> Design of CROP-seq experiment. <bold>(B)</bold> Bar charts showing the relative expression of the different trancripts by RT-qPCR following infection of A549-KRAB-MeCP2 cells with lentivirus obtained from individual CROPseq-Guide-Puro plasmids, encoding the indicated guides. Expression levels were evaluated using comparative CT method, and normalized using transcript levels of each targeted gene in control conditions (guide Neg-sg1) as 100%. <bold>(C)</bold> Repartition of cells classified as Doublet (more than 1 assigned gRNA), Singlet (1 assigned gRNA), and Negative (no assigned gRNA) after demulplexing of gRNA counts in each condition, prior to low quality cells filtering. <bold>(D)</bold> Heatmap of target gene RNA in each cell, labelled according to assigned gRNA and condition after demultiplexing.</p>
</caption>
<graphic xlink:href="fbinf-04-1340339-g002.tif"/>
</fig>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>gRNA library.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">gRNA</th>
<th align="center">Target</th>
<th align="center">Type</th>
<th align="center">% Inhibition</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">HIF1A-sg1</td>
<td rowspan="2" align="center">HIF1A</td>
<td rowspan="3" align="center">Hypoxic response regulator</td>
<td align="center">&#x3e;95%</td>
</tr>
<tr>
<td align="center">HIF1A-sg2</td>
<td align="center">&#x3e;95%</td>
</tr>
<tr>
<td align="center">HIF2-sg5</td>
<td align="center">HIF2/EPAS1</td>
<td align="center">&#x3e;95%</td>
</tr>
<tr>
<td align="center">LINC00152-sg3</td>
<td align="center">CYTOR</td>
<td rowspan="9" align="center">Hypoxia-regulated lncRNA</td>
<td align="center">&#x3e;75%</td>
</tr>
<tr>
<td align="center">LUCAT-sg3</td>
<td rowspan="2" align="center">LUCAT1</td>
<td align="center">&#x3e;97%</td>
</tr>
<tr>
<td align="center">LUCAT-sg5</td>
<td align="center">&#x3e;90%</td>
</tr>
<tr>
<td align="center">MALAT-sg1</td>
<td align="center">MALAT1</td>
<td align="center">&#x3e;95%</td>
</tr>
<tr>
<td align="center">NEAT1-sg2</td>
<td rowspan="2" align="center">NEAT1</td>
<td align="center">&#x3e;85%</td>
</tr>
<tr>
<td align="center">NEAT1-sg6</td>
<td align="center">&#x3e;95%</td>
</tr>
<tr>
<td align="center">SNHG12-sg1</td>
<td rowspan="2" align="center">SNHG12</td>
<td align="center">&#x3e;75%</td>
</tr>
<tr>
<td align="center">SNHG12-sg3</td>
<td align="center">&#x3e;90%</td>
</tr>
<tr>
<td align="center">SNHG21-sg5</td>
<td align="center">SNHG21</td>
<td align="center">&#x3e;85%</td>
</tr>
<tr>
<td align="center">Neg-sg1</td>
<td rowspan="2" align="center">None</td>
<td rowspan="2" align="center">Negative control</td>
<td rowspan="2" align="center">None</td>
</tr>
<tr>
<td align="center">Neg-sg2</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Overall, we found a balanced representation for each treatment and for each gRNA among the sequenced cells, except for the cells targeted by &#x201d;SNHG12-sg3&#x201d; which were depleted in all conditions (<xref ref-type="fig" rid="F2">Figure 2C</xref>, <xref ref-type="sec" rid="s11">Supplemental Figure SA</xref> and <xref ref-type="sec" rid="s11">Supplemental Table S3</xref>). Moreover, the expression of this particular gRNA was lowly detected in those cells, confirming previous observations that this gRNA induced cell death and that only cells with low expression survive. Inhibition of target gene expression in the presence of their corresponding gRNA were validated in all 4 conditions, as well as their progressive increase (CYTOR, LUCAT1, NEAT1, SNHG12) or decrease (HIF1A and SNHG21) during hypoxia exposure (<xref ref-type="fig" rid="F2">Figure 2D</xref>).</p>
</sec>
<sec id="s3-2">
<title>3.2 Mathematical and biological validation of the SSAE approach</title>
<p>In order to validate the SSAE approach, we first evaluated the effect of feature and cell selection on the accuracy of the model, and then the biological relevance of the detected transcriptomic perturbations induced by the knock-down of the two main regulators of the hypoxic response, HIF1A and HIF2.</p>
<sec id="s3-2-1">
<title>3.2.1 Feature and cell selection improve the accuracy of the model</title>
<p>We compared the performance of SSAE, for different criterion (MSE and Huber), with or without gene selection using <italic>&#x2113;</italic>
<sub>1,1</sub> constraint projection, and with or without cell selection on the dataset of HIF2-targeted cells and negative control cells cultured in hypoxic condition for 24&#xa0;h. <xref ref-type="table" rid="T2">Table 2</xref> indicates that the SSAE with <italic>&#x2113;</italic>
<sub>1,1</sub> constraint projection and Huber criterion is able to discriminate both classes by selecting only a fraction of measured genes (19.46%), as it is shown in the matrix of connections between the first and second layer (<xref ref-type="fig" rid="F3">Figure 3A</xref>). Moreover, the improvement of 5.39% of the model accuracy by using SSAE with <italic>&#x2113;</italic>
<sub>1,1</sub> constraint projection compared to SSAE without projection shows the efficiency of selecting only the most relevant features.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Comparison of different criterion on HIF2 datase (CE is the cross entropy).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Criterion</th>
<th align="center">Gene selection (%)</th>
<th align="center">F1 score (%)</th>
<th align="center">Accuracy (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">CE &#x2b; Huber and No-proj</td>
<td align="center">97</td>
<td align="center">80.54</td>
<td align="center">87.29</td>
</tr>
<tr>
<td align="center">CE &#x2b; Huber &#x2b; <italic>&#x2113;</italic>
<sub>1,1</sub>
</td>
<td align="center">19.46</td>
<td align="center">90.48</td>
<td align="center">92.68</td>
</tr>
<tr>
<td align="center">CE &#x2b; MSE &#x2b; <italic>&#x2113;</italic>
<sub>1,1</sub>
</td>
<td align="center">19.46</td>
<td align="center">89.86</td>
<td align="center">92.16</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Impact of feature and cell selection on the accuracy of the model comparing HIF2-targeted cells and negative control cells after 24&#xa0;h of hypoxic exposure. <bold>(A)</bold> Sparsity of the first layer: Left: using no projection, Right: using our <italic>&#x2113;</italic>
<sub>1,1</sub> projection. <bold>(B)</bold> Distribution of the perturbation score for non-targeted and HIF2-targeted cells.</p>
</caption>
<graphic xlink:href="fbinf-04-1340339-g003.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="F3">Figure 3B</xref> shows the distribution of perturbation scores computed with the softmax formula (<xref ref-type="disp-formula" rid="e2">2</xref>) for non-targeted and HIF2-targeted cells. Running the SSAE after selecting only perturbed HIF2-targeted cells, without feature selection, improves the accuracy by 4.67% (<xref ref-type="table" rid="T3">Table 3</xref>). Combining feature and cell selection improves the accuracy by 10.4%. To further demonstrate the efficiency of the SSAE classification, we compared the average expression of the top 15 genes of the perturbation signature induced by HIF1A and HIF2 inhibition between control, non-perturbed and perturbed cells in the 3 time points of hypoxia exposure. <xref ref-type="sec" rid="s11">Supplemental Figure SB</xref> shows that the effect of HIF1A or HIF2 inhibition is systematically mitigated in cells classified as non-perturbed compared to perturbed cells. The percentage of cells classified as perturbed among targeted cells for each gRNA in each condition are shown in <xref ref-type="fig" rid="F4">Figure 4</xref>.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>SSAE Accuracy without or with feature and cell selection.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">SSAE (without feature and cell selection) (%)</th>
<th align="center">SSAE (only with feature selection) (%)</th>
<th align="center">SSAE (with feature and cell selection) (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">87.29</td>
<td align="center">92.68</td>
<td align="center">97.67</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Percentages of targeted cells classified as perturbed or non-perturbed for each gRNA in each condition.</p>
</caption>
<graphic xlink:href="fbinf-04-1340339-g004.tif"/>
</fig>
<p>Finally, we tested the stability of the SSAE according of the number of cells by progressively sub-sampling the cells in each condition and evaluated the classification accuracy for each condition. The data showed overall a very good stability, with however some differences between the conditions (<xref ref-type="sec" rid="s11">Supplemental Figure SC</xref>), indicating that the number of cells required to obtain an optimal result also depends on the intensity of the biological signal.</p>
</sec>
<sec id="s3-2-2">
<title>3.2.2 Knock-down of HIF1A and HIF2 differentially modulate the hypoxic response</title>
<p>Globally, the inhibition of HIF1A induced a strong transcriptomic perturbation which affected more than 85% of targeted cells in all conditions (<xref ref-type="table" rid="T4">Table 4</xref>; <xref ref-type="fig" rid="F4">Figure 4</xref>). Even in normoxic condition, the signature amplitude was sufficient to allow a classification accuracy above 93%. Among the genes modulated independently from the hypoxic status, we found SNAPC1, IGFL2-AS1, BNIP3L and LDHA, whereas PGK1, PDK1, or BNIP3 modulations were specific to hypoxic conditions (<xref ref-type="fig" rid="F5">Figure 5A</xref>). We also found gene modulations specific to early (KDM3A, HIPLDA, ZNF292, EGLN3) or late (SLC16A3, GPI, PGAM1, TPI1) hypoxic response, which correspond to the progressive establishment of the HIF1A-mediated metabolic switch <xref ref-type="bibr" rid="B30">Kim et al. (2006)</xref>. In normoxia, the knock-down of HIF2 did not produce stable perturbations, except for its own target gene EPAS1 (<xref ref-type="fig" rid="F5">Figure 5B</xref>). The 2 early time points of hypoxia exposure showed an improvement of the associated classification accuracy, which reflected a slight increase of the transcriptomic perturbation induced by HIF2 knock-down in these experimental settings. This early signature was mainly driven by genes involved in lipid metabolism ANGPTL4, IGFBP3 and HILPGA. Discrepancies between the results at 3&#xa0;h or 6&#xa0;h were mainly due to the lower number or targeted cells at 6&#xa0;h (104 instead of 202), which impacted the classification. At 24&#xa0;h of hypoxic exposure, the effect of HIF2 inhibition reached its maximum, with 84% perturbed cells and an accuracy of 97% (<xref ref-type="table" rid="T4">Table 4</xref>; <xref ref-type="fig" rid="F4">Figure 4</xref>). However, this signature was quite different from that of HIF1A-targeted cells under the same condition. Indeed, some upregulated (ALDH3A1, CPLX2, FTL, PAPPA) or downregulated (ATP1B1, FXYD2, ANXA4, LOXL2) genes in HIF2-targeted cells were not modulated in HIF1A-targeted cells (<xref ref-type="fig" rid="F5">Figure 5C</xref>). Moreover, several genes showed an opposite perturbation between the two groups of cells. This was the case for BNIP3, PGK1, GPI, FAM162A, SLC16A3, TPI1, or PGAM1 which were downregulated upon HIF1A inhibition but were found upregulated upon HIF2 inhibition after 24&#xa0;h of culture in hypoxia. These results were consistent with the known role of HIF2, which is activated upon prolonged exposure to hypoxia and is involved in the regulation of the chronic hypoxic response <xref ref-type="bibr" rid="B34">Koh and Powis (2012)</xref>. They also confirm that in LUAD cells, HIF1A and HIF2-regulated functions are specific, or even antagonistic for certain genes, which has been previously demonstrated in other cancers <xref ref-type="bibr" rid="B48">Raval et al. (2005)</xref>.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>SSAE classification and accuracy for HIF1A and HIF2 gRNA targeted cells.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="left">Treatment</th>
<th align="left">Targeted cells</th>
<th align="left">Perturbed cells (%)</th>
<th align="left">Accuracy (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">HIF1A</td>
<td align="left">Normoxia</td>
<td align="left">475</td>
<td align="left">86.7</td>
<td align="left">95.33</td>
</tr>
<tr>
<td align="left">HIF1A</td>
<td align="left">Hypoxia 3&#xa0;h</td>
<td align="left">554</td>
<td align="left">87.5</td>
<td align="left">94</td>
</tr>
<tr>
<td align="left">HIF1A</td>
<td align="left">Hypoxia 6&#xa0;h</td>
<td align="left">372</td>
<td align="left">85.8</td>
<td align="left">93.67</td>
</tr>
<tr>
<td align="left">HIF1A</td>
<td align="left">Hypoxia 24&#xa0;h</td>
<td align="left">554</td>
<td align="left">85.7</td>
<td align="left">94.67</td>
</tr>
<tr>
<td align="left">HIF2</td>
<td align="left">Normoxia</td>
<td align="left">147</td>
<td align="left">11.6</td>
<td align="left">66.67</td>
</tr>
<tr>
<td align="left">HIF2</td>
<td align="left">Hypoxia 3&#xa0;h</td>
<td align="left">202</td>
<td align="left">51</td>
<td align="left">86.33</td>
</tr>
<tr>
<td align="left">HIF2</td>
<td align="left">Hypoxia 6&#xa0;h</td>
<td align="left">104</td>
<td align="left">38.5</td>
<td align="left">82.33</td>
</tr>
<tr>
<td align="left">HIF2</td>
<td align="left">Hypoxia 24&#xa0;h</td>
<td align="left">213</td>
<td align="left">84</td>
<td align="left">97.67</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Knock-down of HIF1A and HIF2 differentially modulate the hypoxic response. <bold>(A)</bold> Top 20 discriminant features between perturbed and control cells for HIF1A for each treatment. Upregulated or downregulated genes are written in red or blue respectively. <bold>(B)</bold> Top 20 discriminant features between perturbed and control cells for HIF2/EPAS1 for each treatment. <bold>(C)</bold> Differentially expressed genes between perturbed and control cells for HIF1A and HIF2 for each treatment, expressed as log2FC (Fold Change).</p>
</caption>
<graphic xlink:href="fbinf-04-1340339-g005.tif"/>
</fig>
</sec>
</sec>
<sec id="s3-3">
<title>3.3 Knock-down of hypoxia-regulated lncRNA LUCAT1 leads to hypoxic condition-dependent transcriptomic modulations</title>
<p>We then applied the SSAE method to classify cells treated with the 6 gRNA targeting hypoxia-regulated lncRNAs and cultured in the 4 conditions. Globally, the SSAE was able to classify perturbed and control cells with a good overall accuracy around 80%, except for SNHG12 and SNHG21 (<xref ref-type="table" rid="T5">Table 5</xref>; <xref ref-type="fig" rid="F4">Figure 4</xref>). However, despite their promising accuracies, we did not detect any other stable perturbations than the target gene for both MALAT1 and NEAT1 targeted cells, as indicated by the obtained high means and standard deviations of the computed ranks, while those two lncRNAs were previously associated with various gene regulation functions <xref ref-type="bibr" rid="B16">Dong et al. (2018)</xref>; <xref ref-type="bibr" rid="B3">Amodio et al. (2018)</xref> (<xref ref-type="fig" rid="F6">Figures 6A,B</xref>). The SSAE outcomes were different for the classification of LUCAT1-targeted cells. Indeed, the transcriptomic inhibition of LUCAT1 resulted in a stable upregulation for PCOLCE2 and ISCA1 in normoxia, HDHD2 after 3&#xa0;h, and 6&#xa0;h of hypoxia (<xref ref-type="fig" rid="F6">Figure 6C</xref>). ISCA1 and HDHD2 encode for metal ion binding proteins, whereas TFCP2 is a known oncogene. After 24&#xa0;h of hypoxia, a completely different perturbation signature was found, with at least 6 stably modulated genes, including the upregulation of KDM5C, TMEM175 and NIT1, as well as the downregulation of ATP6AP1, PEX1 and PHF20. ATP6AP1 and PEX1 are respectively components of the V-ATPase and the peroxisomal ATPase complexes, while TMEM175 is a proton channel also involved in pH regulation. KDMC5 and PHF20 are both involved in chromatin remodeling and transcriptomic regulation, while NIT1 is associated to tumor suppressor functions. This particular signature allowed the classification of 90,4% of targeted cells with an accuracy of 85,67% (<xref ref-type="table" rid="T5">Table 5</xref>; <xref ref-type="fig" rid="F4">Figure 4</xref>). These results indicate that LUCAT1 inhibition may induce hypoxic condition-dependent transcriptomic modulations that potentially impact tumor survival and gene regulatory processes during prolonged exposure to hypoxic conditions, completing our previous observations <xref ref-type="bibr" rid="B42">Moreno Leon et al. (2019)</xref>.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Supervised autoencoder classification and accuracy for hypoxia regulated lncRNAs gRNA targeted cells (&#x2a;obtained without cell selection).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="left">Treatment</th>
<th align="left">Targeted cells</th>
<th align="left">Perturbed cells (%)</th>
<th align="left">Accuracy</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">LUCAT1</td>
<td align="left">Normoxia</td>
<td align="left">438</td>
<td align="left">51.8</td>
<td align="left">73.33%</td>
</tr>
<tr>
<td align="left">LUCAT1</td>
<td align="left">Hypoxia 3&#xa0;h</td>
<td align="left">583</td>
<td align="left">77.7</td>
<td align="left">82.33%</td>
</tr>
<tr>
<td align="left">LUCAT1</td>
<td align="left">Hypoxia 6&#xa0;h</td>
<td align="left">391</td>
<td align="left">63.9</td>
<td align="left">79%</td>
</tr>
<tr>
<td align="left">LUCAT1</td>
<td align="left">Hypoxia 24&#xa0;h</td>
<td align="left">666</td>
<td align="left">90.4</td>
<td align="left">85.67%</td>
</tr>
<tr>
<td align="left">MALAT1</td>
<td align="left">Normoxia</td>
<td align="left">205</td>
<td align="left">37.1</td>
<td align="left">82.67%</td>
</tr>
<tr>
<td align="left">MALAT1</td>
<td align="left">Hypoxia 3&#xa0;h</td>
<td align="left">269</td>
<td align="left">45</td>
<td align="left">82.33%</td>
</tr>
<tr>
<td align="left">MALAT1</td>
<td align="left">Hypoxia 6&#xa0;h</td>
<td align="left">194</td>
<td align="left">37.1</td>
<td align="left">78.00%</td>
</tr>
<tr>
<td align="left">MALAT1</td>
<td align="left">Hypoxia 24&#xa0;h</td>
<td align="left">241</td>
<td align="left">26.1</td>
<td align="left">78.67%</td>
</tr>
<tr>
<td align="left">NEAT1</td>
<td align="left">Normoxia</td>
<td align="left">274</td>
<td align="left">59.1</td>
<td align="left">86.33%</td>
</tr>
<tr>
<td align="left">NEAT1</td>
<td align="left">Hypoxia 3&#xa0;h</td>
<td align="left">390</td>
<td align="left">73.3</td>
<td align="left">89.00%</td>
</tr>
<tr>
<td align="left">NEAT1</td>
<td align="left">Hypoxia 6&#xa0;h</td>
<td align="left">237</td>
<td align="left">52.7</td>
<td align="left">85.33%</td>
</tr>
<tr>
<td align="left">NEAT1</td>
<td align="left">Hypoxia 24&#xa0;h</td>
<td align="left">566</td>
<td align="left">58.7</td>
<td align="left">87.33%</td>
</tr>
<tr>
<td align="left">LINC00152</td>
<td align="left">Normoxia</td>
<td align="left">147</td>
<td align="left">55.1</td>
<td align="left">87.67%</td>
</tr>
<tr>
<td align="left">LINC00152</td>
<td align="left">Hypoxia 3&#xa0;h</td>
<td align="left">200</td>
<td align="left">59</td>
<td align="left">91.67%</td>
</tr>
<tr>
<td align="left">LINC00152</td>
<td align="left">Hypoxia 6&#xa0;h</td>
<td align="left">150</td>
<td align="left">58</td>
<td align="left">86.33%</td>
</tr>
<tr>
<td align="left">LINC00152</td>
<td align="left">Hypoxia 24&#xa0;h</td>
<td align="left">209</td>
<td align="left">42.6</td>
<td align="left">85%</td>
</tr>
<tr>
<td align="left">SNHG12</td>
<td align="left">Normoxia</td>
<td align="left">196</td>
<td align="left">14.8</td>
<td align="left">65%&#x2a;</td>
</tr>
<tr>
<td align="left">SNHG12</td>
<td align="left">Hypoxia 3&#xa0;h</td>
<td align="left">217</td>
<td align="left">14.7</td>
<td align="left">69%&#x2a;</td>
</tr>
<tr>
<td align="left">SNHG12</td>
<td align="left">Hypoxia 6&#xa0;h</td>
<td align="left">154</td>
<td align="left">3.9</td>
<td align="left">67.4%&#x2a;</td>
</tr>
<tr>
<td align="left">SNHG12</td>
<td align="left">Hypoxia 24&#xa0;h</td>
<td align="left">213</td>
<td align="left">8.5</td>
<td align="left">71%&#x2a;</td>
</tr>
<tr>
<td align="left">SNHG21</td>
<td align="left">Normoxia</td>
<td align="left">211</td>
<td align="left">5.2</td>
<td align="left">60.7%&#x2a;</td>
</tr>
<tr>
<td align="left">SNHG21</td>
<td align="left">Hypoxia 3&#xa0;h</td>
<td align="left">256</td>
<td align="left">3.5</td>
<td align="left">61.1%&#x2a;</td>
</tr>
<tr>
<td align="left">SNHG21</td>
<td align="left">Hypoxia 6&#xa0;h</td>
<td align="left">192</td>
<td align="left">6.8</td>
<td align="left">59.1%&#x2a;</td>
</tr>
<tr>
<td align="left">SNHG21</td>
<td align="left">Hypoxia 24&#xa0;h</td>
<td align="left">299</td>
<td align="left">2</td>
<td align="left">63.5%&#x2a;</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Top 20 discriminant features between perturbed and control cells for LUCAT1 <bold>(A)</bold>, MALAT1 <bold>(B)</bold> and NEAT1 <bold>(C)</bold> for each treatment. Upregulated or downregulated genes are written in red or blue respectively.</p>
</caption>
<graphic xlink:href="fbinf-04-1340339-g006.tif"/>
</fig>
<p>For LINC00152, the combined inhibition of CYTOR/LINC00152 with MIR4435-2HG (<xref ref-type="fig" rid="F7">Figure 7A</xref>), whose sequences are highly homologous (99% in the 220&#xa0;bp region including the most efficient gRNA), was sufficient to select half of the targeted cells with an accuracy above 85% regardless of the condition (<xref ref-type="table" rid="T4">Table 4</xref>; <xref ref-type="fig" rid="F4">Figure 4</xref>).</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Top 20 discriminant features between perturbed and control cells for LINC00152 <bold>(A)</bold>, SNHG21 <bold>(B)</bold> and SNHG12 <bold>(C)</bold> for each treatment. Upregulated or downregulated genes are written in red or blue respectively.</p>
</caption>
<graphic xlink:href="fbinf-04-1340339-g007.tif"/>
</fig>
<p>For the SNHG12 and SNHG21 datasets, the first round of SAE selected only around 10% of perturbed cells (<xref ref-type="table" rid="T5">Table 5</xref>; <xref ref-type="fig" rid="F4">Figure 4</xref>). Thus we could not run the SAE for the second round because of a too low number of cells for the 4 fold cross validation. For those 2 genes, we just reported the average accuracies obtained after the first round of the SSAE (<xref ref-type="table" rid="T5">Table 5</xref>). As SNHG21 expression is relatively low in LUAD cells and is decreased by hypoxic stress, the extent of its inhibition was therefore weaker and not sufficient to distinguish targeted from control cells. Combined with the lack of transcriptomic effect induced by its knock-down, it explains the poor classification results and the randomness of features selected for cells targeted by this particular gene under all conditions (<xref ref-type="fig" rid="F7">Figure 7B</xref>).</p>
</sec>
<sec id="s3-4">
<title>3.4 SSAE classification revealed an anti-apoptotic signature expressed by a subset of SNHG12-targeted cells in response to the cytotoxic effect of one of its gRNA</title>
<p>Looking at the SSAE classification outcomes for SNHG12-targeted cells, only about 15% of them were classified as perturbed in normoxia and after 3&#xa0;h of hypoxia, with a poor accuracy (<xref ref-type="table" rid="T5">Table 5</xref>; <xref ref-type="fig" rid="F4">Figure 4</xref>). The number of selected cells was even worse for a longer exposure to hypoxia.</p>
<p>Nevertheless, the ranked list of top discriminant features between the few perturbed cells and control cells obtained for the first two time points showed a notable perturbation signature. In normoxia, it was only composed of BAG1 upregulation, whereas after 3&#xa0;h of hypoxia exposure, this signature was completed by GAS5 (snoRNAs-containing lncRNA gene), ARRB2, ATF5, and ETHE1 upregulations (<xref ref-type="fig" rid="F7">Figure 7C</xref>). These 5 genes are all known anti-apoptotic factors. We hypothesized that this anti-apoptotic signature was expressed by a subset of LUAD cells that were actively escaping the cytotoxic effect we systematically observed for the most efficient of the two gRNA selected for targeting SNHG12, SNHG12-sg3 (<xref ref-type="table" rid="T1">Table 1</xref>). Indeed, most of the cells classified as perturbed were specific to this particular gRNA (<xref ref-type="fig" rid="F4">Figure 4</xref>). As this signature progressively attenuated over time under hypoxic conditions, we speculate that the activation of this anti-apoptotic response may be inhibited by hypoxic stress, or that hypoxia may protect against the cytotoxic effect of this guide. These results demonstrate the precision of the SAE-based approach to detect a short signature, even restricted to a small subset of cells.</p>
</sec>
<sec id="s3-5">
<title>3.5 Comparison of SSAE with others machine learning methods</title>
<p>We compared the classification performance and the biological relevance of extracted features between the SSAE, with and without cell selection of the most responsive cells, SAE <xref ref-type="bibr" rid="B35">Le et al. (2018)</xref> and Random Forests using 400 estimators and the Gini importance (GI) for feature ranking. We performed this comparison for 2 representative datasets, namely, HIF2-targeted cells <italic>versus</italic> control cells and LUCAT1-targeted cells <italic>versus</italic> control cells following 24&#xa0;h of hypoxia exposure.</p>
<p>For the first dataset, as HIF2 inhibition induced a strong perturbation signature, the 10 first selected features between all methods were highly similar, even between SSAE/SAE and Random Forest (<xref ref-type="fig" rid="F8">Figure 8A</xref>). However, using the SSAE with cell selection outperforms Random Forest with an increase of 25.25% of accuracy.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Comparison of the 10 first selected features between SSAE (with or without cell selection), SAE and Random Forests, for HIF2/EPAS1 <bold>(A)</bold> or LUCAT1 <bold>(B)</bold> targeted cells in hypoxia 24&#xa0;h. Associated accuracies are indicated in brackets.</p>
</caption>
<graphic xlink:href="fbinf-04-1340339-g008.tif"/>
</fig>
<p>For LUCAT1 dataset, few overlaps were found between the first 10 selected features with each method. The inhibition of the target gene LUCAT1 was the only feature commonly detected (<xref ref-type="fig" rid="F8">Figure 8B</xref>). ATP6AP1 and PEX1 were the only two overlapping genes between SSAE and SAE obtained signatures, while KDMC5, NIT1, PHF20 and CCDC142 were specific to SSAE regardless of the cell selection step. While the signature of Random Forests was very specific, note that the author of RF proposes two measures for feature ranking, the variable importance (VI) and Gini importance (GI): <xref ref-type="bibr" rid="B2">Altmann et al. (2010)</xref> showed that if predictors are real with multimodal Gaussian distributions, both measures are biased. Moreover, since using the SSAE with cell selection outperforms RF by 31.07%, SAE by 27% and SSAE without cell selection by 1.67%, it is reasonable to claim that the SSAE perturbation signature is the most relevant.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<p>Single-cell CRISPR(i)-based transcriptome screenings are powerful tools for simultaneously accessing the expression profiles of cells targeted by different gRNA, in order to infer target genes functions from the observed perturbations. However, these approaches are limited by the low molecule capture rate and sequencing depth provided by droplet-based scRNA-seq, which produce sparse and noisy data. Furthermore, the outcome of CRISPR-induced modification in each cell is a stochastic event, depending among other things, on the expression levels of the transcribed gRNA and dCas9, as well as the accessibility of the target gene locus, that may be heterogeneously regulated at the epigenomic levels in the different cells. For these reasons, the induced perturbation signature and its detection are likely heterogeneous between cells, even when dCas9-expressing cells receiving the same gRNA have been cloned. Deciphering this heterogeneity in sparse data is even more complex when the targeted genes are not master genes involved in signaling or regulatory pathways, such as transcription factors and receptors. In this respect, a previous study <xref ref-type="bibr" rid="B43">Papalexi et al. (2021)</xref> has shown that this particular challenge cannot be met using conventional scRNA-seq analysis tools such as differential expression, which is clearly limited to the detection of weak and heterogeneous perturbation signals. This challenge seems even more complex for the study of perturbations mediated by knockdown of non-coding RNAs, which have been largely involved in the fine-tuning of gene expression regulation. To increase the sensitivity of single-cell CRISPR(i)-based transcriptome screenings, we propose here a powerful feature selection and classification approach based on a sparse supervised autoencoder (SSAE). It leverages in particular on the known cell labels initially given by gRNA counts demultiplexing to constrain the latent space to fit the original data distribution. Beyond high statistical accuracy, our SSAE offered relevant properties that distinguishes it from classical classification methods: i) a stringent feature selection producing an interpretable readout of ranked top discriminant genes associated to their weights; ii) a classification score which allow the selection of the most perturbed cells and the eventual signal to obtain a more robust perturbation signature. We first validated this approach by analyzing the perturbations associated with the knock-down of the two master regulators of the hypoxic response, HIF1A and HIF2. We showed that the SSAE was able to learn a latent space and a perturbation signature which can for exemple almost perfectly discriminate HIF2-targeted cells from their control in condition of prolonged hypoxia. The SSAE classification accuracy provided a global perturbation score associated with HIF1A and HIF2 at each time point, reflecting the biological activity of each factor during the hypoxic response. We were able to recapitulate the known distinct influence and target specificity of HIF1 and HIF2 during the hypoxia time course <xref ref-type="bibr" rid="B34">Koh and Powis (2012)</xref>, with notably i) a strong perturbation driven by HIF1 at early time points; ii) a progressive influence of HIF2 with a maximum effect observed at 24&#xa0;h of hypoxia; iii) a specificity regarding their targets, with sometimes an opposite regulation for some genes. Finally, this unique dataset provides a global and dynamic description of the transcriptomic modulations mediated by the two main regulators of the hypoxic response in LUAD A549 cells. Surprisingly, we did not detect any relevant and stable perturbation in cells targeted for LINC00152, MALAT1, NEAT1 and SNHG21, in the four culture conditions. This result appears quite unexpected for MALAT1 and NEAT1, two of the most studied lncRNAs that are associated with various functions in cancer, including proliferation, migration, and invasion <xref ref-type="bibr" rid="B4">Arun et al. (2020)</xref>; <xref ref-type="bibr" rid="B16">Dong et al. (2018)</xref>. In particular, it has been shown that MALAT1 knockout in the same cellular model (A549) modulated a set of metastasis-associated genes <xref ref-type="bibr" rid="B22">Gutschner et al. (2013)</xref>. Although CRISPRi-mediated knock-down achieved an efficient knock-down <inline-formula id="inf6">
<mml:math id="m8">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x3e;</mml:mo>
<mml:mn>95</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, it is however possible that based on the very high level of MALAT1, the remaining transcripts are sufficient to mediate the cellular function. Another possibility could be due to differences in methodology, notably the need to isolate single clones for the knockout protocol, a long procedure that can profoundly affect the transcriptome, compared with the CROP-seq approach performed on a bulk population prior to immediate single-cell isolation. A similar situation may occur for NEAT1, a highly abundant lncRNA acting as a structural scaffold of membraneless paraspeckle nuclear bodies. Moreover, NEAT1 can produces two isoforms, with a differential regulation upon stress and distinct functions <xref ref-type="bibr" rid="B1">Adriaens et al. (2019)</xref>. Additional work will be thus necessary to further analyze the relative proportion of the two isoforms in A549 cells and their potential function during hypoxia.</p>
<p>However, for LUCAT1-targeted cells after 24&#xa0;h of hypoxia exposure, we found a stable signature of 6 modulated genes, which are associated with pH or gene regulation. It suggested a potential capacity of LUCAT1 to promote tumor cell survival during prolonged hypoxia and to contribute to an aggressive phenotype in LUAD cells, as we previously demonstrated <xref ref-type="bibr" rid="B42">Moreno Leon et al. (2019)</xref>. Finally, we also found a relevant signature in SNHG12-targeted cells, characterized by the upregulation of anti-apoptotic genes. As this signature is almost exclusive to cells targeted by the most effective gRNA against SNHG-12, which appeared to systematically induce cell death, we hypothesized that it is expressed by surviving cells. The potential pro-oncogenic role of the complex SNHG12 locus, producing a lncRNA and 3 snoRNAs, should be pursued to decipher the molecular components associated with this phenotype, as also suggested by previous studies <xref ref-type="bibr" rid="B54">Tamang et al. (2019)</xref>.</p>
<p>Autoencoder neural networks are state of the art for biological analyses <xref ref-type="bibr" rid="B17">Eltager et al. (2023)</xref> and do not represent a computational issue thanks to their implementation in dedicated frameworks such as PyTorch. Compared to others autoencoders, the SSAE add the projection step to achieve feature selection. However, its execution time is very fast: for example, it is only 60&#xa0;ms using a Macbook with a M3 Processor for a Matrix of 1000 samples and 10,000 features <xref ref-type="bibr" rid="B45">Perez et al. (2023)</xref>.</p>
<p>The size of the perturbation signatures obtained for LUCAT1 and SNHG12 datasets prevented the utilization of functional enrichment analysis to characterize their modulated functions. Moreover, as these small signatures were found in specific subsets of targeted cells and dynamically during the hypoxic response, it appears very difficult to validate them using a global experimental approach that will average the signal across all cells. Despite these limitations, we believe that our approach is well suited to the particular deciphering of single cell CRISPR-based screen with omics readout, or for other similar assays to assess the effect of perturbation at the single cell level.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The original contributions presented in the study are publicly available. This data can be found here: GSE249597 and <ext-link ext-link-type="uri" xlink:href="https://github.com/marintruchi/lncRNAs_CROPseq_SSAE">https://github.com/marintruchi/lncRNAs_CROPseq_SSAE</ext-link>.</p>
</sec>
<sec id="s6">
<title>Ethics statement</title>
<p>Ethical approval was not required for the studies on humans in accordance with the local legislation and institutional requirements because only commercially available established cell lines were used.</p>
</sec>
<sec id="s7">
<title>Author contributions</title>
<p>MT: Conceptualization, Writing&#x2013;original draft, Writing&#x2013;review and editing, Formal Analysis, Investigation, Methodology. CL: Formal Analysis, Investigation, Methodology, Writing&#x2013;review and editing. CG: Investigation, Methodology, Writing&#x2013;review and editing. JF: Investigation, Methodology, Writing&#x2013;review and editing. VM: Investigation, Writing&#x2013;review and editing. RL: Investigation, Writing&#x2013;review and editing. CG-R: Investigation, Writing&#x2013;review and editing. IM-P: Investigation, Writing&#x2013;review and editing. MG-I: Investigation, Writing&#x2013;review and editing. KL: Data curation, Investigation, Writing&#x2013;review and editing. PB: Resources, Writing&#x2013;review and editing. SS: Funding acquisition, Resources, Supervision, Writing&#x2013;review and editing. GV: Supervision, Writing&#x2013;review and editing, Investigation, Validation. RR: Investigation, Supervision, Validation, Writing&#x2013;review and editing, Funding acquisition. MB: Conceptualization, Formal Analysis, Investigation, Methodology, Supervision, Writing&#x2013;original draft, Writing&#x2013;review and editing. BM: Conceptualization, Funding acquisition, Project administration, Resources, Supervision, Writing&#x2013;original draft, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. We acknowledge the support from the Centre National de la Recherche Scientifique (CNRS), Universit&#xe9; C&#xf4;te d&#x2019;Azur, Cancerop&#xf4;le PACA (Action Structurante CRISPR SCREEN), the French Government (National Research Agency, ANR) program &#x201d;Investissements d&#x2019;Avenir&#x201d; UCAJEDI n ANR-15-IDEX-01 (PERTURB&#x2013;ENCODER) and ANR-22-CE17-0046-01 MIR-ASO, Plan Cancer 2018 &#x201c;ARN non-codants en canc&#xe9;rologie: du fondamental au translationnel&#x201d; (number 18CN045) and Fondation ARC (number PJA 20191209562).</p>
</sec>
<ack>
<p>The authors thank the technical support of the UCA GenomiX of the University C&#xf4;te d&#x2019;Azur, the CRISPR SCREEN UniCA platform, and the CoBioDA bioinformatics hub of the IPMC. We also thank A. Monteil and C. Lemmers from the Vectorology facility, PVM, Biocampus Montpellier, CNRS UMS3426. We are grateful to Sergio Covarrubias and Susan Carpenter (University of California Santa Cruz) for their advice and the exchange of equipment during the set-up of the CRISPR technology.</p>
</ack>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s11">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fbinf.2024.1340339/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fbinf.2024.1340339/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.PDF" id="SM1" mimetype="application/PDF" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table1.XLSX" id="SM2" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Adriaens</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Rambow</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Bervoets</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Silla</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Mito</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chiba</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>The long noncoding RNA NEAT1_1 is seemingly dispensable for normal tissue homeostasis and cancer cell growth</article-title>. <source>RNA (New York, N.Y.)</source> <volume>25</volume>, <fpage>1681</fpage>&#x2013;<lpage>1695</lpage>. <pub-id pub-id-type="doi">10.1261/rna.071456.119</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Altmann</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tolo&#x15f;i</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Sander</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Lengauer</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Permutation importance: a corrected feature importance measure</article-title>. <source>Bioinformatics</source> <volume>26</volume>, <fpage>1340</fpage>&#x2013;<lpage>1347</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btq134</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Amodio</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Raimondi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Juli</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Stamato</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Caracciolo</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Tagliaferri</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>MALAT1: a druggable long non-coding RNA for targeted anti-cancer approaches</article-title>. <source>J. Hematol. Oncol.</source> <volume>11</volume>, <fpage>63</fpage>. <pub-id pub-id-type="doi">10.1186/s13045-018-0606-4</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arun</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Aggarwal</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Spector</surname>
<given-names>D. L.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>MALAT1 long non-coding RNA: functional implications</article-title>. <source>Non-coding RNA</source> <volume>6</volume>, <fpage>22</fpage>. <pub-id pub-id-type="doi">10.3390/ncrna6020022</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barlaud</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Belhajali</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Combettes</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Fillatre</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Classification and regression using an outer approximation projection-gradient method</article-title>. <source>IEEE Trans. Signal Process.</source> <volume>65</volume>, <fpage>4635</fpage>&#x2013;<lpage>4644</lpage>. <pub-id pub-id-type="doi">10.1109/tsp.2017.2709262</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Barlaud</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chambolle</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Caillau</surname>
<given-names>J.-B.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Classification and feature selection using a primal-dual method and projection on structured constraints</article-title>,&#x201d; in <conf-name>2020 25th International Conference on Pattern Recognition (ICPR)</conf-name>, <conf-loc>Milan, Italy</conf-loc>, <conf-date>10-15 Jan. 2021</conf-date> (<publisher-name>IEEE</publisher-name>), <fpage>6538</fpage>&#x2013;<lpage>6545</lpage>. <pub-id pub-id-type="doi">10.1109/ICPR48806.2021.9412873</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Barlaud</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Guyard</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Learning sparse deep neural networks using efficient structured projections on convex constraints for green ai</article-title>,&#x201d; in <conf-name>2020 25th International Conference on Pattern Recognition (ICPR)</conf-name>, <conf-loc>Milan, Italy</conf-loc> (<publisher-name>IEEE</publisher-name>), <fpage>1566</fpage>&#x2013;<lpage>1573</lpage>. <pub-id pub-id-type="doi">10.1109/ICPR48806.2021.9412162</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Barlaud</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Guyard</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2021a</year>). &#x201c;<article-title>Learning a sparse generative non-parametric supervised autoencoder</article-title>,&#x201d; in <conf-name>ICASSP 2021 - 2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</conf-name>, <conf-loc>Toronto, ON, Canada</conf-loc>, <conf-date>6-11 June 2021</conf-date> (<publisher-name>IEEE</publisher-name>), <fpage>3315</fpage>&#x2013;<lpage>3319</lpage>. <comment>ISSN: 2379-190X</comment>. <pub-id pub-id-type="doi">10.1109/ICASSP39728.2021.9414410</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Barlaud</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Guyard</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2021b</year>). &#x201c;<article-title>Learning a sparse generative non-parametric supervised autoencoder</article-title>,&#x201d; in <conf-name>Proceedings of the International Conference on Acoustics, Speech and Signal Processing</conf-name>, <conf-loc>TORONTO , Canada</conf-loc>.</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barth</surname>
<given-names>D. A.</given-names>
</name>
<name>
<surname>Prinz</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Teppan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jonas</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Klec</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Pichler</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Long-Noncoding RNA (lncRNA) in the regulation of hypoxia-inducible factor (HIF) in cancer</article-title>. <source>Non-coding RNA</source> <volume>6</volume>, <fpage>27</fpage>. <pub-id pub-id-type="doi">10.3390/ncrna6030027</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Bertero</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Rezzonico</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Pottier</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Mari</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Chapter three - impact of MicroRNAs in the cellular response to hypoxia</article-title>,&#x201d; in <conf-name>International Review of Cell and Molecular Biology of MiRNAs in Differentiation and Development</conf-name>. Editors <person-group person-group-type="editor">
<name>
<surname>Galluzzi,</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Vitale</surname>
<given-names>I.</given-names>
</name>
</person-group> (<publisher-name>Academic Press</publisher-name>), <fpage>91</fpage>&#x2013;<lpage>158</lpage>. <pub-id pub-id-type="doi">10.1016/bs.ircmb.2017.03.006</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Carlevaro-Fita</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lanz&#xf3;s</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Feuerbach</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Hong</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Mas-Ponte</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Pedersen</surname>
<given-names>J. S.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Cancer LncRNA Census reveals evidence for deep functional conservation of long noncoding RNAs in tumorigenesis</article-title>. <source>Commun. Biol.</source> <volume>3</volume>, <fpage>56</fpage>&#x2013;<lpage>16</lpage>. <comment>Number: 1 Publisher: Nature Publishing Group</comment>. <pub-id pub-id-type="doi">10.1038/s42003-019-0741-7</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Choudhry</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Harris</surname>
<given-names>A. L.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Advances in hypoxia-inducible factor biology</article-title>. <source>Cell Metab.</source> <volume>27</volume>, <fpage>281</fpage>&#x2013;<lpage>298</lpage>. <pub-id pub-id-type="doi">10.1016/j.cmet.2017.10.005</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Condat</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Fast projection onto the simplex and the l1 ball</article-title>. <source>Math. Program. Ser. A</source> <volume>158</volume>, <fpage>575</fpage>&#x2013;<lpage>585</lpage>. <pub-id pub-id-type="doi">10.1007/s10107-015-0946-6</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Datlinger</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Rendeiro</surname>
<given-names>A. F.</given-names>
</name>
<name>
<surname>Schmidl</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Krausgruber</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Traxler</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Klughammer</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Pooled CRISPR screening with single-cell transcriptome readout</article-title>. <source>Nat. Methods</source> <volume>14</volume>, <fpage>297</fpage>&#x2013;<lpage>301</lpage>. <comment>Number: 3 Publisher: Nature Publishing Group</comment>. <pub-id pub-id-type="doi">10.1038/nmeth.4177</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dong</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Xiong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yue</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hanley</surname>
<given-names>S. J. B.</given-names>
</name>
<name>
<surname>Kobayashi</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Todo</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Long non-coding RNA NEAT1: a novel target for diagnosis and therapy in human tumors</article-title>. <source>Front. Genet.</source> <volume>9</volume>, <fpage>471</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2018.00471</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Eltager</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Abdelaal</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Charrout</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mahfouz</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Reinders</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Makrodimitris</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Benchmarking variational autoencoders on cancer transcriptomics data</article-title>. <source>PLoS ONE</source> <volume>18</volume>, <fpage>e0292126</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0292126</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Emmert-Streib</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Tripathi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Dehmer</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>An introductory review of deep learning for prediction models with big data</article-title>. <source>Front. Artif. Intell.</source> <volume>3</volume>, <fpage>4</fpage>. <pub-id pub-id-type="doi">10.3389/frai.2020.00004</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Esposito</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Polidori</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Meise</surname>
<given-names>D. F.</given-names>
</name>
<name>
<surname>Pulido-Quetglas</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Chouvardas</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Forster</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Multi-hallmark long noncoding RNA maps reveal non-small cell lung cancer vulnerabilities</article-title>. <source>Cell Genomics</source> <volume>2</volume>, <fpage>100171</fpage>. <pub-id pub-id-type="doi">10.1016/j.xgen.2022.100171</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Frankle</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Carbin</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>The lottery ticket hypothesis: finding sparse, trainable neural networks</article-title>,&#x201d; in <conf-name>International Conference on Learning Representations</conf-name>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gapp</surname>
<given-names>B. V.</given-names>
</name>
<name>
<surname>Konopka</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Penz</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Dalal</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>B&#xfc;rckst&#xfc;mmer</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Bock</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Parallel reverse genetic screening in mutant human cells using transcriptomics</article-title>. <source>Mol. Syst. Biol.</source> <volume>12</volume>, <fpage>879</fpage>. <comment>Publisher: John Wiley and Sons, Ltd</comment>. <pub-id pub-id-type="doi">10.15252/msb.20166890</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gutschner</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>H&#xe4;mmerle</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ei&#xdf;mann</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hsu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hung</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>The noncoding RNA MALAT1 is a critical regulator of the metastasis phenotype of lung cancer cells</article-title>. <source>Cancer Res.</source> <volume>73</volume>, <fpage>1180</fpage>&#x2013;<lpage>1189</lpage>. <pub-id pub-id-type="doi">10.1158/0008-5472.CAN-12-2850</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hastie</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Rosset</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tibshirani</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>The entire regularization path for the support vector machine</article-title>. <source>J. Mach. Learn. Res.</source> <volume>5</volume>, <fpage>1391</fpage>&#x2013;<lpage>1415</lpage>. <pub-id pub-id-type="doi">10.5555/1005332.1044706</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hastie</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Tibshirani</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>1996</year>). <article-title>Discriminant analysis by Gaussian mixtures</article-title>. <source>J. R. Stat. Soc. Ser. B Methodol.</source> <volume>58</volume>, <fpage>155</fpage>&#x2013;<lpage>176</lpage>. <pub-id pub-id-type="doi">10.1111/j.2517-6161.1996.tb02073.x</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hill</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>McFaline-Figueroa</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Starita</surname>
<given-names>L. M.</given-names>
</name>
<name>
<surname>Gasperini</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Matreyek</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Packer</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>On the design of CRISPR-based single-cell molecular screens</article-title>. <source>Nat. Methods</source> <volume>15</volume>, <fpage>271</fpage>&#x2013;<lpage>274</lpage>. <comment>Number: 4 Publisher: Nature Publishing Group</comment>. <pub-id pub-id-type="doi">10.1038/nmeth.4604</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hinton</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Zemel</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>1994</year>). &#x201c;<article-title>Autoencoders, minimum description length and helmholtz free energy</article-title>,&#x201d; in <source>Advances in neural information processing systems</source>, <fpage>3</fpage>&#x2013;<lpage>10</lpage>.</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Hypoxia exposure upregulates MALAT-1 and regulates the transcriptional activity of PTB-associated splicing factor in A549 lung adenocarcinoma cells</article-title>. <source>Oncol. Lett.</source> <volume>16</volume>, <fpage>294</fpage>&#x2013;<lpage>300</lpage>. <comment>Publisher: Spandidos Publications</comment>. <pub-id pub-id-type="doi">10.3892/ol.2018.8637</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Huber</surname>
<given-names>P. J.</given-names>
</name>
</person-group> (<year>2011</year>). &#x201c;<article-title>Robust statistics</article-title>,&#x201d; in <source>International encyclopedia of statistical science</source>. Editor <person-group person-group-type="editor">
<name>
<surname>Lovric</surname>
<given-names>M.</given-names>
</name>
</person-group> (<publisher-loc>Berlin, Heidelberg</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>1248</fpage>&#x2013;<lpage>1251</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-642-04898-2_594</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kaelin</surname>
<given-names>W. G.</given-names>
</name>
<name>
<surname>Ratcliffe</surname>
<given-names>P. J.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Oxygen sensing by metazoans: the central role of the HIF hydroxylase pathway</article-title>. <source>Mol. Cell</source> <volume>30</volume>, <fpage>393</fpage>&#x2013;<lpage>402</lpage>. <pub-id pub-id-type="doi">10.1016/j.molcel.2008.04.009</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>J.-w.</given-names>
</name>
<name>
<surname>Tchernyshyov</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Semenza</surname>
<given-names>G. L.</given-names>
</name>
<name>
<surname>Dang</surname>
<given-names>C. V.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>HIF-1-mediated expression of pyruvate dehydrogenase kinase: a metabolic switch required for cellular adaptation to hypoxia</article-title>. <source>Cell Metab.</source> <volume>3</volume>, <fpage>177</fpage>&#x2013;<lpage>185</lpage>. <pub-id pub-id-type="doi">10.1016/j.cmet.2006.02.002</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xing</surname>
<given-names>E. P.</given-names>
</name>
</person-group> (<year>2010</year>). &#x201c;<article-title>Tree-guided group lasso for multi-task regression with structured sparsity</article-title>,&#x201d; in <conf-name>Proceedings of the 27th International Conference on International Conference on Machine Learning (USA: Omnipress)</conf-name> (<publisher-name>ICML&#x2019;10</publisher-name>), <fpage>543</fpage>&#x2013;<lpage>550</lpage>.</citation>
</ref>
<ref id="B32">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Kingma</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Welling</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>Auto-encoding variational bayes</article-title>,&#x201d; in <conf-name>International Conference on Learning Representation</conf-name>.</citation>
</ref>
<ref id="B33">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kingma</surname>
<given-names>D. P.</given-names>
</name>
<name>
<surname>Ba</surname>
<given-names>L. J.</given-names>
</name>
</person-group> (<year>2015</year>). <source>Adam: a method for stochastic optimization publisher: ithaca, NYarXiv.org</source>.</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Koh</surname>
<given-names>M. Y.</given-names>
</name>
<name>
<surname>Powis</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Passing the baton: the HIF switch</article-title>. <source>Trends Biochem. Sci.</source> <volume>37</volume>, <fpage>364</fpage>&#x2013;<lpage>372</lpage>. <pub-id pub-id-type="doi">10.1016/j.tibs.2012.06.004</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Le</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Patterson</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>White</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2018</year>). <source>Supervised autoencoders: improving generalization performance with unsupervised regularizers</source>.</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Leclercq</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Vittrant</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Martin-Magniette</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Scott Boyer</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Perin</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Bergeron</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Large-scale automatic feature selection for biomarker discovery in high-dimensional omics data</article-title>. <source>Front. Genet.</source> <volume>16</volume> (<issue>10</issue>), <fpage>452</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2019.00452</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Horlbeck</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Cho</surname>
<given-names>S. W.</given-names>
</name>
<name>
<surname>Birk</surname>
<given-names>H. S.</given-names>
</name>
<name>
<surname>Malatesta</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>CRISPRi-based genome-scale identification of functional long noncoding RNA loci in human cells</article-title>. <source>Sci. (New York, N.Y.)</source> <volume>355</volume>, <fpage>aah7111</fpage>. <pub-id pub-id-type="doi">10.1126/science.aah7111</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lopez</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Regier</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cole</surname>
<given-names>M. B.</given-names>
</name>
<name>
<surname>Jordan</surname>
<given-names>M. I.</given-names>
</name>
<name>
<surname>Yosef</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Deep generative modeling for single-cell transcriptomics</article-title>. <source>Nat. Methods</source> <volume>15</volume>, <fpage>1053</fpage>&#x2013;<lpage>1058</lpage>. <comment>Number: 12 Publisher: Nature Publishing Group</comment>. <pub-id pub-id-type="doi">10.1038/s41592-018-0229-2</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lotfollahi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wolf</surname>
<given-names>F. A.</given-names>
</name>
<name>
<surname>Theis</surname>
<given-names>F. J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Scgen predicts single-cell perturbation responses</article-title>. <source>Nat. Methods</source> <volume>16</volume>, <fpage>715</fpage>&#x2013;<lpage>721</lpage>. <pub-id pub-id-type="doi">10.1038/s41592-019-0494-8</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Lundberg</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>S.-I.</given-names>
</name>
</person-group> (<year>2017</year>). <source>A unified approach to interpreting model predictions</source>. <publisher-loc>Barcelone, Spain</publisher-loc>: <publisher-name>Neural Information Processing Systems</publisher-name>, <fpage>30</fpage>.</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mimitou</surname>
<given-names>E. P.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Montalbano</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Stoeckius</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Legut</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Multiplexed detection of proteins, transcriptomes, clonotypes and CRISPR perturbations in single cells</article-title>. <source>Nat. Methods</source> <volume>16</volume>, <fpage>409</fpage>&#x2013;<lpage>412</lpage>. <pub-id pub-id-type="doi">10.1038/s41592-019-0392-0</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moreno Leon</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gautier</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Allan</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Ili&#xe9;</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Nottet</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Pons</surname>
<given-names>N.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>The nuclear hypoxia-regulated NLUCAT1 long non-coding RNA contributes to an aggressive phenotype in lung adenocarcinoma through regulation of oxidative stress</article-title>. <source>Oncogene</source> <volume>38</volume>, <fpage>7146</fpage>&#x2013;<lpage>7165</lpage>. <pub-id pub-id-type="doi">10.1038/s41388-019-0935-y</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Papalexi</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Mimitou</surname>
<given-names>E. P.</given-names>
</name>
<name>
<surname>Butler</surname>
<given-names>A. W.</given-names>
</name>
<name>
<surname>Foster</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bracken</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Mauck</surname>
<given-names>W. M.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Characterizing the molecular regulation of inhibitory immune checkpoints with multimodal single-cell screens</article-title>. <source>Nat. Genet.</source> <volume>53</volume>, <fpage>322</fpage>&#x2013;<lpage>331</lpage>. <pub-id pub-id-type="doi">10.1038/s41588-021-00778-2</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Perez</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Barlaud</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Fillatre</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>R&#xe9;gin</surname>
<given-names>J.-C.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A filtered bucket-clustering method for projection onto the simplex and the <italic>&#x2113;</italic>
<sub>1</sub>-ball</article-title>. <source>Math. Program.</source> <volume>182</volume>, <fpage>445</fpage>&#x2013;<lpage>464</lpage>. <pub-id pub-id-type="doi">10.1007/s10107-019-01401-3</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Perez</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Condat</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Barlaud</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <source>Near-linear time projection onto the l1,infty ball application to sparse autoencoders</source>. <comment>
<italic>arXiv: 2307.09836</italic>
</comment>.</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rankin</surname>
<given-names>E. B.</given-names>
</name>
<name>
<surname>Giaccia</surname>
<given-names>A. J.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Hypoxic control of metastasis</article-title>. <source>Sci. (New York, N.Y.)</source> <volume>352</volume>, <fpage>175</fpage>&#x2013;<lpage>180</lpage>. <pub-id pub-id-type="doi">10.1126/science.aaf4405</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rankin</surname>
<given-names>E. B.</given-names>
</name>
<name>
<surname>Nam</surname>
<given-names>J.-M.</given-names>
</name>
<name>
<surname>Giaccia</surname>
<given-names>A. J.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Hypoxia: signaling the metastatic cascade</article-title>. <source>Trends Cancer</source> <volume>2</volume>, <fpage>295</fpage>&#x2013;<lpage>304</lpage>. <pub-id pub-id-type="doi">10.1016/j.trecan.2016.05.006</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Raval</surname>
<given-names>R. R.</given-names>
</name>
<name>
<surname>Lau</surname>
<given-names>K. W.</given-names>
</name>
<name>
<surname>Tran</surname>
<given-names>M. G. B.</given-names>
</name>
<name>
<surname>Sowter</surname>
<given-names>H. M.</given-names>
</name>
<name>
<surname>Mandriota</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.-L.</given-names>
</name>
<etal/>
</person-group> (<year>2005</year>). <article-title>Contrasting properties of hypoxia-inducible factor 1 (HIF-1) and HIF-2 in von Hippel-Lindau-associated renal cell carcinoma</article-title>. <source>Mol. Cell. Biol.</source> <volume>25</volume>, <fpage>5675</fpage>&#x2013;<lpage>5686</lpage>. <pub-id pub-id-type="doi">10.1128/MCB.25.13.5675-5686.2005</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Replogle</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Norman</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hussmann</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cogan</surname>
<given-names>J. Z.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Combinatorial single-cell CRISPR screens by direct guide RNA capture and targeted sequencing</article-title>. <source>Nat. Biotechnol.</source> <volume>38</volume>, <fpage>954</fpage>&#x2013;<lpage>961</lpage>. <comment>Number: 8 Publisher: Nature Publishing Group</comment>. <pub-id pub-id-type="doi">10.1038/s41587-020-0470-y</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Semenza</surname>
<given-names>G. L.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Hypoxia-inducible factors in physiology and medicine</article-title>. <source>Cell</source> <volume>148</volume>, <fpage>399</fpage>&#x2013;<lpage>408</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2012.01.021</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Slack</surname>
<given-names>F. J.</given-names>
</name>
<name>
<surname>Chinnaiyan</surname>
<given-names>A. M.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>The role of non-coding RNAs in oncology</article-title>. <source>Cell</source> <volume>179</volume>, <fpage>1033</fpage>&#x2013;<lpage>1055</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2019.10.017</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Snoek</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Adams</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Larochelle</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2012</year>). &#x201c;<article-title>On non parametric guidance for learning autoencoder representations (PMLR)</article-title>,&#x201d; in <conf-name>Proceedings of the Fifteenth International Conference on Artificial Intelligence and Statistics</conf-name> (<publisher-name>PMLR</publisher-name>), <fpage>1073</fpage>&#x2013;<lpage>1080</lpage>.</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stoeckius</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Houck-Loomis</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Hao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yeung</surname>
<given-names>B. Z.</given-names>
</name>
<name>
<surname>Mauck</surname>
<given-names>W. M.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Cell Hashing with barcoded antibodies enables multiplexing and doublet detection for single cell genomics</article-title>. <source>Genome Biol.</source> <volume>19</volume>, <fpage>224</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-018-1603-1</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tamang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Acharya</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Roy</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Sharma</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Aryaa</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sharma</surname>
<given-names>U.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>SNHG12: an LncRNA as a potential therapeutic target and biomarker for human cancer</article-title>. <source>Front. Oncol.</source> <volume>9</volume>, <fpage>901</fpage>. <pub-id pub-id-type="doi">10.3389/fonc.2019.00901</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Townes</surname>
<given-names>F. W.</given-names>
</name>
<name>
<surname>Hicks</surname>
<given-names>S. C.</given-names>
</name>
<name>
<surname>Aryee</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Irizarry</surname>
<given-names>R. A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Feature selection and dimension reduction for single-cell RNA-seq based on a multinomial model</article-title>. <source>Genome Biol.</source> <volume>20</volume>, <fpage>295</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-019-1861-6</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vincent</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Larochelle</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Lajoie</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Manzagol</surname>
<given-names>P.-A.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Stacked denoising autoencoders: learning useful representations in a deep network with a local denoising criterion</article-title>. <source>J. Mach. Learn. Res.</source> <volume>11</volume>, <fpage>3371</fpage>&#x2013;<lpage>3408</lpage>. <pub-id pub-id-type="doi">10.5555/1756006.1953039</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yuan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Model selection and estimation in regression with grouped variables</article-title>. <source>J. R. Stat. Soc. Ser. B Stat. Methodol.</source> <volume>68</volume>, <fpage>49</fpage>&#x2013;<lpage>67</lpage>. <pub-id pub-id-type="doi">10.1111/j.1467-9868.2005.00532.x</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Lan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Yosinski</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Deconstructing lottery tickets: zeros, signs, and the supermask</article-title>,&#x201d; in <source>Advances in neural information processing systems (curran associates, inc.)</source>, <volume>32</volume>.</citation>
</ref>
</ref-list>
</back>
</article>