<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Immunol.</journal-id>
<journal-title>Frontiers in Immunology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Immunol.</abbrev-journal-title>
<issn pub-type="epub">1664-3224</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fimmu.2024.1394965</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Immunology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Transcriptomics profiling of the non-small cell lung cancer microenvironment across disease stages reveals dual immune cell-type behaviors</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes" corresp="yes">
<name>
<surname>Hurtado</surname>
<given-names>Marcelo</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2021;</sup>
</xref>
<xref ref-type="author-notes" rid="fn005">
<sup>&#xa7;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2672140"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Khajavi</surname>
<given-names>Leila</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<xref ref-type="author-notes" rid="fn004">
<sup>&#x2021;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2321242"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Essabbar</surname>
<given-names>Abdelmounim</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1151519"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Kammer</surname>
<given-names>Michael</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2618004"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xie</surname>
<given-names>Ting</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Coullomb</surname>
<given-names>Alexis</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2798363"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Pradines</surname>
<given-names>Anne</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Casanova</surname>
<given-names>Anne</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Kruczynski</surname>
<given-names>Anna</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/19281"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Gouin</surname>
<given-names>Sandrine</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2691026"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Clermont</surname>
<given-names>Estelle</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Boutillet</surname>
<given-names>L&#xe9;a</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2721355"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Senosain</surname>
<given-names>Maria Fernanda</given-names>
</name>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<xref ref-type="aff" rid="aff7">
<sup>7</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/828661"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zou</surname>
<given-names>Yong</given-names>
</name>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<xref ref-type="aff" rid="aff7">
<sup>7</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Zhao</surname>
<given-names>Shillin</given-names>
</name>
<xref ref-type="aff" rid="aff8">
<sup>8</sup>
</xref>
<xref ref-type="author-notes" rid="fn004">
<sup>&#x2021;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1519404"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Burq</surname>
<given-names>Prosper</given-names>
</name>
<xref ref-type="aff" rid="aff9">
<sup>9</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Mahfoudi</surname>
<given-names>Abderrahim</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Besse</surname>
<given-names>Jerome</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Launay</surname>
<given-names>Pierre</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2832408"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Passioukov</surname>
<given-names>Alexandre</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2793981"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chetaille</surname>
<given-names>Eric</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Favre</surname>
<given-names>Gilles</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Maldonado</surname>
<given-names>Fabien</given-names>
</name>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Cruzalegui</surname>
<given-names>Francisco</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Delfour</surname>
<given-names>Olivier</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2831417"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Mazi&#xe8;res</surname>
<given-names>Julien</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1080620"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Pancaldi</surname>
<given-names>Vera</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff10">
<sup>10</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/127214"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>CRCT, Universit&#xe9; de Toulouse, Institut national de la sant&#xe9; et de la recherche m&#xe9;dicale (Inserm), Centre national de la recherche scientifique (CNRS), Universit&#xe9; Toulouse III-Paul Sabatier, Centre de Recherches en canc&#xe9;rologie de Toulouse</institution>, <addr-line>Toulouse</addr-line>, <country>France</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Medicine, Vanderbilt University Medical Center</institution>, <addr-line>Nashville, TN</addr-line>, <country>United States</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Laboratory Medicine, Oncopole Claudius Regaud</institution>, <addr-line>Toulouse</addr-line>, <country>France</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Institut de Recherche Pierre Fabre</institution>, <addr-line>Toulouse</addr-line>, <country>France</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Pulmonology Department, Larrey Hospital, University Hospital of Toulouse</institution>, <addr-line>Toulouse</addr-line>, <country>France</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Division of Allergy, Pulmonary, and Critical Care Medicine, Department of Medicine, Vanderbilt University Medical Center</institution>, <addr-line>Nashville, TN</addr-line>, <country>United States</country>
</aff>
<aff id="aff7">
<sup>7</sup>
<institution>Cancer Early Detection and Prevention Initiative, Vanderbilt-Ingram Cancer Center, Vanderbilt University Medical. Center</institution>, <addr-line>Nashville, TN</addr-line>, <country>United States</country>
</aff>
<aff id="aff8">
<sup>8</sup>
<institution>Department of Biostatistics, Vanderbilt University Medical Center</institution>, <addr-line>Nashville, TN</addr-line>, <country>United States</country>
</aff>
<aff id="aff9">
<sup>9</sup>
<institution>Data Science, Centre Hospitalier Universitaire de Toulouse</institution>, <addr-line>Toulouse</addr-line>, <country>France</country>
</aff>
<aff id="aff10">
<sup>10</sup>
<institution>Life Sciences Department, Barcelona Supercomputing Center</institution>, <addr-line>Barcelona</addr-line>, <country>Spain</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Chiara Porta, University of Eastern Piedmont, Italy</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Xinjun Wang, Memorial Sloan Kettering Cancer Center, United States</p>
<p>Davide Cora, University of Eastern Piedmont, Italy</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Marcelo Hurtado, <email xlink:href="mailto:marcelo.hurtado@inserm.fr">marcelo.hurtado@inserm.fr</email>; Vera Pancaldi, <email xlink:href="mailto:vera.pancaldi@inserm.fr">vera.pancaldi@inserm.fr</email>
</p>
</fn>
<fn fn-type="present-address" id="fn003">
<p>&#x2020;Present addresses: Leila Khajavi, Bioinformatics Department, Evotec, Toulouse, France; Ting Xie, Institut national de la sant&#xe9; et de la recherche m&#xe9;dicale (INSERM) U981, Gustave Roussy Institute, Universit&#xe9; Paris-Saclay, Paris, France; Alexis Coullomb, RESTORE Research Center, Universit&#xe9; de Toulouse, INSERM 1301, Centre national de la recherche scientifique (CNRS) 5070, &#xc9;tablissement fran&#xe7;ais du sang (EFS), &#xc9;cole nationale v&#xe9;t&#xe9;rinaire de Toulouse (ENVT), Toulouse, France; L&#xe9;a Boutillet, Facult&#xe9; de M&#xe9;decine et de Pharmacie, Universit&#xe9; de Poitiers, France; Abderrahim Mahfoudi, Abdul Latif Jameel Health, Dubai, United Arab Emirates; Eric Chetaille, EC Medical Consulting, Paris, France; Francisco Cruzalegui, In Vitro Pharmacology Department, Evotec, Toulouse, France</p>
</fn>
<fn fn-type="equal" id="fn004">
<p>&#x2021;These authors have contributed equally to this work</p>
</fn>
<fn fn-type="other" id="fn005">
<p>&#xa7;Equipe Labellis&#xe9;e LIGUE Contre le Cancer</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>31</day>
<month>10</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1394965</elocation-id>
<history>
<date date-type="received">
<day>02</day>
<month>03</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>24</day>
<month>09</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Hurtado, Khajavi, Essabbar, Kammer, Xie, Coullomb, Pradines, Casanova, Kruczynski, Gouin, Clermont, Boutillet, Senosain, Zou, Zhao, Burq, Mahfoudi, Besse, Launay, Passioukov, Chetaille, Favre, Maldonado, Cruzalegui, Delfour, Mazi&#xe8;res and Pancaldi</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Hurtado, Khajavi, Essabbar, Kammer, Xie, Coullomb, Pradines, Casanova, Kruczynski, Gouin, Clermont, Boutillet, Senosain, Zou, Zhao, Burq, Mahfoudi, Besse, Launay, Passioukov, Chetaille, Favre, Maldonado, Cruzalegui, Delfour, Mazi&#xe8;res and Pancaldi</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Background</title>
<p>Lung cancer is the leading cause of cancer death worldwide, with poor survival despite recent therapeutic advances. A better understanding of the complexity of the tumor microenvironment is needed to improve patients&#x2019; outcome.</p>
</sec>
<sec>
<title>Methods</title>
<p>We applied a computational immunology approach (involving immune cell proportion estimation by deconvolution, transcription factor activity inference, pathways and immune scores estimations) in order to characterize bulk transcriptomics of 62 primary lung adenocarcinoma (LUAD) samples from patients across disease stages. Focusing specifically on early stage samples, we validated our findings using an independent LUAD cohort with 70 bulk RNAseq and 15 scRNAseq datasets and on TCGA datasets.</p>
</sec>
<sec>
<title>Results</title>
<p>Through our methodology and feature integration pipeline, we identified groups of immune cells related to disease stage as well as potential immune response or evasion and survival. More specifically, we reported a duality in the behavior of immune cells, notably natural killer (NK) cells, which was shown to be associated with survival and could be relevant for immune response or evasion. These distinct NK cell populations were further characterized using scRNAseq data, showing potential differences in their cytotoxic activity.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>The dual profile of several immune cells, most notably T-cell populations, have been discussed in the context of diseases such as cancer. Here, we report the duality of NK cells which should be taken into account in conjunction with other immune cell populations and behaviors in predicting prognosis, immune response or evasion.</p>
</sec>
</abstract>
<kwd-group>
<kwd>lung adenocarcinoma</kwd>
<kwd>natural killer cells</kwd>
<kwd>immune landscape</kwd>
<kwd>cell deconvolution</kwd>
<kwd>transcription factor activity</kwd>
</kwd-group>
<counts>
<fig-count count="13"/>
<table-count count="1"/>
<equation-count count="0"/>
<ref-count count="63"/>
<page-count count="23"/>
<word-count count="10564"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Cancer Immunity and Immunotherapy</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Background</title>
<p>Lung adenocarcinoma exhibits diverse clinical behaviors, ranging from indolent to aggressive metastatic disease. However, the biological underpinnings of this heterogeneity remain poorly understood. Non Small Cell Lung Cancer (NSCLC) is often diagnosed at an advanced stage and its management is currently undergoing significant transformation. Molecular testing, targeted therapies, and immunotherapy are now part of routine clinical care (<xref ref-type="bibr" rid="B1">1</xref>). However, despite major progress in the therapeutic management of NSCLC cancer, many patients are still refractory to the initial treatment or develop resistance leading to tumor recurrence. Furthermore, the clinical and pathological diversity of NSCLC is associated with a highly complex genomic landscape and heterogenous immune tumor microenvironment. Interactions between tumor cells and the immune microenvironment are known to profoundly impact cancer pathogenesis and progression (<xref ref-type="bibr" rid="B2">2</xref>).</p>
<p>Lung cancer tumor biopsies contain a heterogeneous mix of cancer cells, healthy cells, immune cells, and extracellular factors that constitute the tumor microenvironment (TME). The specific composition and functional profiles of immune cells within the TME can profoundly influence tumor pathogenesis. Detailed characterization of immune cell diversity in the TME has therefore become a major goal in cancer research. However, dissecting the immune landscape from bulk tumor profiling remains challenging (<xref ref-type="bibr" rid="B3">3</xref>&#x2013;<xref ref-type="bibr" rid="B5">5</xref>). Single cell RNA sequencing enables high-resolution dissection of tumor-immune interactions, but remains prohibitively costly for large-scale or clinical applications (<xref ref-type="bibr" rid="B6">6</xref>). Additionally, each single cell isolation approach introduces distinct technical biases that can skew rare cell detection. Computational deconvolution approaches can leverage unique gene expression signatures to estimate immune cell subsets from bulk transcriptomics in a more accessible and standardized way (<xref ref-type="bibr" rid="B7">7</xref>). However, numerous deconvolution algorithms exist with little consensus on best practices. In this study, we performed an integrated analysis using bulk RNAseq and validating our results with single cell RNA sequencing data. We applied this multi-omics pipeline to understand heterogeneity specifically in the microenvironment of early-stage lung adenocarcinomas, for which could validate our results on an independent cohort and on early stage lung adenocarcinoma (LUAD) samples from TCGA. We further correlated immune deconvolution features with clinical outcomes, highlighting the potential value of our approaches to reveal clinically relevant cellular populations and potentially implicating distinct NK cell phenotypes in survival. By correlating the deconvolution immune cell estimates and inferred transcription factor activities, we aimed to overcome limitations of individual methods. This study provides a framework for robust characterization of tumor immune landscapes from bulk transcriptomics.</p>
</sec>
<sec id="s2">
<title>Methods</title>
<sec id="s2_1">
<title>Patient summary</title>
<p>The primary analysis cohort was derived from a pilot study stemming from a collaborative effort between l&#x2019;Institut Universitaire du Cancer de Toulouse (IUCT) and Institut de Recherche Pierre Fabre (IRPF) aimed at assessing the technical feasibility of developing molecular characterization of lung tumors in order to enrich the activities already initiated by the IUCT. Patients were enrolled in the study if they were diagnosed with non-small cell lung cancer (NSCLC). Patients were excluded from this study if they were treated for any NSCLC prior to study enrollment. All individuals involved signed a non-objection form to part-take in the research program under the LUNG PREDICT protocol. Blood samples were gathered as part of a collection declared to the Ministry of Research under the number DC-2011-1382. Tissue samples are the remaining parts of the whole tissue belonging to the patient coming from the tumor library of CHU Biological Resource Center (IUCT-O) declared to the Ministry of Research under the number DC-2008-463. All clinical, pathological and molecular data were prospectively collected. Patients&#x2019; therapeutics and outcome were collected overtime with a 33 months median follow-up.</p>
</sec>
<sec id="s2_2">
<title>Sample selection and extraction</title>
<p>A certified pathologist made the selection of slides with haematoxylin eosin slide coloration. The paraffin embedded block was cut, 1 HE to control the extraction and 4 to 16 sections of 10 <italic>&#xb5;</italic>m for the RNA extraction, which was performed with High Pure FFPE RNA extraction kit from Roche (Ref 0665077500). The purified RNA samples were analyzed with Fragment Analyser (Advanced Analytical Technologies Inc., Agilent Technologies, US) and High Sensitivity RNA Kit (DNF-472-0500, Agilent Technologies, US) to determine the RIN and the DV200 (percentage of RNA <italic>&#x2265;</italic> 200 bp).</p>
</sec>
<sec id="s2_3">
<title>RNA sequencing</title>
<p>The libraries were prepared with the KAPA RNA HyperPrep Kit with RiboErase (HMR) (Kapa/Roche KK8560) for whole transcriptome sequencing as recommended by the supplier using 1 <italic>&#xb5;</italic>g input of RNA. Briefly, rRNA was hybridized with DNA probes to 5S, 8.8S, 18S, 28S, 12S and 16S rRNA, then the hybrids were depleted by enzymatic depletion using RNAse H. After, DNA digestion and fragmentation with high temperature were done. First strand, second strand synthesis and A-tailing were performed. Next, adapters from 1.5-7 <italic>&#xb5;</italic>M depending on the DV200 were ligated and the library was amplified. Library size and quality were confirmed on Fragment Analyzer (Advanced Analytical Technologies Inc., Agilent Technologies, US) and High Sensitivity NGS Fragment Analysis Kit (DNF-474-0500, Agilent Technologies, US). Qubit (ThermoFisher Scientific, US) was used to quantify libraries. Samples were pooled in equimolar fashion (10nM), then denatured and 1.8 pM was sequenced on NextSeq 550 (Illumina, US) in pair-end sequencing (76 bp reads) and double index 8 bp with NextSeq 500/550 High Output kit v2.5, 150 cycles (20024907, Illumina, US) and 1% PhiX (FC-110-3001, Illumina, US).</p>
</sec>
<sec id="s2_4">
<title>Bulk RNAseq sample processing</title>
<p>Raw sequences were quality checked using FastQC (<xref ref-type="bibr" rid="B8">8</xref> (v0.11.2)) and FastqScreen (<xref ref-type="bibr" rid="B9">9</xref> (v0.15.2)) prior to aligning to the Homo sapiens primary genome sequence (Gencode: GRCh38, v27) using STAR (<xref ref-type="bibr" rid="B10">10</xref> (v2.7.10a)) with encode options. FastQC was again used to assess the mapping quality. RSEM (<xref ref-type="bibr" rid="B11">11</xref> (v1.3.1)) was used to generate the expression matrix (featureCounts from Rsubread R package (<xref ref-type="bibr" rid="B12">12</xref> (v1.22.2)) was used for validation data).</p>
</sec>
<sec id="s2_5">
<title>Differential expression analysis</title>
<p>Expression matrices from bulk RNAseq were analyzed with DESeq2 (<xref ref-type="bibr" rid="B13">13</xref> (v1.42.1)) in the R environment (<xref ref-type="bibr" rid="B14">14</xref>); R Core Team (<xref ref-type="bibr" rid="B15">15</xref>) (version v4.2.3, BioConductor version v3.9 (<xref ref-type="bibr" rid="B16">16</xref>, <xref ref-type="bibr" rid="B17">17</xref>) to identify differentially expressed genes (DEGs) between samples groups. ClusterProfiler (<xref ref-type="bibr" rid="B18">18</xref> (v4.4.4)) was used to classify the DEGs into KEGG pathways. Heatmaps were generated using both pheatmap (v1.0.12) and ComplexHeatmap (<xref ref-type="bibr" rid="B19">19</xref> (v2.0.0)) R packages. Volcano plots were generated using the EnhancedVolcano (<xref ref-type="bibr" rid="B20">20</xref> (v1.2.0)) R package. Counts were normalized by Log2(TPM + 1) using the R package ADImpute (<xref ref-type="bibr" rid="B21">21</xref> (v1.12.0)).</p>
</sec>
<sec id="s2_6">
<title>Pathway activity calculation</title>
<p>Log<sub>2</sub>(TPM + 1) counts were used to calculate pathway activities using the PROGENy database (<xref ref-type="bibr" rid="B22">22</xref>), a compendium of publicly available signaling perturbation experiments based on footprint genes to yield a common core of 14 signaling pathways. Pathways regulatory activities were calculated using the Multivariate Linear Model (MLM) from the package decoupleR (<xref ref-type="bibr" rid="B23">23</xref> (v2.9.7)).</p>
</sec>
<sec id="s2_7">
<title>Immune cell-type deconvolution</title>
<p>In computational biology, deconvolution is an approach to quantitatively estimate the proportions of cell types in a mixed sample (e.g. bulk RNAseq) based on the observed gene expression profiles for separate cell types. Log<sub>2</sub>(TPM + 1) (transcript per million) normalized raw counts were used to estimate immune cell-type proportions for lymphocytes (B, T and NK cells), myeloid cells (monocytes, macrophages and dendritic cells) as well as cancer, endothelial, eosinophils, plasma, myocytes, mast cells and cancer-associated fibroblasts (CAFs). These cell-type proportion estimates were obtained by applying different reference-based deconvolution methods and several cell type signatures (see <xref ref-type="supplementary-material" rid="SF10">
<bold>Supplementary Table&#xa0;1</bold>
</xref>). These methods can provide absolute cell abundance quantification using signatures derived from single cell and bulk RNA seq data.</p>
</sec>
<sec id="s2_8">
<title>Transcription factor activity inference</title>
<p>Log<sub>2</sub> (TPM + 1) counts were used to infer transcription factor (TF) activity. We use prior knowledge networks (PKN) to infer the activity of different TFs from the gene expression of its direct target genes quantified in the gene count matrix. We used CollecTRI (<xref ref-type="bibr" rid="B24">24</xref>) from the package decoupleR (<xref ref-type="bibr" rid="B23">23</xref> (v2.9.7)), a collection of transcriptional regulatory interactions, which provides regulons containing signed transcription factor (TF) - target gene interactions compiled from 12 different resources as database and VIPER (<xref ref-type="bibr" rid="B25">25</xref> (v1.30.0)) as the inference algorithm. Depending on the level of the counts and considering that one TF can have many targets and one target can be regulated by more than one TF, the algorithm can estimate the level of activity of the regulator based on correlation between gene expression values.</p>
</sec>
<sec id="s2_9">
<title>Estimation of immune response scores estimation</title>
<p>Immune-scores were estimated on the TPM normalized raw counts using the EasieR package (<xref ref-type="bibr" rid="B26">26</xref> (v1.4.0)) to generate immune profiles on a per sample basis. Briefly, immune-scores are calculated using gene sets that have been validated in different publications (see <xref ref-type="supplementary-material" rid="SF11">
<bold>Supplementary Table&#xa0;2</bold>
</xref>) as signatures to estimate certain hallmarks of the immune response.</p>
</sec>
<sec id="s2_10">
<title>Feature selection</title>
<p>The Boruta algorithm was applied using the R package Boruta (<xref ref-type="bibr" rid="B27">27</xref> (v8.0.0)) using a bootstrapping approach to ensure consistency in the selection of features. Briefly, the algorithm performs feature selection and it was applied 100 times using different seeds, each time labeling features as &#x2018;Confirmed&#x2019;, &#x2018;Tentative&#x2019; or &#x2018;Rejected&#x2019;. Features labeled as &#x2018;Confirmed&#x2019; more than 90% of the times are finally selected.</p>
</sec>
<sec id="s2_11">
<title>Processing of deconvolution features</title>
<p>Applying several combinations of deconvolution methods and signatures leads to several hundreds of features describing the TME landscapes in the samples. We applied specifically 6 methods (quanTIseq, XCell, MCPcounter, DeconRNASeq, EpidISH and CibersortX) and 9 signatures (BPRNACan, BPRNACanPro, BPRNACan3DProMet, TIL10, LM22, CCLE.TIL10, CBSX.HNSCC.scRNAseq, CBSX.Melanoma.scRNAseq and CBSX.NSCLC.PBMCs.scRNAseq), see <xref ref-type="supplementary-material" rid="SF10">
<bold>Supplementary Table&#xa0;1</bold>
</xref> generating 351 features related to 13 cell types (and 30 subtypes). To reduce the dimensionality and eliminate redundancies we then applied a combination of unsupervised filtering techniques and iterative linear and proportionality based correlations within each cell type to form deconvolution feature subgroups. Applying an unsupervised approach, we removed features with a high proportion of zeros or low variance across samples. We then set out to eliminate redundant features calculating pairwise correlations of these filtered features to identify highly correlated (<italic>&#x2265;</italic> 0.7) feature pairs. We interpret these high correlations as evidence that those features are estimating the presence of the same cell-type despite potential differences in signature nomenclature and hence combine these features into a single feature subgroup. This procedure is carried out until no correlations above the specified threshold remain.</p>
</sec>
<sec id="s2_12">
<title>Processing of TF activity features</title>
<p>The other set of descriptors of our samples stem from TF activity analysis, which returns a score of TF activity for each TF in each sample, amounting to 769 features. Adapting the Weighted correlation network analysis (WGCNA) approach (<xref ref-type="bibr" rid="B28">28</xref> (v1.72-5)), we performed dimensionality reduction on these features by constructing what we defined as Weighted TFs co-activity networks (WTCNA) to detect highly correlated modules of TFs based on pairwise correlation of their inferred activity. Modules are defined as densely connected groups of nodes in the TF network, where connections represent correlation of activities, and they are arbitrarily named using colors. These TF modules were functionally characterized using pathway activities estimated for each sample (see Pathway Activity calculation above) and calculating the Pearson correlation between these TF module scores and the pathways activity scores. A PCA using the correlation matrix between the TFs module scores and the pathways activities allowed us to identify clusters of TF modules with correlated pathway activities, further grouping TF modules into broader functional groups. These combined TF module groups are named by combining the names of TF modules included, thus generating names that include multiple colors.</p>
</sec>
<sec id="s2_13">
<title>TF modules functional enrichment analysis</title>
<p>TFs module enrichment was done by identifying the hub TFs from each module, these are genes which play a central role in the network&#x2019;s module structure and function due to their high connectivity and influence on other genes. Thus, they often represent key regulators or drivers of important biological processes. We considered as hub TFs those which exhibit high module membership, meaning their activity is strongly correlated with the module&#x2019;s score, indicating that they are highly representative of the module&#x2019;s overall behavior. Also, since these genes are typically connected to many other genes within the network, we also considered the level of connectivity for the hub selection. We selected TFs with a high &#x201c;degree&#x201d;, a measure of the number of direct connections or edges a TF has with other TFs in the network. Overall, TFs with high module membership (r&gt;0.8) and belonging to the top 10% of genes with high degree were selected as hub TFs. From the hub TFs, we identified their corresponding target genes using the CollecTRI database (<xref ref-type="bibr" rid="B24">24</xref>). We considered only the top 20% most variable (based on gene expression) and unique target genes per TF module. Using these lists, we performed an over representation analysis (ORA) using the R package ReactomePA (<xref ref-type="bibr" rid="B29">29</xref> (v1.46.0)) and the Reactome database (<xref ref-type="bibr" rid="B30">30</xref> (v1.86.2)) to provide functional interpretation of the modules.</p>
</sec>
<sec id="s2_14">
<title>Integration of deconvolution and TF features</title>
<p>Using both deconvolution and TF activity features across samples we set out to define combined features as groups of cells that share TF activity profiles, potentially describing their phenotypic states. We performed hierarchical clustering using ward.D2 as the agglomeration method of the matrix of correlation between grouped deconvolution features and TF module scores. This leads to clustering of grouped deconvolution features that each refer to specific cell types, producing further grouping of <italic>different</italic> cell types. We refer to these as Cell type groups. The existence of these cell type groups suggests that several cell types could be activating specific biological processes, as reflected by similar activities of the TF modules, potentially revealing different cell states (e.g. cell growth profiles could be observed in cancer cells or fibroblasts by detecting similar TF activities across patients). These new cell type groups are new features composed of different grouped deconvolution features referring to different cell types, named using a specific nomenclature (e.g Dendrogram_red_turquoise.group_2) where &#x201c;dendrogram&#x201d; indicates that the feature came from a hierarchical clustering, colors indicate which TFs module were merged to produce the dendrogram and &#x201c;group_x&#x201d; refers to the actual cluster in this dendrogram. This definition of cell type groups implies cell types cluster together because they share similar biological activity as measured by the TFs activity profiles. Finally, the Boruta feature selection algorithm (see above) was applied to measure the importance of these integrated features in the classification of samples.</p>
</sec>
<sec id="s2_15">
<title>Validation cohort</title>
<p>An independent cohort consisting of 77 surgically resected adenocarcinomas (1 = stage 0, 44 = stage I, 26 = stage II and 6 = stage III) (<xref ref-type="bibr" rid="B31">31</xref>), from which only 70 early stage (I, II) samples were used to validate the findings from the primary analysis cohort. The validation cohort samples were collected at Vanderbilt University Medical Center, Nashville, TN, from treatment-naive patients undergoing surgical resection. The dataset included both bulk and scRNAseq samples, with 15 patients for this last one. From this, only 9 patients have both bulk and scRNAseq information.</p>
</sec>
<sec id="s2_16">
<title>Single-cell RNAseq analysis</title>
<p>Preprocessed single-cell RNAseq data was obtained from (<xref ref-type="bibr" rid="B31">31</xref>). The Seurat package (<xref ref-type="bibr" rid="B6">6</xref>, <xref ref-type="bibr" rid="B32">32</xref>&#x2013;<xref ref-type="bibr" rid="B34">34</xref> (v4.3.0.1)) was used for downstream analysis of the data in the R environment. We computed a principal component analysis for dimensionality reduction followed by the neighborhood graph on the first 20 principal components, using the elbow plot, obtaining 24 clusters. Cell annotation was done with the following references: Human Primary Cell Atlas, Immune Cell Expression, Monaco and Blueprint Encode using the celldex R package (<xref ref-type="bibr" rid="B35">35</xref> (v1.10.1)). A consensus of all three annotations was taken for identification of NK clusters. Reference-based deconvolution was done using the scRNAseq object and the BayesPrism method (<xref ref-type="bibr" rid="B36">36</xref> (v.2.0.0)), obtained from the Omnideconv R package (<xref ref-type="bibr" rid="B37">37</xref> (v.0.1.0)).</p>
</sec>
<sec id="s2_17">
<title>Survival analysis</title>
<p>Patients from the validation cohort with early stage disease (Stage I and II) were included in a survival analysis. &#x201c;Time&#x201d; is measured in days and &#x201c;event&#x201d; was defined as either death, recurrence or progression. Cox proportional hazards modeling was performed using the R packages rms (v.6.8.0) and survival (<xref ref-type="bibr" rid="B38">38</xref>, <xref ref-type="bibr" rid="B39">39</xref> (v3.5.5)), and Kaplan-Meier curves were prepared using ggplot2 (<xref ref-type="bibr" rid="B40">40</xref> (v3.4.3)) and Survminer (<xref ref-type="bibr" rid="B41">41</xref> (v0.4.9)). Univariate and multivariate cox proportional hazards (coxPH) models were evaluated across selected cell type groups to investigate whether the effect of a single or multiple cell type groups on the hazard of an event (death/progression/recurrence) was significant for the survival outcomes. After fitting the CoxPH models to different cell type groups combinations, we stratified our patients based on the linear predictors of the model (risk scores) from which we define as &#x2018;high&#x2019; the patients with risk scores above the median value of the cox model&#x2019;s linear predictors and as &#x2018;low&#x2019; the patients below it. We then performed a Kaplan Meier analysis and plotted the survival curves for each risk group. Finally both survival curves were assessed via a log rank test to see if there was a statistically significant difference between risk groups (p value &lt; 0.01).</p>
</sec>
<sec id="s2_18">
<title>TCGA analysis</title>
<p>Samples counts from TCGA were retrieved using TCGAbiolinks R package (<xref ref-type="bibr" rid="B42">42</xref>&#x2013;<xref ref-type="bibr" rid="B44">44</xref> (v2.30.4)). We selected open-access cases from the project TCGA-LUAD, using transcriptome profiling as data category, RNA-seq as experimental assay and STAR-counts as analysis workflow type. Applying these filters, 600 cases were retrieved. Since our focus was only on early stage samples (I, II) we selected the corresponding 399 cases. Survival analysis was done following the same pipeline described above; for this, the variables &#x201c;vital_status&#x201d;, &#x201c;days_to_last_follow_up&#x201d;, &#x201c;days_to_death&#x201d; were considered to determine the overall_survival (PFS) and whether the event (death) occurred. Six patients were removed from this analysis due to the presence of missing values in these variables.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<title>Results</title>
<sec id="s3_1">
<title>The Lung Predict cohort</title>
<p>Bulk RNAseq was performed on surgically resected tumor tissue or tumor biopsies from 82 patients with NSCLC, 62 of which were diagnosed as lung adenocarcinoma (LUAD) and were considered further (the &#x201c;Primary Analysis Cohort&#x201d;). Of these 62 adenocarcinomas, 30 are female and 32 are male; 21 were enrolled at Stage I, 10 at Stage II, 11 at Stage III and 20 at Stage IV. A full breakdown of this cohort is presented in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> and a patient inclusion flow chart is included in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>, together with details of the validation cohort (see the following sections).</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Summary of the total number of patients included in the Lung Predict and the Vanderbilt validation cohort (VUMC) (percentages of totals in brackets).</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left"/>
<th valign="top" colspan="2" align="center">Lung Predict</th>
<th valign="top" colspan="2" align="center">VUMC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">
<bold>Total</bold>
</td>
<td valign="top" colspan="2" align="center">62</td>
<td valign="top" colspan="2" align="center">77</td>
</tr>
<tr>
<td valign="top" align="left">
<bold>Sex (Female)</bold>
</td>
<td valign="top" align="center">30</td>
<td valign="top" align="center">
<italic>(48)</italic>
</td>
<td valign="top" align="center">42</td>
<td valign="top" align="center">
<italic>(55)</italic>
</td>
</tr>
<tr>
<td valign="top" align="left">
<bold>Age (&lt;70)</bold>
</td>
<td valign="top" align="center">46</td>
<td valign="top" align="center">
<italic>(74)</italic>
</td>
<td valign="top" align="center">47</td>
<td valign="top" align="center">
<italic>(61)</italic>
</td>
</tr>
<tr>
<td valign="top" align="left">
<bold>Smoking Status: Never</bold>
</td>
<td valign="top" align="center">10</td>
<td valign="top" align="center">
<italic>(16)</italic>
</td>
<td valign="top" align="center">14</td>
<td valign="top" align="center">
<italic>(18)</italic>
</td>
</tr>
<tr>
<td valign="top" align="left">
<bold>Smoking status: Former</bold>
</td>
<td valign="top" align="center">9</td>
<td valign="top" align="center">
<italic>(15)</italic>
</td>
<td valign="top" align="center">53</td>
<td valign="top" align="center">
<italic>(69)</italic>
</td>
</tr>
<tr>
<td valign="top" align="left">
<bold>Smoking status: Current</bold>
</td>
<td valign="top" align="center">43</td>
<td valign="top" align="center">
<italic>(69)</italic>
</td>
<td valign="top" align="center">10</td>
<td valign="top" align="center">
<italic>(13)</italic>
</td>
</tr>
<tr>
<th valign="top" colspan="5" align="left">Stage</th>
</tr>
<tr>
<td valign="top" align="left">0</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">
<italic>-</italic>
</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">(1)</td>
</tr>
<tr>
<td valign="top" align="left">I</td>
<td valign="top" align="center">21</td>
<td valign="top" align="center">
<italic>(34)</italic>
</td>
<td valign="top" align="center">44</td>
<td valign="top" align="center">
<italic>(57)</italic>
</td>
</tr>
<tr>
<td valign="top" align="left">II</td>
<td valign="top" align="center">10</td>
<td valign="top" align="center">
<italic>(16)</italic>
</td>
<td valign="top" align="center">26</td>
<td valign="top" align="center">
<italic>(34)</italic>
</td>
</tr>
<tr>
<td valign="top" align="left">III</td>
<td valign="top" align="center">11</td>
<td valign="top" align="center">
<italic>(18)</italic>
</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">(8)</td>
</tr>
<tr>
<td valign="top" align="left">IV</td>
<td valign="top" align="center">20</td>
<td valign="top" align="center">
<italic>(32)</italic>
</td>
<td valign="top" align="center"/>
<td valign="top" align="center">-</td>
</tr>
<tr>
<td valign="top" align="left">
<bold>Metastatic (non-primary)</bold>
</td>
<td valign="top" align="center">20</td>
<td valign="top" align="center">
<italic>(32)</italic>
</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">&#x2013;</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>An overview of the Lung Predict cohort. A description of the cohort is presented on the left with some summary graphics on the right specifically detailing tumor stages in male and female patients, RNAseq batches, presence of the most frequent somatic mutations as detected by a gene panel assay (STK11, EGFR, KRAS), smoking status, metastasis occurrence, age category, type of sample (primary or metastatic sample), location of the sample.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1394965-g001.tif"/>
</fig>
<p>We applied a computational immunology approach integrating several features derived from transcriptomics data to better characterize and profile the TME of LUAD tumor samples in our cohort. The features extracted included cell-type proportions, level of activity of specific Transcription Factors (TFs) and scores of immunogenicity commonly used in the literature.</p>
<p>Briefly, reference-based deconvolution involves applying statistical methods to infer cell type proportions in biological tissue samples starting from transcriptomic profiles of specific cell types (signatures) and bulk transcriptomics from the samples, such as tumoural tissues in this case. We applied several deconvolution methods to the transcriptomes from our LUAD samples and used different cell type signatures to generate estimates of cell type proportions (see Methods, see <xref ref-type="supplementary-material" rid="SF10">
<bold>Supplementary Table&#xa0;1</bold>
</xref>).</p>
<p>Normally, the application of dimensionality reduction methods, such as pathway activity analysis or the calculation of immune cell type proportions allows better interpretation of the signal from bulk gene expression data but at the cost of introducing artificial noise or removing potentially interesting data features. Selecting deconvolution methods is not trivial and the different results obtained with different methods and signatures suggest that they capture different aspects of the samples. In this study, we aimed to use a variety of different methods and signatures instead of choosing a single one. However, using several deconvolution methods and signatures, each covering a range of cell types, produced over 300 different deconvolution-related features, which led us to an increase in dimensionality, exposed high variability between features related to the same cell types, ultimately hindering interpretation and imposing the need for much larger sample sizes to achieve statistical power. This pushed us to address these issues by engineering novel approaches to produce meaningful cell deconvolution features integrated with TF activity profiles.</p>
<p>We therefore performed TF activity estimation, which is an approach to quantify the strength of activity of specific TFs (essentially an estimated combination of their abundance as proteins and their post-translational modifications if required for their activity) based on the expression level of their targets. These approaches involve a prior-knowledge network of TF-target interactions in combination with gene expression levels from bulk transcriptomics data and they allow to identify activation of specific regulons (TFs and their targets) despite the fact that TF activities are rarely regulated at the transcriptional level (see Methods). Complementary to this analysis, we have calculated scores of activation of specific pathways using PROGENy, which help us to define the processes that dominate the transcriptome of our patient samples (see Methods).</p>
<p>Finally, several scores have been proposed in the literature to estimate the level of immunogenicity in tumor samples from bulk transcriptomics data and we have calculated these immuno-scores across our cohort (see Methods).</p>
</sec>
<sec id="s3_2">
<title>Deconvolution features along with inferred TF activity profiles reveal different immune profiles describing the tumor microenvironment across patients of different stages</title>
<p>As a result of applying cell type deconvolution, we considered 351 features for each sample. Multiple signatures were used for each cell type, leading to several estimates of proportions of the same cell type (e.g. monocytes). This multiplication of features referred to the same cell type is due either to the fact that they capture different cell subtypes (e.g. classical or non-classical monocytes), or simply to differences in the generation of the signatures from the literature (from <italic>in-vitro</italic> co-cultures, from tumor samples, etc.). To reduce the redundancy and dimensionality in our data we first grouped deconvolution features quantifying the same cell types based on their correlation across patients and generated deconvolution feature subgroups (see Methods). Briefly, if multiple features estimating the same cell type have high correlation it suggests that they do not differ biologically and do not capture distinct subtypes, so we merge the corresponding features) (see <xref ref-type="supplementary-material" rid="SF12">
<bold>Supplementary Table&#xa0;3</bold>
</xref> for details).</p>
<p>We then performed unsupervised hierarchical clustering of patient samples based on the grouped deconvolution features, to identify patient clusters with correlated immune cell proportion. We identified three patient groups based on the grouped deconvolution features and interpreted them based on the immuno-scores in each sample. Patient cluster 1 contained mostly &#x201c;intermediate&#x201d; tumors, patient cluster 2 contained mainly &#x201c;hot&#x201d; tumors and patient cluster 3 was constituted by a mixture of &#x201c;cold/intermediate&#x201d; tumors (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>). We also visually notice more early stage samples (I, II) in patient cluster 1, a high presence of late stage samples (IV) in patient cluster 3 and a high presence of intermediate stages (III) in patient cluster 2.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Overview of patient sample clustering based on immune deconvolution subgroups. Our immune deconvolution features after being processed identified three clusters of patients corresponding to &#x201c;cold/intermediate&#x201d;, &#x201c;hot&#x201d; and intermediate tumors. Heatmap showing patients within the 3 clusters identified by hierarchical clustering. The gray scale at the top corresponds to the stage of the disease: the darker the color, the later the stage. The orange to brown scale corresponds to the immune-scores (see Methods): the darker the color, the higher the immune scores.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1394965-g002.tif"/>
</fig>
<p>To estimate the main processes driving the transcriptomic profiles of our samples, we calculated TF activities and constructed weighted co-activity TFs networks, identifying TF modules (named with colors), which are groups of TFs showing correlated activity profiles across samples (see Methods and <xref ref-type="supplementary-material" rid="SF13">
<bold>Supplementary Table&#xa0;4</bold>
</xref> for TF module composition). We then applied a Boruta feature selection approach to select the most important deconvolution features driving the patient classification into these three patient clusters, identifying 27 deconvolution features to be the most influential.</p>
<p>To further investigate the mechanistic processes that might underlie the 3 patient clusters, we observed how different features (cell type composition, TF module scores, and pathway scores) correlated with each other across patients (c.f. <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>). We note that several cell types appear in multiple rows as separated features, possibly indicating that the different signatures capture distinct cell subtypes. For example we observe multiple features related to NK cells and B cells. The name of the feature reflects the name of the public signature that generated this feature and often suggests which subtype is captured (activated/naive etc.), while the names including &#x2018;subgroup&#x2019; denote several features that were combined in the earlier step of deconvolution feature grouping since they displayed strong correlations across patients. We characterized the 3 patient subgroups as follows (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>) according to the values of different cell type features:</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Annotation of the three patient clusters from (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>) using TFs module scores, pathways scores, and values of Boruta selected immune deconvolution features. <bold>(A)</bold> Three feature groups (in rows) are identified from the deconvolution features (values shown on scale red to blue from high to low). The panel also shows as column annotations of each sample the immuno-score (brown to white from high to low), the TF module scores (red to green from high to low, see composition of each module in (<xref ref-type="supplementary-material" rid="SF13">
<bold>Supplementary Table&#xa0;4</bold>
</xref>) and the pathway scores (yellow to blue from high to low). <bold>(B)</bold> Heatmap showing significant Pearson correlation between pathway activities and TF modules scores shown in panel A (denoted by the colors at the top of the columns: blue, cyan, yellow, brown, red, black, green, pink from left to right). Heatmap colors represent levels of correlation (darker red implies high positive correlation, darker blue implies high negative correlation). Statistics are shown as text only for significant correlations (p value &lt; 0.05).</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1394965-g003.tif"/>
</fig>
<p>
<italic>Patient cluster 1 (&#x201c;intermediate&#x201d; tumors):</italic> associated to low presence of cancer cells, several CAF signatures, some myeloid cells (M1 macrophages and monocytes), some lymphocytes (CD4 T helper), some type of unspecified NK cells and higher abundance of B cells, resting CD4 and dendritic cells with NK cells denoted as activated. TF activity analysis showed an involvement of TFs modules yellow, brown, red and blue, involved in biological pathway activities related to Androgen, Trail and p53, suggesting a relation with apoptosis and tumor suppression.</p>
<p>
<italic>Patient cluster 2 (&#x201c;hot&#x201d; tumors):</italic> associated to intermediate presence of cancer cells, several CAF signatures, some myeloid cells (M1 macrophages and monocytes), some lymphocytes (CD4 T helper) and some type of unspecified NK cells and low abundance of naive B cells, resting CD4 and dendritic cells and NK cells denoted as activated with varying levels of a group of non-naive B cells. TF activity analysis showed high scores in modules black, green, red and brown, which seems to be related to high scores of JAK/STAT, VEGF, MAPK and hypoxia pathways, as well as low levels of modules blue, and yellow, with particularly low scores for Trail and p53, suggesting activation of immunity, stress response and proliferation.</p>
<p>
<italic>Patient cluster 3 (&#x201c;cold/intermediate&#x201d; tumors):</italic> mostly late stage, showing particularly high proportions of cancer cells, CAF cells and some macrophages. TF activity showed particularly high scores for module black and low scores for Trail, p53 but also NFkB, VEGF and JAK/STAT, MAPK, suggesting a highly aggressive, immunosuppressive and proliferative phenotype.</p>
<p>To better interpret the duality of specific cell type features we consider how they correlate with each other (row feature groups 1 to 3 on <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>).</p>
<p>Interestingly, we identified a complex behavior profile for the different features estimating the presence of the same cell types. In some cases, signature names can suggest which cell subtype we are considering but there are known issues with signatures for myeloid cell subtypes, for example, despite the importance of these details to understand whether the TME is immunosuppressive or not. Here we focus on NK cells, for which several signatures appear to give conflicting results. The first NK profile (exemplified by the <italic>NK CBSX_melanoma &#x2026;</italic> feature from feature group 1) is associated with a presence of cancer cells and CAFs and is found in samples with lower immune scores (subset of Patient cluster 3). This profile may imply the presence of dysfunctional NK cells, which are characterized by reduced proliferation and cytotoxic capabilities. The other profile (NK from <italic>EpiDISH_CCLE_NK &#x2026;</italic> from feature group 2) is associated with endothelial cells and the presence of certain B and CD4 T-cells found in samples with intermediate immune-scores, perhaps signifying the presence of tertiary lymphoid structures (TLS), organized immune cell aggregates that can be good prognosis markers when identified through spatial omics. Another group of NK cells (NK from feature group 3) more associated with the presence of neutrophils, dendritic and M2 macrophages does not seem to be associated to the 3 patient clusters identified, showing variable values in all sample clusters, Taken together, these findings suggest that NK cells of different kinds, associated with other immune cells, can be found in either immune-desert tumors, where they are likely to be dysfunctional, typically in late stage samples, but also in intermediate tumors and early stage samples, where they can be associated with different kinds of immune landscapes, depending on their partners.</p>
</sec>
<sec id="s3_3">
<title>Data integration to uncover associations between cell-type deconvolution features and TF activity profiles</title>
<p>Having observed interesting associations between TF activity and pathway scores and TME landscapes, and the duality of certain cell types (NK for example) we decided to investigate whether a combination of these features could reveal connections between cell states across the different cell types in the TME. As an example, we reasoned that the presence of specific cytokines in the tumor could have an impact on the state of specific immune cells (say cytotoxicity of NK cells) in specific patients. We therefore set out to develop a computational method to integrate cell type proportion estimates and TF activity scores to evaluate the state of the different cell populations present in the samples.</p>
<p>Briefly, we start by considering the grouped deconvolution features and TF module activity scores as descriptors of our samples. Since TF module scores and grouped deconvolution features can both be calculated in each sample, we can visualize the association between each TF module and the different grouped deconvolution features as a heatmap. Hierarchical clustering of this matrix (deconvolution feature by TF module) allows us to cluster deconvolution features, even grouping those that estimate proportions of different cell types, allowing us to define Cell Type Groups (see Methods). The appearance of clusters of deconvolution features referring to different cell types but having similar TF activity profiles suggests some commonality of biological processes ongoing in the distinct cell types present in specific patients (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>).</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Overview of the deconvolution and TF activity integration pipeline. Immune cell deconvolution features and modules of TFs sharing inferred activity profiles are integrated together using a combination of clustering methods in order to reduce the dimensionality of the results (see Methods). The output are groups of immune cells characterized by the same TF activities.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1394965-g004.tif"/>
</fig>
</sec>
<sec id="s3_4">
<title>Integrated analysis of cell type composition and TF activity profiles in early stage LUAD samples uncovers two distinct patient groups</title>
<p>Our results highlighted a possible difference in the immune profile of samples according to stage, with most late stage samples (stage IV) being in &#x2018;cold&#x2019; patient groups. To avoid any confounding effect of stage and sample type (primary vs. metastasis biopsy) and reduce the heterogeneity of processes likely to take place in our samples, we decided to focus on the early stage samples (stages I and II).</p>
<p>To assess the immune landscape in the stage I and II samples of the Lung Predict cohort, we performed immune cell-type deconvolution and inferred TF activities across these samples (see Methods).</p>
<p>Focusing specifically on stage I and II from the Lung Predict cohort (31 samples), we applied our integrative approach to combine grouped deconvolution features and TF module activity scores. (99 deconvolution features including 46 cell subgroups (<xref ref-type="supplementary-material" rid="SF14">
<bold>Supplementary Table&#xa0;5</bold>
</xref>) and 53 non-grouped features, 7 TF modules (<xref ref-type="supplementary-material" rid="SF15">
<bold>Supplementary Table&#xa0;6</bold>
</xref>), each containing different groups of TFs (<xref ref-type="supplementary-material" rid="SF1">
<bold>Supplementary Figure&#xa0;1A</bold>
</xref>) correlating with different biological activities (<xref ref-type="supplementary-material" rid="SF1">
<bold>Supplementary Figure&#xa0;1B</bold>
</xref>).</p>
<p>In order to further study the composition of these modules, we identified the most central (hub) TFs in each module (see Methods), which highlighted 20 hub TFs in total (6 for module red, 6 for module brown, 3 for module black, 2 for module green and 3 for module blue). No hub TFs were found for module turquoise and yellow (<xref ref-type="supplementary-material" rid="SF1">
<bold>Supplementary Figure&#xa0;1C</bold>
</xref>). Further enrichment of these TFs modules was done by identifying the corresponding target genes of the hub TFs (see Methods). Using only target genes that belong to only one module to avoid overlapping ones, we performed an over-representation analysis (ORA) and identified enriched pathways using the Reactome database. Results showed an enrichment for neutrophil degranulation and chemokines binding for TFs module black, suggesting a potential role of this module in the interaction of neutrophils with other cells. The Brown module is mostly enriched in pathways related to EGFR signaling, suggesting a potential role of these TFs in regulation of cell growth. Module blue showed enrichment for toll-like receptor pathway components, suggesting an association with immune suppression factors and tumor progression. Module green showed an enrichment in transmembrane transporters, this might suggest the metabolic uptake and efflux of nutrients and the metabolic crosstalk between cells in the TME. Finally, module red is enriched in cell cycle checkpoint terms, confirming its role in regulation of cell proliferation (<xref ref-type="supplementary-material" rid="SF1">
<bold>Supplementary Figure&#xa0;1D</bold>
</xref>).</p>
<p>In order to reduce the dimensionality of these TF modules, we identified different module categories by using information of signaling pathways from PROGENy (see Methods). From this we performed a PCA analysis to see which TF modules cluster together based on their association with these pathways. TF modules blue, green and yellow clustered together and showed a common activation of p53 and apoptosis pathways while TF modules black, brown, turquoise and red grouped together by showing a similar association to VEGF, NFkB and TGFb (<xref ref-type="supplementary-material" rid="SF2">
<bold>Supplementary Figure&#xa0;2</bold>
</xref>).</p>
<p>Taken together and considering both enrichment and PCA analysis, our results showed that modules blue, green and yellow are more associated with cancer-related pathways, including tumor suppression and progression; while brown, black, turquoise and red have an association with cell growth.</p>
<p>Associations between these two categories of TFs modules and deconvolution features were investigated defining several cell type groups with correlated TF module scores (see Methods). As a reminder, cell type groups consist of subsets of grouped deconvolution features that share similar TF profiles, for example, Dendrogram_red_turquoise_black_brown.group_1, which contains several deconvolution features related to B cells, cancer cells and dendritic cells (<xref ref-type="supplementary-material" rid="SF16">
<bold>Supplementary Table&#xa0;7</bold>
</xref>). With this approach, 14 cell type groups containing deconvolution features with significant Pearson correlations with TF module scores (p-value &lt; 0.05, cut.height = 5) were identified. These cell type groups naturally divide patients into two groups (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5A</bold>
</xref>). A feature selection algorithm (see Methods) was applied to estimate the importance of different cell type groups in the classification of patients in the two patient clusters identified by unsupervised hierarchical clustering. After 100 repeats, 10 cell type group features were selected as important to classify Lung Predict early stage samples into the two patient clusters shown in (<xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5A, B</bold>
</xref>). These cell type group features can themselves be grouped into two main broader categories (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5C</bold>
</xref>).</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Selected cell type group features reveal two profiles in the Lung Predict early stage cohort. <bold>(A)</bold> Hierarchical clustering dendrogram of early stage patient samples using the 14 cell type group scores. <bold>(B)</bold> Feature selection based on importance for predicting the two patient clusters in 5A using a Boruta algorithm, showing confirmed features (green) and rejected features (red) after 100 repeats (see Methods). <bold>(C)</bold> Heatmap showing the 10 cell groups selected after feature selection. The panel also shows as column annotations of each sample the immuno-score (brown to white from high to low), the TF module scores (red to green from high to low, see composition of each module in <xref ref-type="supplementary-material" rid="SF14">
<bold>Supplementary Table&#xa0;5</bold>
</xref>) and the pathway scores (yellow to blue from high to low). <bold>(D)</bold> Contribution of cell type groups to the PCA variation in the classification of patient clusters.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1394965-g005.tif"/>
</fig>
<p>Performing a PCA using these cell type groups as features across the samples, we observed that two cell type groups (PCA variance explained &gt;20%) were mostly driving this separation of the two patient groups (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5D</bold>
</xref>). The first, namely &#x201c;Dendrogram_red_turquoise_black_brown.group_3&#x201d;.</p>
<p>is composed mainly by resting NK cells, cancer cells, fibroblasts, CAF, NKT cells, T helper cells, dendritic cells and M1/M0 macrophages (<xref ref-type="supplementary-material" rid="SF16">
<bold>Supplementary Table&#xa0;7</bold>
</xref>) and is significantly associated with pathways related to cell growth and angiogenesis based on the TF modules involved (red, turquoise, black and brown, c.f. <xref ref-type="supplementary-material" rid="SF1">
<bold>Supplementary Figures&#xa0;1B, C</bold>
</xref>).</p>
<p>The second cell type group, namely &#x201c;Dendrogram_yellow_blue_green.group_2&#x201d;, is highly present in patients with intermediate immune-scores and is composed mostly by CD4 T cells, dendritic cells, M2 macrophages, neutrophils, monocytes, mast cells, endothelial cells and NK cells, while being associated to pathways related to immune response activation and tumor suppression based on TF modules involved (yellow, blue, green, c.f. <xref ref-type="supplementary-material" rid="SF1">
<bold>Supplementary Figures&#xa0;1B, C</bold>
</xref>).</p>
</sec>
<sec id="s3_5">
<title>The two patient subgroups identified in the LungPredict early stage samples are validated in an external early stage LUAD cohort</title>
<p>Senosain et&#xa0;al. have recently published an in-depth characterisation of an early stage clinically annotated LUAD cohort (<xref ref-type="bibr" rid="B31">31</xref>). This cohort, to which we will refer as Vanderbilt, including 70 early-stage (stage I and II) lung adenocarcinomas, for which bulk RNAseq as well as 15 scRNAseq samples are available (with an overlap of 9 patients), was used as external validation.</p>
<p>Before using the validation cohort, we verified that these two datasets Lung Predict and Vanderbilt were comparable (<xref ref-type="supplementary-material" rid="SF9">
<bold>Supplementary Text 1</bold>
</xref>, <xref ref-type="supplementary-material" rid="SF12">
<bold>Supplementary Figures&#xa0;3</bold>
</xref>, <xref ref-type="supplementary-material" rid="SF13">
<bold>4</bold>
</xref>).</p>
<p>To validate our newly identified patient clusters, we considered the same 10 most important cell type groups identified via the feature selection algorithm using data from our validation cohort to see if the identified groups can also classify an independent cohort, namely the stage I and II samples from the Vanderbilt cohort. We performed a cell group projection analysis, which consists of identifying the same TFs modules based on the gene expression from the independent cohort. We then projected the same deconvolution subgroups into the unprocessed deconvolution features from the Vanderbilt samples and recreated the same cell type groups identified in the Lung Predict cohort. The independent validation cohort samples also display a separation into two patient groups based on the values of the selected cell type groups (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6A</bold>
</xref>). A PCA analysis suggests that the feature with the highest contribution (&gt;40%) is the same as in the Lung Predict analysis (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6B</bold>
</xref>). This important cell type group is composed mainly by resting NK cells and M1 cells and associated with cell growth and angiogenesis. This cell type group is present mostly in patients with intermediate and high immune-scores and lacking in patients with low immune-scores (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6C</bold>
</xref>).</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Selected cell type groups features found in LP cohort projected in early stage samples of validation cohort. <bold>(A)</bold> Dendrogram obtained by hierarchical clustering revealed two groups of patients based on the cell type groups values. <bold>(B)</bold> Cell type groups feature contribution to the PCA variation in the classification of patient clusters. <bold>(C)</bold> Early stage patient samples from the validation cohort are divided in two main clusters based on the values of the selected cell type groups from LP analysis.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1394965-g006.tif"/>
</fig>
<p>Taken together, these results suggest that the two patient groups we identified in the LP cohort are also identified in the validation cohort. Our findings suggest the importance of resting NK and M1 cells and activation of cell growth and angiogenesis in the separation of the two patient clusters observed similarly in the two cohorts.</p>
</sec>
<sec id="s3_6">
<title>Differential expression analysis between patients with alternative profiles of NK cells hints at differing cytotoxicity of these cells</title>
<p>In order to understand the difference between two clusters of patients defined by the selected cell type groups, we performed a differential expression analysis between Vanderbilt patients from cluster 1 (green) and Cluster 2 (red) in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6A</bold>
</xref>. We obtained 665 differential expressed genes (p.adj &lt; 0.05, abs(log2FoldChange) &gt; 1) between the two patient subgroups (<xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7A</bold>
</xref>). We then summarized these DEGs into KEGG pathways identifying enrichment of deregulated genes in several immunologically and oncologically relevant pathways (p value &lt; 0.01) (<xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7B</bold>
</xref>), including the NK cell-mediated cytotoxicity pathway. A network plot was generated linking enriched pathways and the genes contained in them in order to interrogate the genes present in this pathway and understand the overlap with other immunologically relevant pathways (<xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7C</bold>
</xref>). In this network plot, we see the downregulation of CD3 (epsilon and delta) as well as CD8 alpha suggesting a reduction in the activation - and, perhaps, no involvement - of CD8 T-cells in the functional profile of NK cells from cluster 1. Many KIR genes, important for NK cytotoxicity) appear downregulated in this cluster, confirming the potential presence of dysfunctional or resting NK cells in this first cluster of patients. In a deeper analysis of the NK-cell mediated cytotoxicity pathway (<xref ref-type="supplementary-material" rid="SF14">
<bold>Supplementary Figure&#xa0;5</bold>
</xref>), we observe a downregulation in inhibitory receptors KLRC1 (NKG2A) and KIR3DL2. The inhibitory potential of NKG2A is dependent on its dimerization with CD94, which is not differentially expressed in our analysis (<xref ref-type="bibr" rid="B45">45</xref>). Anfossi et&#xa0;al. reported that KIR+NKG2A+ NK cells were responsive upon stimulation with tumor targets whereas NK cells lacking these inhibitory markers are hyporesponsive (<xref ref-type="bibr" rid="B46">46</xref>). Furthermore, the observed downregulation in protein kinase C (PKC) can have a direct effect on the granulization and cytotoxic effect of these NK cells (<xref ref-type="bibr" rid="B47">47</xref>). Taken together, these results suggest that these two patient clusters might be defined by the presence of either functional or dysfunctional (resting) NK cells.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Supervised analysis of the patient groups identified in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6A</bold>
</xref>. <bold>(A)</bold> Volcano plot summarizing the 665 differentially expressed genes (DEGs) (padj &lt; 0.05, abs(log2FoldChange) &gt; 1) identified by comparing the green and red clusters from <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6A</bold>
</xref> <bold>(B)</bold> Top results from the KEGG pathway enrichment analysis (p value &lt; 0.01) on the DEGs summarized in the volcano plot. <bold>(C)</bold> Network plot of immunological pathways showing the genes involved in each pathway and overlapping among pathways. Node colors communicate the log2FoldChange of the genes between the two patient groups.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1394965-g007.tif"/>
</fig>
</sec>
<sec id="s3_7">
<title>Single-cell analysis in the validation cohort confirms multiple subgroups of NK cells</title>
<p>In an effort to better characterize the dual behavior associated with NK cells detected at the bulk RNAseq level, we analyzed single cell transcriptomics data from 15 patients from our validation cohort. Following standard procedures for scRNAseq analysis, we performed graph-based clustering of cells to identify cell groups sharing similar gene expression (<xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8A</bold>
</xref>) using annotations already provided in the scRNAseq object from the validation cohort (<xref ref-type="bibr" rid="B31">31</xref>) and didn&#x2019;t identify any batch effect (<xref ref-type="supplementary-material" rid="SF6">Supplementary Figure&#xa0;6</xref>). Since our focus was to identify different NK subclusters, we then re-annotated these cells. We performed annotation using reference expression datasets with curated cell type labels for automatic annotation in order to establish a consensus for the NK cell annotation (see Methods) (<xref ref-type="supplementary-material" rid="SF7">
<bold>Supplementary Figure&#xa0;7</bold>
</xref>). We extracted the cell clusters identified as NK (cluster 8) and performed an additional clustering step to identify subclusters within this population. We obtained 3 subclusters of NK cells (<xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8B</bold>
</xref>) that we investigated based on specific NK markers. All three subclusters showed a high expression of KLRK1, which is expressed on all NK cells as well as on a small subset of cytotoxic CD8 T-cells. Interestingly, when profiling the expression of GNLY (cytolytic compound expressed by cytotoxic cells) and KLRC2 (activation receptor, expressed on NK cells), cluster 0 did not show any detectable expression. Cluster 1 also lacks expression of KLRC2 while Cluster 2 shows expression of both markers, with higher expression of GNLY. Further analysis revealed that cluster 1 had the lowest expression of perforin (PRF1), granzyme B (GZMB) and interferon-<italic>&#x3b3;</italic> (IFNG), suggesting that this cluster may include resting or dysfunctional NK cells, with reduced cytotoxic potential. Clusters 0 and 2 display high expression of PRF1, GZMB and IFNG suggesting that they are functionally competent sub-types of NK cells. Cluster 0 is the only NK cluster expressing FCGR3A (Fc-gamma receptor III, also known as CD16), which suggests that it may contain cytotoxic, peripheral blood NK cells (<xref ref-type="bibr" rid="B48">48</xref>). Cluster 2 has high expression of ITGAE (CD103) and ZNF683 (HOBIT - regulates immune cell development (<xref ref-type="bibr" rid="B49">49</xref>) without any expression of S1PR5 (plays a role in migration of immune cells) and low expression of KLF2 (plays a role in the regulation of NK cell maturation), which suggest that this cluster may include cytotoxic, tissue-resident NK cells (<xref ref-type="bibr" rid="B50">50</xref>) (<xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8C</bold>
</xref>). For details about the differential expression markers between the NK clusters refer to <xref ref-type="supplementary-material" rid="SF17">
<bold>Supplementary Tables&#xa0;8,</bold>
</xref>&#x2013;<xref ref-type="supplementary-material" rid="SF19">
<bold>10</bold>
</xref>. We observe varying proportions of NK cell subtypes across our patient cohort, but unfortunately only 9 patients had both scRNAseq and bulk RNAseq, from which only 7 correspond to early stage samples (I, II) (<xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8D</bold>
</xref>), so we could not confidently estimate whether our grouping of bulk RNAseq samples into two patient groups according to NK subtype (indicated by numbers on each barplot) could be associated to the dominance of dysfunctional NK cells in the scRNAseq data.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Single-cell RNAseq characterization of natural killer (NK)-cell clusters in LUAD samples from the Vanderbilt cohort. <bold>(A)</bold> Graph-based UMAP clustering. <bold>(B)</bold> UMAP of cluster 8 identified as NK cells after automatic annotation showing the 3 NK subclusters. <bold>(C)</bold> Characterization of the three NK subclusters using several cell surface markers. <bold>(D)</bold> Proportions of each NK cluster, labeled according to the marker analysis. The numbers at the bottom correspond to the patient cluster to which the corresponding bulk RNAseq sample belongs (Cluster 1= green, Cluster 2 = red) according to (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6A)</bold>
</xref>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1394965-g008.tif"/>
</fig>
</sec>
<sec id="s3_8">
<title>Reference-based bulk RNA-seq deconvolution using the scRNAseq from the validation cohort to estimate cell type proportions in our primary cohort reveals the different annotated NK profiles in the LP early stage patients</title>
<p>To strengthen and validate our findings regarding cell type composition in the bulk data from our LungPredict cohort, we performed single-cell reference-based bulk RNAseq deconvolution using the scRNAseq data from Vanderbilt as our reference for extracting signatures. We used BayesPrism as implemented in the Omnideconv R package (see Methods) to deconvolve our early stage LungPredict samples. We identified the three different annotated NK subtypes across our samples and found the peripheral cytotoxic NK cell subtype to be the most predominant and the dysfunctional NK subtype to be the least abundant (<xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>).</p>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Reference-based deconvolution of primary cohort using BayesPrism method. <bold>(A)</bold> Deconvolution proportions from early stage samples from the LungPredict cohort. NK cells are subdivided into the three subgroups considered above: dysfunctional, peripheral and tissue resident. <bold>(B)</bold> NK subtypes proportions in early stage samples using the cell-type annotations from the scRNAseq object of the validation cohort.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1394965-g009.tif"/>
</fig>
</sec>
<sec id="s3_9">
<title>Cell type groups are associated with recurrence-free-survival in the validation cohort</title>
<p>Focusing on early stage disease, we can evaluate the potential association of the immune landscape and disease recurrence. The association of the immune profiles determined through the integration of shared inferred TF activity and the deconvolution features with recurrence was assessed using the mature follow-up available for patients from the validation cohort. CoxPH models were evaluated across all the 10 selected cell type groups and then used to stratify samples based on the linear predictors of the model. Kaplan Meier analysis and log rank tests were used to assess the difference between risk groups (see Methods). Two multivariate models were found as significant after log rank test (p value = 0.007 and p value = 0.0068) (<xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>). In model 1, the variables (covariates) that are most associated to recurrence free survival were Dendrogram_red_turquoise_black_brown.group_3, including resting NK cells, Dendrogram_red_turquoise_black_brown.group_9 and Dendrogram_red_turquoise_black_brown.group_combined_1, including more active/cytotoxic NK cells with other immune cells like neutrophils, T cells and activated dendritic cells.</p>
<fig id="f10" position="float">
<label>Figure&#xa0;10</label>
<caption>
<p>Multivariate cox proportional hazards (Cox PH) models were developed across all selected 10 cell type groups (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5B</bold>
</xref>). <bold>(A)</bold> Survival curves based on high and low risk groups using linear predictors after fitting Cox PH model using as covariates cell type groups corresponding to Dendrogram_red_turquoise_black_brown.group_3, Dendrogram_red_turquoise_black_brown.group_9 and Dendrogram_red_turquoise_black_brown.group_combined_1 (p value = 0.007). <bold>(B)</bold> Survival curves based on high and low risk groups using linear predictors after fitting Cox PH model using as covariates cell type groups corresponding to Dendrogram_red_turquoise_black_brown.group_3, Dendrogram_yellow_blue_green.group_2 and Dendrogram_yellow_blue_green.group_3 (p value = 0.0068).</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1394965-g010.tif"/>
</fig>
<p>Model 2 also contains as covariate the Dendrogram_red_turquoise_black_brown.group_3 feature, and additionally two other cell type groups: Dendrogram_yellow_blue_green.group_2, containing the NK resting subgroup as well as other resting immune cells (CD4, dendritic, Mast), and Dendrogram_yellow_blue_green.group_3, containing the more active NK subgroup in combination with T cells (CD4 and CD8) and dendritic cells in their active state (see <xref ref-type="supplementary-material" rid="SF16">
<bold>Supplementary Table&#xa0;7</bold>
</xref> for detailed composition of the cell groups). This result is limited by small sample size (n=70) and a low event rate (n=11), however the results serve as preliminary evidence for the applicability of transcriptomically defined immune patient profiles in real world outcomes among early stage lung adenocarcinoma patients.</p>
</sec>
<sec id="s3_10">
<title>TCGA LUAD cohort analysis confirms similar immune infiltration profiles across early stage patients</title>
<p>To further test the validity of our findings, we selected the 399 early stage (I,II) lung adenocarcinoma (LUAD) from TCGA. We performed immune cell type deconvolution and inferred TF activity across these samples as described above. We then projected and recreated the 10 selected cell type groups (see above) using the same TF modules found in the analysis mentioned above using early stage samples in the primary and in the validation cohorts. Our results showed three patient clusters related to distinct immune infiltration profiles. Two of the three patient clusters revealed similar expression patterns as the ones found in the LP and Vanderbilt cohorts (<xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11A</bold>
</xref>) and we identified patient clusters 1 (red) and 3 (green) as the clusters defined by two opposite NK profiles (<xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11B</bold>
</xref>). We then performed a differential expression analysis and a functional enrichment analysis using the KEGG database, identifying 1518 differentially expressed genes (padj &lt;.00001, abs(log<sub>2</sub>FoldChange) &gt; 1.5) revealing an enrichment in immunological and cytotoxic related pathways (p value &lt; 0.05) (<xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11C</bold>
</xref>).</p>
<fig id="f11" position="float">
<label>Figure&#xa0;11</label>
<caption>
<p>TCGA analysis using selected cell type groups from <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9A</bold>
</xref>. <bold>(A)</bold> Heatmap showing cell type groups scores after projection using the computed deconvolution and the inferred TF activity.  <bold>(B)</bold>Samples dendrogram using hierarchical clustering based on the cell type groups scores <bold>(C)</bold>. Dotplot showing KEGG pathways (p value &lt; 0.05) related to the enrichment of DEG (padj &lt; 0.05, abs(log<sub>2</sub>FoldChange) &gt; 1) after comparison between patient Cluster 1 and 3 (red and blue in panel <bold>B</bold>, respectively).</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1394965-g011.tif"/>
</fig>
</sec>
<sec id="s3_11">
<title>Survival analysis in TCGA revealed that both resting and activated NK subtypes are significant predictors of survival</title>
<p>Linear predictors from univariate cox proportional hazards (coxPH) models across all the 10 selected cell groups were evaluated to stratify patients based on their risk-scores, subsequently computing the survival curves through Kaplan Meier analysis and testing whether the survival between the two groups is significantly different (p value &lt; 0.01). In this dataset we applied stricter filtering due to the high number of patients (n=393), stratifying as high-risk only the top 34% of patients (based on their risk scores) and the remaining 66% as &#x201c;low-risk&#x201d;. Two models were found to be significantly associated with the survival time of patients (<xref ref-type="fig" rid="f12">
<bold>Figure&#xa0;12</bold>
</xref>). Cell type groups dendrogram_red_turquoise_black_brown.group_3 and dendrogram_red_turquoise_black_brown.group_4 with p value = 0.0063 and p value = 0.0027, respectively. The first cell type group corresponds to the subgroup of resting NK cells with macrophages M1 and the second one corresponds to the NK subgroup in combination with cancer, fibroblasts, dendritic, and Thelper cells (see <xref ref-type="supplementary-material" rid="SF16">
<bold>Supplementary Table&#xa0;7</bold>
</xref>). Both these features were predictors of survival in the univariate models. These results suggest an important association between these NK subtypes and patient survival.</p>
<fig id="f12" position="float">
<label>Figure&#xa0;12</label>
<caption>
<p>Survival curves corresponding to the analysis done for TCGA-LUAD (393 early stage patients). <bold>(A)</bold> Survival curves showed a significant difference (p value = 0.0063) of survival using formula 1 (Surv(time, status) ~ dendrogram_red_turquoise_black_brown.group_3) when comparing high-risk patients (yellow) and low-risk (blue) patients defined based on the risk scores. <bold>(B)</bold> Survival curves showed a significant difference (p value = 0.0027) of survival using formula 2 (Surv(time, status) ~ dendrogram_red_turquoise_black_brown.group_4) when comparing high-risk patients (yellow) and low-risk (blue) patients defined based on the risk scores.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1394965-g012.tif"/>
</fig>
</sec>
<sec id="s3_12">
<title>Patient subgroups identified are related to oncogene and tumor suppressor TF modules</title>
<p>To further investigate the functional mechanisms leading to the subgrouping of patients into 2 categories according to their TME landscapes, we further explored the association between TF modules and deconvolution features. In particular we highlight the modules that are associated with abundance of cancer cells as potentially capturing oncogenic processes while other modules negatively correlated with cancer cells could be considered as tumor suppressor processes (<xref ref-type="fig" rid="f13">
<bold>Figure&#xa0;13</bold>
</xref>). The module that is more strongly positively correlated with cancer cell estimates is red, which shows strong repression of Trail and p53 pathways and activation of MAPK, VEGF and Hypoxia and is strongly positively correlated to the presence of resting NK cells and negatively to the presence of active NK cells. The black and brown modules are negatively correlated with the same features and show instead strong activation of immune processes (NFkB and TFGb). The repression of module red clearly sets patients in cluster 1 apart (c.f. <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5C</bold>
</xref>). The TF activity profiles across early stage Lung Predict samples of TFs contained in each module are shown in <xref ref-type="supplementary-material" rid="SF17">
<bold>Supplementary Figure&#xa0;8.</bold>
</xref>
</p>
<fig id="f13" position="float">
<label>Figure&#xa0;13</label>
<caption>
<p>TF module characterisation based on association with grouped deconvolution features in early stage Lung Predict samples. The heatmap shows Pearson correlation between TF module scores and deconvolution features, highlighting cancer-related features. Colors represent levels of correlation (darker red implies high positive correlation, darker blue implies high negative correlation). Statistics are shown only for significantly correlated pairs (p value &lt; 0.05).</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1394965-g013.tif"/>
</fig>
<p>Since TF activities are estimated based on bulk RNAseq, we cannot be sure of whether these pathways are activated mainly in the cancer cells or the correlation directly reflects the tumor sample purity. However, combining these two types of features we have demonstrated that discordance between deconvolution signatures might simply reflect substantial differences in the subtypes of cells they refer to.</p>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<title>Discussion</title>
<p>This study leveraged integrative computational approaches to dissect immune heterogeneity in the tumor microenvironment of lung adenocarcinomas. Integrating bulk transcriptomics with bioinformatic analyses for cell type deconvolution and TF activity inference, we identified profiles associated with dual immune cell phenotypes (<xref ref-type="bibr" rid="B51">51</xref>).</p>
<p>Specifically, our combined analysis suggested the presence of two subgroups of natural killer (NK) cells. One subgroup is associated with a high proportion of cancer cells and CAFs and could be potentially associated with a &#x201c;resting&#x201d; or &#x201c;dysfunctional&#x201d; behavior. Dysfunctional NK cells are characterized by reduced proliferation and cytotoxic capabilities. In contrast, we inferred a high presence of B-cells, T-cells and NK cells in early stage samples with high immune-scores. This different group of NK cells may display cytotoxic capabilities and might even be subdivided into two NK profiles, depending on co-occurrence of other cell types, namely endothelial cells. Focusing on early stage (stage I and II) patient samples, we confirmed these dual NK subgroups in an independent LUAD cohort and in the 399 stage I and II LUAD samples from TCGA after further characterizing them in an scRNAseq dataset. Interestingly, in the scRNAseq data analysis, we identified three major NK clusters. We characterized these three clusters as resting/dysfunctional, circulating cytotoxic and tissue-resident cytotoxic NK cells. The single-cell analysis provided independent validation of the computationally defined NK cell subtypes/states, and provided further resolution into tissue-resident versus circulating NK cell subsets. Finally, we were able to show that our engineered features based on cell type groups, which take into account TF activity profiles to estimate presence of groups of different cell types, have predictive value on recurrence free survival (in our validation cohort) and on overall survival (in the TCGA cohort).</p>
<p>To summarize, we revealed a striking duality in NK cell phenotypes across three independent cohorts, with NK subsets displaying signatures of dysfunctional exhaustion versus cytotoxic competence. Dysfunctional NK cells have reduced proliferative and functional capacity, resulting from constant exposure to immune suppressive signals in the tumor microenvironment. Our findings align with other recent studies showing phenotypic heterogeneity in NK cells and other immune cell types in the context of cancer (<xref ref-type="bibr" rid="B52">52</xref>, <xref ref-type="bibr" rid="B53">53</xref>) and with reports that NK cell states might be essential for response to PD-1/PD-L1 blockers (<xref ref-type="bibr" rid="B54">54</xref>) and key players in immunotherapy (<xref ref-type="bibr" rid="B55">55</xref>, <xref ref-type="bibr" rid="B56">56</xref>). Beyond those results, our approach is a first step towards delineating the type of inter-cellular interactions that could be established in the TME in connection to the presence of these two NK cell subtypes.</p>
<p>Overall, our study sheds light on the significant diversity of immune cells in the lung cancer microenvironment. The integrated computational frameworks provide an accessible, robust and general methodology for immune profiling of tumor samples via bulk RNAseq.</p>
<p>Immune cell dysfunction arises from continuous stimulation in a persistent inflammatory environment. In the tumor microenvironment (TME), the presence of various immune suppressive signals exacerbates immune cell dysfunction leading to tumor progression and metastasis (<xref ref-type="bibr" rid="B57">57</xref>). The ability to resolve immune cell dysfunction versus activation states could significantly improve prognostic models and prediction of immunotherapy response (<xref ref-type="bibr" rid="B58">58</xref>). Whether these dysfunctional characteristics are a result of exhaustion or senescence will need to be determined (<xref ref-type="bibr" rid="B59">59</xref>). Our approach is a very step towards delineating the type of inter-cellular interactions that could be established in the TME in connection to the presence of these two NK cell subtypes.</p>
<p>Our exploration of the single cell data further strengthens the hypothesis that there are two major subgroups of NK cells, dysfunctional/resting and functional, associated with immune cells presence and that patients might be characterized based on the dominance of either of these two NK cell subgroups. It could be speculated that the profile of NK cell subtypes present could be related to response to immune checkpoint blockers. However, early stage LUAD patients are still rarely treated with this type of therapy, while only a few patients in Lung Predict received it, requiring alternative cohorts to validate this hypothesis. However, we note that in any non-pharmacologically treated tumor a strong immune response is likely to improve survival, potentially explaining why the active NK subtype, which associates with M1-like macrophages, could also improve survival in cases that are treated by surgery alone, as those included in our primary and validation cohorts.</p>
<p>We note that our initial analysis on the Lung Predict cohort across stages suggests that the duality in NK cells populations is not limited to early stage disease. Looking forward, extension of these analyses across lung cancer stages and histological subtypes could provide valuable insights into reprogramming of the immune microenvironment during progression. Incorporating spatial and proteomic data could help further resolve the tissue localization and functional capacities of distinct immune cell subsets in lung tumors. Ultimately, comprehensive mapping of immune heterogeneity in lung cancer provides a path towards more precise immunotherapeutic strategies (<xref ref-type="bibr" rid="B53">53</xref>, <xref ref-type="bibr" rid="B60">60</xref>).</p>
<p>Nevertheless, this study has several limitations to be considered. First, the sample size was relatively small, with only 62 lung adenocarcinomas in the primary analysis cohort and 70 in the validation cohort. The number of samples included in our analysis from TCGA is considerable (399) and helped us confirm our findings, but the cohort is likely to be less homogeneous. Larger studies on deeply clinically characterized samples will be needed to further validate the findings. Second, we utilized only transcriptomic data, which provides an incomplete picture of cellular states compared to integrating proteomics and adding spatial resolution. Third, our study lacked longitudinal samples, with which we could assess how immune profiles change over time and with therapy. Fourth, bulk transcriptomics may underestimate certain rare cell populations that are better captured by single-cell sequencing. Our in-depth analysis of 15 samples for which scRNAseq was available and using NK populations identified therein helped us confirm the presence of the NK subtypes in our bulk RNAseq datasets. Fifth, the specific deconvolution algorithms used can impact results, and incorporating additional methods could provide further validation. Finally, functional validations to directly test immune cell cytotoxicity or dysfunctional profiles in NK cells were not performed. This would require either <italic>in-vitro</italic> experiments or very deep characterisation of clinical samples that are beyond the scope of this study.</p>
<p>Overall, this proof-of-concept study demonstrated the potential of integrated computational immunology techniques to identify signatures of immune cell dysfunction from bulk tumor profiling. However, further experimental and clinical validations are needed to fully characterize the phenotypic diversity of anti-tumor immune responses in lung adenocarcinoma patients.</p>
</sec>
<sec id="s5" sec-type="conclusions">
<title>Conclusion</title>
<p>In summary, our multi-omics computational framework elucidated heterogeneous immune microenvironments in lung adenocarcinoma. Deconvolution and TF activity analysis identified groups of immune cells with coordinated regulation/states. The ability to resolve dysfunctional/resting versus activated immune cell states from bulk tumor profiling could have important implications for prognosis and prediction of response to immunotherapy, as suggested by our preliminary evidence of an association to survival in 3 early LUAD cohorts. Further characterization of dynamic immune reprogramming during cancer progression and therapy response represents an important future direction. We make the RNAseq datasets from our Lung Predict cohort and all the code available to the research community, hoping to contribute to reproducibility and open-research practices for the ultimate benefit of patients.</p>
</sec>
</body>
<back>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>The primary LUAD cohort (Lung Predict) transcriptomics data is available on NCBI GEO with study number GSE251840. The validation LUAD cohort (Vanderbilt) data is available on Zenodo under accession number 7878082. The code to reproduce the analysis and figures is available on github at <uri xlink:href="https://github.com/VeraPancaldiLab/LungPredict1_paper">https://github.com/VeraPancaldiLab/LungPredict1_paper</uri>.</p>
</sec>
<sec id="s7" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>The studies involving human participants were reviewed and approved by the Ministry of Research under the number DC-2008-463. The patients/participants signed a non-opposition form to participate in this study under the LUNG PREDICT protocol. 2018. For the validation dataset, tumor tissue samples were collected from patients undergoing lung resection surgery following an Institutional Review Board&#x2013;approved protocol 000616 at the Vanderbilt University Medical Center (Nashville, TN). Written informed consent was obtained from all subjects.</p>
</sec>
<sec id="s8" sec-type="author-contributions">
<title>Author contributions</title>
<p>MH: Conceptualization, Investigation, Methodology, Software, Visualization, Writing &#x2013; original draft. LK: Formal analysis, Investigation, Methodology, Software, Visualization, Writing &#x2013; original draft. AE: Methodology, Software, Writing &#x2013; review &amp; editing. MK: Investigation, Software, Supervision, Validation, Writing &#x2013; review &amp; editing. TX: Investigation, Methodology, Software, Writing &#x2013; review &amp; editing. AC: Software, Writing &#x2013; review &amp; editing. AP: Investigation, Methodology, Writing &#x2013; review &amp; editing. AnC: Investigation, Writing &#x2013; review &amp; editing. AK: Funding acquisition, Project administration, Writing &#x2013; review &amp; editing. SG: Project administration, Writing &#x2013; review &amp; editing. EC: Investigation, Writing &#x2013; review &amp; editing. LB: Data curation, Writing &#x2013; review &amp; editing. MFS: Data curation, Formal analysis, Investigation, Methodology, Software, Writing &#x2013; review &amp; editing. YZ: Formal analysis, Investigation, Writing &#x2013; review &amp; editing. SZ: Software, Writing &#x2013; review &amp; editing. PB: Data curation, Writing &#x2013; review &amp; editing. AM: Funding acquisition, Writing &#x2013; review &amp; editing. JB: Data curation, Writing &#x2013; review &amp; editing. PL: Methodology, Writing &#x2013; review &amp; editing. AlP: Project administration, Funding acquisition, Writing &#x2013; review &amp; editing. ErC: Funding acquisition, Writing &#x2013; review &amp; editing. GF: Funding acquisition, Writing &#x2013; review &amp; editing. FM: Resources, Supervision, Writing &#x2013; review &amp; editing. FC: Supervision, Writing &#x2013; review &amp; editing. OD: Supervision, Writing &#x2013; review &amp; editing. JM: Resources, Supervision, Writing &#x2013; review &amp; editing. VP: Conceptualization, Funding acquisition, Investigation, Methodology, Project administration, Supervision, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s9" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This work was supported by the Lung Predict pilot project as part of an alliance between the Pierre Fabre Research Institute and the IUCT. Work in the Pancaldi lab was funded by the Chair of Bioinformatics in Oncology of the CRCT (INSERM; Fondation Toulouse Cancer Sant&#xe9; and Pierre Fabre Research Institute) and Ligue Nationale Contre le Cancer. while the work on the Vanderbilt cohort was funded by the National Institutes of Health of the USA (U01CA196405 &amp; U01CA152662). This study has been partially supported through the grant EUR CARe N&#xb0;ANR-18-EURE-0003 in the framework of the Programme des Investissements d&#x2019;Avenir and an Eiffel Excellence doctoral fellowship to M. H.</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>We would like to express our sincerest gratitude for all the patients who took part in this and other studies. Without their consent and contributions, there would be no progress and advancements in this field of research.</p>
</ack>
<sec id="s10" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
<p>The author(s) declared that they were an editorial board member of Frontiers, at the time of submission. This had no impact on the peer review process and the final decision.</p>
</sec>
<sec id="s11" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fimmu.2024.1394965/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fimmu.2024.1394965/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Presentation1.pdf" id="SF1" mimetype="application/pdf">
<label>Supplementary Figure&#xa0;1</label>
<caption>
<p>TFs modules characterization from early stage Lung predict samples. <bold>(A)</bold> Number of TFs across each of the 7 modules. <bold>(B)</bold> Module association between TFs modules scores and pathway values (only showing significant correlations considering p value &lt; 0.05). <bold>(C)</bold> Heatmap of the TF activity of the 20 hub TFs across samples, showing their related module as the color annotation on the right. <bold>(D)</bold> Reactome enrichment results from unique target genes from hub TFs of each module.</p>
</caption>
</supplementary-material>
<supplementary-material xlink:href="Presentation1.pdf" id="SF2" mimetype="application/pdf">
<label>Supplementary Figure&#xa0;2</label>
<caption>
<p>TFs modules classification and characterization from analysis on early stage samples from Lung Predict cohort. <bold>(A)</bold> Construction of weighted TFs modules based on inferred co-activity. <bold>(B)</bold> Hierarchical clustering based on association values between TFs modules and pathway activities. <bold>(C)</bold> Biplot representing the contribution of the top 6 pathways classifying the TFs modules. <bold>(D)</bold> Contribution of each pathway on the TFs module classification.</p>
</caption>
</supplementary-material>
<supplementary-material xlink:href="Presentation1.pdf" id="SF3" mimetype="application/pdf">
<label>Supplementary Figure&#xa0;3</label>
<caption>
<p>Analysis of combined LungPredict and Vanderbilt validation cohort A. <bold>(A)</bold> difference between the LP and Vanderbilt cohorts on normalized counts was evident and treated as a batch effect <bold>(B)</bold> Heatmap showing the Pearson correlation between the principal components and the metadata variables (the darker the green the higher the correlation). p values 0, 0.0001, 0.001, 0.01, 0.05, 1 correspond to &#x2018;****&#x2019;, &#x2018;***&#x2019;, &#x2018;**&#x2019;, &#x2018;*&#x2019;, &#x2018;&#x2018; respectively. <bold>(C)</bold> PCA plot using TFs activity values after calculating it independently in each cohort, shows the difference between cohorts was removed. <bold>(D)</bold> Heatmap showing no significant correlation between cohorts (treated here as batches) and the principal components (PCs) using TFs activity.</p>
</caption>
</supplementary-material>
<supplementary-material xlink:href="Presentation1.pdf" id="SF4" mimetype="application/pdf">
<label>Supplementary Figure&#xa0;4</label>
<caption>
<p>PCA analysis to assess batch effect within the validation cohort. <bold>(A)</bold> PCA of validation cohort (Vanderbilt) normalized counts before batch correction <bold>(B)</bold>. PCA of validation cohort normalized counts after batch effect removal by Combat_seq from the sva R package (<xref ref-type="bibr" rid="B61">61</xref> (v3.50.0)) to maintain the integrity of the raw counts.</p>
</caption>
</supplementary-material>
<supplementary-material xlink:href="Presentation1.pdf" id="SF5" mimetype="application/pdf">
<label>Supplementary Figure&#xa0;5</label>
<caption>    <p>KEGG pathway diagram of differentially expressed genes between two patient clusters identified in the Vanderbilt cohort early stage samples (c.f. <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9A</bold>
</xref>). The diagram shows the &#x201c;Natural Killer Cell mediated cytotoxicity pathway&#x201d; produced using the pathview R package (<xref ref-type="bibr" rid="B62">62</xref> (v1.42.0)) components and interactions, highlighting downregulation of inhibitory (KIR3DL1/2) receptors as well as protein kinase C (PKC).</p>
</caption>
</supplementary-material>
<supplementary-material xlink:href="Presentation1.pdf" id="SF6" mimetype="application/pdf">
<label>Supplementary Figure&#xa0;6</label>
<caption>
<p>UMAP of scRNAseq data from 15 Vanderbilt cohort patients (<xref ref-type="bibr" rid="B31">31</xref>). UMAP shows no batch effect influence in the cell based clustering.</p>
</caption>
</supplementary-material>
<supplementary-material xlink:href="Presentation1.pdf" id="SF7" mimetype="application/pdf">
<label>Supplementary Figure&#xa0;7</label>
<caption>
<p>Automatic cluster annotation from Vanderbilt scRNA cohort using reference expression datasets with curated cell type labels. <bold>(A)</bold> Cluster automation using Human Primary Cell Atlas. <bold>(B)</bold> Cluster annotation using Database Immune Cell Expression Data. <bold>(C)</bold> Cluster annotation using Monaco database. <bold>(D)</bold> Cluster annotation using Blueprint Encode Data.</p>
</caption>
</supplementary-material>
<supplementary-material xlink:href="Presentation1.pdf" id="SF8" mimetype="application/pdf">
<label>Supplementary Figure&#xa0;8</label>
<caption>
<p>TFs activity of module composition from TF modules. Modules black, red, blue, brown, green, turquoise and yellow correspond to Figures <bold>(A&#x2013;G)</bold> respectively.</p>
</caption>
</supplementary-material>
<supplementary-material xlink:href="Presentation1.pdf" id="SF9" mimetype="application/pdf">
<label>Supplementary Text 1</label>
<caption>
<p>related to <xref ref-type="supplementary-material" rid="SF12">
<bold>Supplementary Figure&#xa0;3</bold>
</xref>, <xref ref-type="supplementary-material" rid="SF13">
<bold>Supplementary Figure&#xa0;4</bold>
</xref> Evaluation of batch effects within and between cohorts: To assess comparability between the Lung Predict and Vanderbilt early stage cohorts, we performed a PCA analysis using the R package PCAtools (<xref ref-type="bibr" rid="B63">63</xref> (v2.14.0)) where we joined the two datasets and tested whether they separated or not. As expected, there is a big difference between the two cohorts based on normalized counts (<xref ref-type="supplementary-material" rid="SF3">
<bold>Supplementary Figure&#xa0;3A</bold>
</xref>) with a pearson correlation of 1 (p value &lt; 0.0001) between cohort (here batch) and the first principal component (<xref ref-type="supplementary-material" rid="SF3">
<bold>Supplementary Figure&#xa0;3B</bold>
</xref>). Instead of removing the batch effect, which potentially can also eliminate some important biological differences, and since our analysis does not directly use normalized counts, we decided to calculate TFs activity independently for each cohort and assess again for batch effects. As expected, calculation of the inferred TFs activity removed the batch effects between the two cohorts (<xref ref-type="supplementary-material" rid="SF3">
<bold>Supplementary Figure&#xa0;3C</bold>
</xref>) showing no correlation (r = 0.01) between the cohorts and the PC1 (<xref ref-type="supplementary-material" rid="SF3">
<bold>Supplementary Figure&#xa0;3D</bold>
</xref>). Once we confirmed that the two datasets can be comparable when looking at the TF activity profiles, we performed the previously described analysis only on the validation cohort to assess for within-dataset batch effects. A PCA analysis identified two main groups confounded by batches (<xref ref-type="supplementary-material" rid="SF4">
<bold>Supplementary Figure&#xa0;4A</bold>
</xref>). For this reason, we performed both our TFs inference analysis and immune cell type deconvolution calculation independently for each batch. We then concatenated our results and saw that even though the TFs analysis was not affected by the batch effect, this was still present in the deconvolution results. We then used Combat_seq from the sva R package (<xref ref-type="bibr" rid="B61">61</xref> (v3.50.0)) to remove batch effects from our counts and maintain the integrity of the raw counts (<xref ref-type="supplementary-material" rid="SF4">
<bold>Supplementary Figure&#xa0;4B</bold>
</xref>). Finally, after log<sub>2</sub>(TPM + 1) normalization we calculated deconvolution features from batch corrected datasets.</p>
</caption>
</supplementary-material>
<supplementary-material xlink:href="Presentation1.pdf" id="SF10" mimetype="application/pdf">
<label>Supplementary Table&#xa0;1</label>
<caption>
<p>Deconvolution methods and signatures.</p>
</caption>
</supplementary-material>
<supplementary-material xlink:href="Presentation1.pdf" id="SF11" mimetype="application/pdf">
<label>Supplementary Table&#xa0;2</label>
<caption>
<p>Immune-scores hallmarks.</p>
</caption>
</supplementary-material>
<supplementary-material xlink:href="Presentation1.pdf" id="SF12" mimetype="application/pdf">
<label>Supplementary Table&#xa0;3</label>
<caption>
<p>Composition of deconvolution features subgroups on all samples from Lung Predict cohort.</p>
</caption>
</supplementary-material>
<supplementary-material xlink:href="Presentation1.pdf" id="SF13" mimetype="application/pdf">
<label>Supplementary Table&#xa0;4</label>
<caption>
<p>Composition of TF modules obtained from all samples from Lung Predict cohort.</p>
</caption>
</supplementary-material>
<supplementary-material xlink:href="Presentation1.pdf" id="SF14" mimetype="application/pdf">
<label>Supplementary Table&#xa0;5</label>
<caption>
<p>Composition of deconvolution features subgroups on early stage samples from Lung Predict cohort.</p>
</caption>
</supplementary-material>
<supplementary-material xlink:href="Presentation1.pdf" id="SF15" mimetype="application/pdf">
<label>Supplementary Table&#xa0;6</label>
<caption>
<p>Composition of TF modules obtained from early stage samples from Lung Predict cohort.</p>
</caption>
</supplementary-material>
<supplementary-material xlink:href="Presentation1.pdf" id="SF16" mimetype="application/pdf">
<label>Supplementary Table&#xa0;7</label>
<caption>
<p>Composition of cell groups obtained from early stage samples from Lung Predict cohort.</p>
</caption>
</supplementary-material>
<supplementary-material xlink:href="Presentation1.pdf" id="SF17" mimetype="application/pdf">
<label>Supplementary Table&#xa0;8</label>
<caption>
<p>Differential expression markers between NK peripheral (pct.1) and NK dysfunctional (pct.2) (p_val_adj &lt;0.05 and abs(avg_log2FC) &gt; 1).</p>
</caption>
</supplementary-material>
<supplementary-material xlink:href="Presentation1.pdf" id="SF18" mimetype="application/pdf">
<label>Supplementary Table&#xa0;9</label>
<caption>
<p>Differential expression markers between NK peripheral (pct.1) and NK Tissue resident (pct.2) (p_val_adj &lt;0.05 and abs(avg_log2FC) &gt; 1).</p>
</caption>
</supplementary-material>
<supplementary-material xlink:href="Presentation1.pdf" id="SF19" mimetype="application/pdf">
<label>Supplementary Table&#xa0;10</label>
<caption>
<p>Differential expression markers between NK Dysfunctional (pct.1) and NK Tissue resident (pct.2) (p_val_adj &lt;0.05 and abs(avg_log2FC) &gt; 1).</p>
</caption>
</supplementary-material>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mazieres</surname> <given-names>J</given-names>
</name>
<name>
<surname>Drilon</surname> <given-names>A</given-names>
</name>
<name>
<surname>Lusque</surname> <given-names>A</given-names>
</name>
<name>
<surname>Mhanna</surname> <given-names>L</given-names>
</name>
<name>
<surname>Cortot</surname> <given-names>A</given-names>
</name>
<name>
<surname>Mezquita</surname> <given-names>L</given-names>
</name>
<etal/>
</person-group>. <article-title>Immune check- point inhibitors for patients with advanced lung cancer and oncogenic driver alterations: results from the IMMUNOTARGET registry</article-title>. <source>Ann Oncol</source>. (<year>2019</year>) <volume>30</volume>:<page-range>1321&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/annonc/mdz167</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>C</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>G</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>F</given-names>
</name>
<etal/>
</person-group>. <article-title>Clinical significance and inflammatory landscapes of a novel recurrence-associated immune signature in early-stage lung adenocarcinoma</article-title>. <source>Cancer Lett</source>. (<year>2020</year>) <volume>479</volume>:<fpage>31</fpage>&#x2013;<lpage>41</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.canlet.2020.03.016</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sturm</surname> <given-names>G</given-names>
</name>
<name>
<surname>Finotello</surname> <given-names>F</given-names>
</name>
<name>
<surname>Petitprez</surname> <given-names>F</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>JD</given-names>
</name>
<name>
<surname>Baumbach</surname> <given-names>J</given-names>
</name>
<name>
<surname>Fridman</surname> <given-names>WH</given-names>
</name>
<etal/>
</person-group>. <article-title>Comprehensive evaluation of transcriptome-based cell-type quantification methods for immuno-oncology</article-title>. <source>Bioinformatics</source>. (<year>2019</year>) <volume>35</volume>:<page-range>i436&#x2013;45</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btz363</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Avila Cobos</surname> <given-names>F</given-names>
</name>
<name>
<surname>Alquicira-Hernandez</surname> <given-names>J</given-names>
</name>
<name>
<surname>Powell</surname> <given-names>JE</given-names>
</name>
<name>
<surname>Powell</surname> <given-names>JE</given-names>
</name>
<name>
<surname>Mestdagh</surname> <given-names>P</given-names>
</name>
<name>
<surname>Preter De</surname> <given-names>K</given-names>
</name>
</person-group>. <article-title>Benchmarking of cell type deconvolution pipelines for transcriptomics data</article-title>. <source>Nat Commun</source>. (<year>2020</year>) <volume>11</volume>:<fpage>5650</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41467-020-19015-1</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Merotto</surname> <given-names>L</given-names>
</name>
<name>
<surname>Zopoglou</surname> <given-names>M</given-names>
</name>
<name>
<surname>Zackl</surname> <given-names>C</given-names>
</name>
<name>
<surname>Finotello</surname> <given-names>F</given-names>
</name>
</person-group>. <article-title>Next-generation deconvolution of transcriptomic data to investigate the tumor microenvironment</article-title>. In: <source>International review of cell and molecular biology</source>. <publisher-name>Academic Press</publisher-name> (<year>2023</year>).</citation>
</ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stuart</surname> <given-names>T</given-names>
</name>
<name>
<surname>Butler</surname> <given-names>A</given-names>
</name>
<name>
<surname>Hoffman</surname> <given-names>P</given-names>
</name>
<name>
<surname>Hafemeister</surname> <given-names>C</given-names>
</name>
<name>
<surname>Papalexi</surname> <given-names>E</given-names>
</name>
<name>
<surname>Mauck</surname> <given-names>WM</given-names> <suffix>3rd</suffix>
</name>
<etal/>
</person-group>. <article-title>Comprehensive integration of single-cell data</article-title>. <source>Cell</source>. (<year>2019</year>) <volume>177</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cell.2019.05.031</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ruan</surname> <given-names>X</given-names>
</name>
<name>
<surname>Ye</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>W</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>L</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>M</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Y</given-names>
</name>
<etal/>
</person-group>. <article-title>Multi-omics integrative analysis of lung adenocarcinoma: An in silico profiling for precise medicine</article-title>. <source>Front Med</source>. (<year>2022</year>) <volume>9</volume>:<elocation-id>894338.</elocation-id> doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fmed.2022.894338</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8</label>
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Andrews</surname> <given-names>S</given-names>
</name>
</person-group>. <source>FastQC: a quality control tool for high throughput sequence data</source> (<year>2010</year>). Available online at: <uri xlink:href="http://www.bioinformatics.babraham.ac.uk/projects/fastqc/">http://www.bioinformatics.babraham.ac.uk/projects/fastqc/</uri>. (accessed <access-date>March 10, 2022</access-date>)</citation>
</ref>
<ref id="B9">
<label>9</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wingett</surname> <given-names>S</given-names>
</name>
<name>
<surname>Andrews</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>FastQ Screen: A tool for multi-genome mapping and quality control</article-title>. <source>F1000Res</source>. (<year>2018</year>) <volume>7</volume>:<fpage>1338</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.12688/f1000research</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dobin</surname> <given-names>A</given-names>
</name>
<name>
<surname>Davis</surname> <given-names>C</given-names>
</name>
<name>
<surname>Schlesinger</surname> <given-names>F</given-names>
</name>
<name>
<surname>Drenkow</surname> <given-names>J</given-names>
</name>
<name>
<surname>Zaleski</surname> <given-names>C</given-names>
</name>
<name>
<surname>Jha</surname> <given-names>S</given-names>
</name>
<etal/>
</person-group>. <article-title>STAR: ultrafast universal RNA-seq aligner</article-title>. <source>Bioinformatics</source>. (<year>2013</year>) <volume>29</volume>:<fpage>15</fpage>&#x2013;<lpage>21</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/bts635</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>B</given-names>
</name>
<name>
<surname>Dewey</surname> <given-names>C</given-names>
</name>
</person-group>. <article-title>RSEM: accurate transcript quantification from RNA-Seq data with or without a reference genome</article-title>. <source>BMC Bioinf</source>. (<year>2011</year>) <volume>12</volume>:<fpage>323</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/1471-2105-12-323</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liao</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Smyth</surname> <given-names>GK</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>W</given-names>
</name>
</person-group>. <article-title>featureCounts: an efficient general-purpose program for assigning sequence reads to genomic features</article-title>. <source>Bioinformatics</source>. (<year>2014</year>) <volume>30</volume>:<page-range>923&#x2013;30</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btt656</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Love</surname> <given-names>M</given-names>
</name>
<name>
<surname>Huber</surname> <given-names>W</given-names>
</name>
<name>
<surname>Anders</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Moderated estimation of fold change and dispersion for RNA-seq data with DESeq2</article-title>. <source>Genome Biol</source>. (<year>2014</year>) <volume>15</volume>:<fpage>550</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13059-014-0550-8</pub-id>
</citation>
</ref>
<ref id="B14">
<label>14</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Morandat</surname> <given-names>F</given-names>
</name>
<name>
<surname>Hill</surname> <given-names>B</given-names>
</name>
<name>
<surname>Osvald</surname> <given-names>L</given-names>
</name>
<name>
<surname>Vitek</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Evaluating the design of the R language</article-title>. In: <person-group person-group-type="editor">
<name>
<surname>Noble</surname> <given-names>J</given-names>
</name>
</person-group>, editor. <source>ECOOP 2012 &#x2013; object-oriented programming</source>, vol. <volume>7313</volume> (<year>2012</year>) (<publisher-loc>Berlin, Heidelberg</publisher-loc>: <publisher-name>Springer</publisher-name>).</citation>
</ref>
<ref id="B15">
<label>15</label>
<citation citation-type="book">
<person-group person-group-type="author">
<collab>R Core Team</collab>
</person-group>. <source>R: A language and environment for statistical computing</source>. <publisher-loc>Vienna, Austria</publisher-loc>: <publisher-name>R Foundation for Statistical Computing</publisher-name> (<year>2020</year>). Available at: <uri xlink:href="https://www.R-project.org/">https://www.R-project.org/</uri>.</citation>
</ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gentleman</surname> <given-names>R</given-names>
</name>
<name>
<surname>Carey</surname> <given-names>V</given-names>
</name>
<name>
<surname>Bates</surname> <given-names>D</given-names>
</name>
<name>
<surname>Bolstad</surname> <given-names>B</given-names>
</name>
<name>
<surname>Dettling</surname> <given-names>M</given-names>
</name>
<name>
<surname>Dudoit</surname> <given-names>S</given-names>
</name>
<etal/>
</person-group>. <article-title>Bioconductor: open software development for computational biology and bioinformatics</article-title>. <source>Genome Biol</source>. (<year>2004</year>) <volume>5</volume>:<fpage>R80</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/gb-2004-5-10-r80</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huber</surname> <given-names>W</given-names>
</name>
<name>
<surname>Carey</surname> <given-names>V</given-names>
</name>
<name>
<surname>Gentleman</surname> <given-names>R</given-names>
</name>
<name>
<surname>Anders</surname> <given-names>S</given-names>
</name>
<name>
<surname>Carlson</surname> <given-names>M</given-names>
</name>
<name>
<surname>Carvalho</surname> <given-names>BS</given-names>
</name>
<etal/>
</person-group>. <article-title>Orchestrating high-throughput genomic analysis with bioconductor</article-title>. <source>Nat Methods</source>. (<year>2015</year>) <volume>12</volume>:<page-range>115&#x2013;21</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nmeth.3252</pub-id>
</citation>
</ref>
<ref id="B18">
<label>18</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname> <given-names>G</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>L</given-names>
</name>
<name>
<surname>Han</surname> <given-names>Y</given-names>
</name>
<name>
<surname>He</surname> <given-names>Q-Y</given-names>
</name>
</person-group>. <article-title>clusterProfiler: an R package for comparing biological themes among gene clusters</article-title>. <source>A J Integr Biol</source>. (<year>2012</year>) <volume>16</volume>:<page-range>284&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1089/omi.2011.0118</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gu</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Eils</surname> <given-names>R</given-names>
</name>
<name>
<surname>Schlesner</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Complex heatmaps reveal patterns and correlations in multi- dimensional genomic data</article-title>. <source>Bioinformatics</source>. (<year>2016</year>) <volume>32</volume>:<page-range>2847&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btw313</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Blighe</surname> <given-names>K</given-names>
</name>
<name>
<surname>Rana</surname> <given-names>S</given-names>
</name>
<name>
<surname>Lewis</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>EnhancedVolcano: Publication- ready volcano plots with enhanced colouring and labeling</article-title>. (<year>2018</year>). doi:&#xa0;<pub-id pub-id-type="doi">10.18129/B9.bioc.EnhancedVolcano</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Leote</surname> <given-names>AC</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>X</given-names>
</name>
<name>
<surname>Beyer</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Regulatory network-based imputation of dropouts in single-cell RNA sequencing data</article-title>. <source>PLOS Comput Biology</source>. (<year>2024</year>) <volume>18</volume>(<issue>2</issue>):<elocation-id>1009849</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pcbi.1009849</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schubert</surname> <given-names>M</given-names>
</name>
<name>
<surname>Klinger</surname> <given-names>B</given-names>
</name>
<name>
<surname>Kl&#xfc;nemann</surname> <given-names>M</given-names>
</name>
<name>
<surname>Sieber</surname> <given-names>A</given-names>
</name>
<name>
<surname>Uhlitz</surname> <given-names>F</given-names>
</name>
<name>
<surname>Sauer</surname> <given-names>S</given-names>
</name>
<etal/>
</person-group>. <article-title>Perturbation-response genes reveal signaling footprints in cancer gene expression</article-title>. <source>Nat Commun</source>. (<year>2018</year>) <volume>9</volume>:<fpage>20</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41467-017-02391-6</pub-id>
</citation>
</ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Badia-i-Mompel</surname> <given-names>P</given-names>
</name>
<name>
<surname>V&#xe9;lez Santiago</surname> <given-names>J</given-names>
</name>
<name>
<surname>Braunger</surname> <given-names>J</given-names>
</name>
<name>
<surname>Geiss</surname> <given-names>C</given-names>
</name>
<name>
<surname>Dimitrov</surname> <given-names>D</given-names>
</name>
<name>
<surname>M&#xfc;ller-Dott</surname> <given-names>S</given-names>
</name>
<etal/>
</person-group>. <article-title>decoupleR: ensemble of computational methods to infer biological activities from omics data</article-title>. <source>Bioinf Adv</source>. (<year>2022</year>) <volume>2</volume>(<issue>1</issue>):<elocation-id>vbac016</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioadv/vbac016</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>M&#xfc;ller-Dott</surname> <given-names>S</given-names>
</name>
<name>
<surname>Tsirvouli</surname> <given-names>E</given-names>
</name>
<name>
<surname>Vazquez</surname> <given-names>M</given-names>
</name>
<name>
<surname>Ramirez Flores</surname> <given-names>RO</given-names>
</name>
<name>
<surname>Badia-I-Mompel</surname> <given-names>P</given-names>
</name>
<name>
<surname>Fallegger</surname> <given-names>R</given-names>
</name>
<etal/>
</person-group>. <article-title>Expanding the coverage of regulons from high-confidence prior knowledge for accurate estimation of transcription factor activities</article-title>. <source>Nucleic Acids Res</source>. (<year>2023</year>) <volume>51</volume>(<issue>20</issue>):<page-range>10934&#x2013;49</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkad841</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alvarez</surname> <given-names>M</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Giorgi</surname> <given-names>F</given-names>
</name>
<name>
<surname>Lachmann</surname> <given-names>A</given-names>
</name>
<name>
<surname>Ding</surname> <given-names>B</given-names>
</name>
<name>
<surname>Ye</surname> <given-names>B</given-names>
</name>
<etal/>
</person-group>. <article-title>Functional characterization of somatic mutations in cancer using network-based inference of protein activity</article-title>. <source>Nat Genet</source>. (<year>2016</year>) <volume>48</volume>:<page-range>838&#x2013;47</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/ng.3593</pub-id>
</citation>
</ref>
<ref id="B26">
<label>26</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lapuente-Santana</surname> <given-names>O</given-names>
</name>
<name>
<surname>van Genderen</surname> <given-names>M</given-names>
</name>
<name>
<surname>Hilbers</surname> <given-names>P</given-names>
</name>
<name>
<surname>Hilbers</surname> <given-names>PAJ</given-names>
</name>
<name>
<surname>Finotello</surname> <given-names>F</given-names>
</name>
<name>
<surname>Eduati</surname> <given-names>F</given-names>
</name>
</person-group>. <article-title>Interpretable systems biomarkers predict response to immune-checkpoint inhibitors</article-title>. <source>Cell</source>. (<year>2021</year>) <volume>2</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1101/2021.02.05.429977</pub-id>
</citation>
</ref>
<ref id="B27">
<label>27</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kursa</surname> <given-names>MB</given-names>
</name>
<name>
<surname>Rudnicki</surname> <given-names>WR</given-names>
</name>
</person-group>. <article-title>Feature selection with the boruta package</article-title>. <source>J Stat Software</source>. (<year>2010</year>) <volume>36</volume>:<fpage>1</fpage>&#x2013;<lpage>13</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.18637/jss.v036.i11</pub-id>
</citation>
</ref>
<ref id="B28">
<label>28</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Langfelder</surname> <given-names>P</given-names>
</name>
<name>
<surname>Horvath</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>WGCNA: an R package for weighted correlation network analysis</article-title>. <source>BMC Bioinf</source>. (<year>2008</year>) <volume>9</volume>:<elocation-id>559</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/1471-2105-9-559</pub-id>
</citation>
</ref>
<ref id="B29">
<label>29</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname> <given-names>G</given-names>
</name>
<name>
<surname>He</surname> <given-names>Q</given-names>
</name>
</person-group>. <article-title>ReactomePA: an R/Bioconductor package for reactome pathway analysis and visualization</article-title>. <source>Mol Biosyst</source>. (<year>2016</year>) <volume>12</volume>:<page-range>477&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1039/C5MB00663E</pub-id>
</citation>
</ref>
<ref id="B30">
<label>30</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ligtenberg</surname> <given-names>W</given-names>
</name>
</person-group>. <source>reactome.db: A set of annotation maps for reactome. R package version 1.68.0</source>.<publisher-name>Bioconductor</publisher-name> (<year>2019</year>).</citation>
</ref>
<ref id="B31">
<label>31</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Senosain</surname> <given-names>M</given-names>
</name>
<name>
<surname>Zou</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Patel</surname> <given-names>K</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>S</given-names>
</name>
<name>
<surname>Coullomb</surname> <given-names>A</given-names>
</name>
<name>
<surname>Rowe</surname> <given-names>DJ</given-names>
</name>
<etal/>
</person-group>. <article-title>Integrated multi-omics analysis of early lung adenocarcinoma links tumor biological features with predicted indolence or aggressiveness</article-title>. <source>Cancer Res Commun</source>. (<year>2023</year>) <volume>7</volume>:<page-range>1350&#x2013;65</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1158/2767-9764.CRC-22-0373</pub-id>
</citation>
</ref>
<ref id="B32">
<label>32</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Satija</surname> <given-names>R</given-names>
</name>
<name>
<surname>Farrell</surname> <given-names>J</given-names>
</name>
<name>
<surname>Gennert</surname> <given-names>D</given-names>
</name>
<name>
<surname>Schier</surname> <given-names>AF</given-names>
</name>
<name>
<surname>Regev</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Spatial reconstruction of single-cell gene expression data</article-title>. <source>Nat Biotechnol</source>. (<year>2015</year>) <volume>33</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nbt.3192</pub-id>
</citation>
</ref>
<ref id="B33">
<label>33</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Butler</surname> <given-names>A</given-names>
</name>
<name>
<surname>Hoffman</surname> <given-names>P</given-names>
</name>
<name>
<surname>Smibert</surname> <given-names>P</given-names>
</name>
<name>
<surname>Papalexi</surname> <given-names>E</given-names>
</name>
<name>
<surname>Satija</surname> <given-names>R</given-names>
</name>
</person-group>. <article-title>Integrating single-cell transcriptomic data across different conditions, technologies, and species</article-title>. <source>Nat Biotechnol</source>. (<year>2018</year>) <volume>36</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nbt.4096</pub-id>
</citation>
</ref>
<ref id="B34">
<label>34</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hao</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Hao</surname> <given-names>S</given-names>
</name>
<name>
<surname>Andersen-Nissen</surname> <given-names>E</given-names>
</name>
<name>
<surname>Mauck</surname> <given-names>WM</given-names> <suffix>3rd</suffix>
</name>
<name>
<surname>Zheng</surname> <given-names>S</given-names>
</name>
<name>
<surname>Butler</surname> <given-names>A</given-names>
</name>
<etal/>
</person-group>. <article-title>Integrated analysis of multimodal single-cell data</article-title>. <source>Cell</source>. (<year>2021</year>) <volume>184</volume>(<issue>13</issue>):<page-range>p3573&#x2013;87</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1101/2020.10.12.335331</pub-id>
</citation>
</ref>
<ref id="B35">
<label>35</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Aran</surname> <given-names>D</given-names>
</name>
<name>
<surname>Looney</surname> <given-names>A</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>L</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>E</given-names>
</name>
<name>
<surname>Fong</surname> <given-names>V</given-names>
</name>
<name>
<surname>Hsu</surname> <given-names>A</given-names>
</name>
<etal/>
</person-group>. <article-title>Reference-based analysis of lung single-cell sequencing reveals a transitional profibrotic macrophage</article-title>. <source>Nat Immunol</source>. (<year>2019</year>) <volume>20</volume>:<page-range>163&#x2013;72</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41590-018-0276-y</pub-id>
</citation>
</ref>
<ref id="B36">
<label>36</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chu</surname> <given-names>T</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Pe&#x2019;er</surname> <given-names>D</given-names>
</name>
<name>
<surname>Danko</surname> <given-names>CG</given-names>
</name>
</person-group>. <article-title>Cell type and gene expression deconvolution with BayesPrism enables Bayesian integrative analysis across bulk and single-cell RNA sequencing in oncology</article-title>. <source>Nat Cancer</source>. (<year>2022</year>) <volume>3</volume>:<page-range>505&#x2013;17</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s43018-022-00356-3</pub-id>
</citation>
</ref>
<ref id="B37">
<label>37</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dietrich</surname> <given-names>A</given-names>
</name>
<name>
<surname>Merotto</surname> <given-names>L</given-names>
</name>
<name>
<surname>Pelz</surname> <given-names>K</given-names>
</name>
<name>
<surname>Eder</surname> <given-names>B</given-names>
</name>
<name>
<surname>Zackl</surname> <given-names>C</given-names>
</name>
<name>
<surname>Reinisch</surname> <given-names>K</given-names>
</name>
<etal/>
</person-group>. <article-title>Benchmarking second-generation methods for cell-type deconvolution of transcriptomic data</article-title>. <source>biorxiv</source>. (<year>2024</year>). doi:&#xa0;<pub-id pub-id-type="doi">10.1101/2024.06.10.598226</pub-id>
</citation>
</ref>
<ref id="B38">
<label>38</label>
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Therneau</surname> <given-names>T</given-names>
</name>
</person-group>. <source>A Package for Survival Analysis in R. R package version 3.7-0</source> (<year>2024</year>). Available online at: <uri xlink:href="https://CRAN.R-project.org/package=survival">https://CRAN.R-project.org/package=survival</uri>. (accessed <access-date>August 05, 2024</access-date>)</citation>
</ref>
<ref id="B39">
<label>39</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Therneau</surname> <given-names>TM</given-names>
</name>
<name>
<surname>Grambsch</surname> <given-names>PM</given-names>
</name>
</person-group>. <source>Modeling survival data: extending the cox model</source>. <publisher-loc>New York</publisher-loc>: <publisher-name>Springer</publisher-name> (<year>2000</year>), ISBN: <isbn>ISBN 0-387-98784-3</isbn>.</citation>
</ref>
<ref id="B40">
<label>40</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wickham</surname> <given-names>H</given-names>
</name>
</person-group>. <source>ggplot2: elegant graphics for data analysis</source>. <publisher-loc>New York</publisher-loc>: <publisher-name>Springer-Verlag</publisher-name> (<year>2016</year>). Available at: <uri xlink:href="https://ggplot2.tidyverse.org">https://ggplot2.tidyverse.org</uri>, ISBN: <isbn>ISBN 978-3-319-24277-4</isbn>.</citation>
</ref>
<ref id="B41">
<label>41</label>
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Kassambara</surname> <given-names>A</given-names>
</name>
<name>
<surname>Kosinski</surname> <given-names>M</given-names>
</name>
<name>
<surname>Biecek</surname> <given-names>P</given-names>
</name>
</person-group>. <source>survminer: Drawing Survival Curves using &#x2018;gg- plot2</source> (<year>2021</year>). Available online at: <uri xlink:href="https://CRAN.R-project.org/package=survminer">https://CRAN.R-project.org/package=survminer</uri>. (accessed <access-date>August 05, 2024</access-date>)</citation>
</ref>
<ref id="B42">
<label>42</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Colaprico</surname> <given-names>A</given-names>
</name>
<name>
<surname>Silva</surname> <given-names>TC</given-names>
</name>
<name>
<surname>Olsen</surname> <given-names>C</given-names>
</name>
<name>
<surname>Garofano</surname> <given-names>L</given-names>
</name>
<name>
<surname>Cava</surname> <given-names>C</given-names>
</name>
<name>
<surname>Garolini</surname> <given-names>D</given-names>
</name>
<etal/>
</person-group>. <article-title>TCGAbiolinks: An R/Bioconductor package for integrative analysis of TCGA data</article-title>. <source>Nucleic Acids Res</source>. (<year>2015</year>) <volume>44</volume>(<issue>8</issue>):<elocation-id>e71</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkv1507</pub-id>
</citation>
</ref>
<ref id="B43">
<label>43</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Silva</surname> <given-names>CT</given-names>
</name>
<name>
<surname>Colaprico</surname> <given-names>A</given-names>
</name>
<name>
<surname>Olsen</surname> <given-names>C</given-names>
</name>
<name>
<surname>D'Angelo</surname> <given-names>F</given-names>
</name>
<name>
<surname>Bontempi</surname> <given-names>G</given-names>
</name>
<name>
<surname>Ceccarelli</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>TCGA Workflow: Analyze cancer genomics and epigenomics data using Bioconductor packages</article-title>. <source>F1000Research</source>. (<year>2016</year>) <volume>5</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.12688/f1000research</pub-id>
</citation>
</ref>
<ref id="B44">
<label>44</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mounir</surname> <given-names>M</given-names>
</name>
<name>
<surname>Lucchetta</surname> <given-names>M</given-names>
</name>
<name>
<surname>Silva</surname> <given-names>CT</given-names>
</name>
<name>
<surname>Olsen</surname> <given-names>C</given-names>
</name>
<name>
<surname>Bontempi</surname> <given-names>G</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>X</given-names>
</name>
<etal/>
</person-group>. <article-title>New functionalities in the TCGAbiolinks package for the study and integration of cancer data from GDC and GTEx</article-title>. <source>PloS Comput Biol</source>. (<year>2019</year>) <volume>15</volume>:<elocation-id>e1006701</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pcbi.1006701</pub-id>
</citation>
</ref>
<ref id="B45">
<label>45</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Xiong</surname>
</name>
<name>
<surname>Ning</surname> <given-names>Z</given-names>
</name>
</person-group>. <article-title>Implications of NKG2A in immunity and immune-mediated diseases</article-title>. <source>Front Immunol</source>. (<year>2022</year>) <volume>13</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fimmu.2022.960852</pub-id>
</citation>
</ref>
<ref id="B46">
<label>46</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Anfossi</surname> <given-names>N</given-names>
</name>
<name>
<surname>Andre</surname> <given-names>P</given-names>
</name>
<name>
<surname>Guia</surname> <given-names>S</given-names>
</name>
<name>
<surname>Falk</surname> <given-names>CS</given-names>
</name>
<name>
<surname>Roetynck</surname> <given-names>S</given-names>
</name>
<name>
<surname>Stewart</surname> <given-names>CA</given-names>
</name>
<etal/>
</person-group>. <article-title>Human NK cell education by inhibitory receptors for MHC class I</article-title>. <source>Immunity</source>. (<year>2006</year>) <volume>25</volume>:<page-range>331&#x2013;42</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.immuni.2006.06.013</pub-id>
</citation>
</ref>
<ref id="B47">
<label>47</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Comet</surname> <given-names>N</given-names>
</name>
<name>
<surname>Aguilo</surname> <given-names>J</given-names>
</name>
<name>
<surname>Rathore</surname> <given-names>M</given-names>
</name>
<name>
<surname>Catal&#xe1;n</surname> <given-names>E</given-names>
</name>
<name>
<surname>Garaude</surname> <given-names>J</given-names>
</name>
<name>
<surname>Uz&#xe9;</surname> <given-names>G</given-names>
</name>
<etal/>
</person-group>. <article-title>IFN<italic>&#x3b1;</italic> signaling through PKC-<italic>&#x3b8;</italic> is essential for antitumor NK cell function</article-title>. <source>OncoImmunol</source>. (<year>2014</year>) <volume>3</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.4161/21624011.2014.948705</pub-id>
</citation>
</ref>
<ref id="B48">
<label>48</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Poli</surname> <given-names>A</given-names>
</name>
<name>
<surname>Michel</surname> <given-names>T</given-names>
</name>
<name>
<surname>The&#xb4;re&#xb4;sine</surname> <given-names>M</given-names>
</name>
<name>
<surname>Andr&#xe8;s</surname> <given-names>E</given-names>
</name>
<name>
<surname>Hentges</surname> <given-names>F</given-names>
</name>
<name>
<surname>Zimmer</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>CD56bright natural killer (NK) cells: an important NK cell subset</article-title>. <source>Immunology</source>. (<year>2009</year>) <volume>4</volume>:<page-range>458&#x2013;65</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/j.1365-2567.2008.03027.x</pub-id>
</citation>
</ref>
<ref id="B49">
<label>49</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Post</surname> <given-names>M</given-names>
</name>
<name>
<surname>Cuapio</surname> <given-names>A</given-names>
</name>
<name>
<surname>Osl</surname> <given-names>M</given-names>
</name>
<name>
<surname>Lehmann</surname> <given-names>D</given-names>
</name>
<name>
<surname>Resch</surname> <given-names>U</given-names>
</name>
<name>
<surname>Davies</surname> <given-names>DM</given-names>
</name>
<etal/>
</person-group>. <article-title>The transcription factor ZNF683/HOBIT regulates human NK-cell development</article-title>. <source>Front Immunol</source>. (<year>2017</year>) <volume>8</volume>:<elocation-id>535.</elocation-id> doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fimmu.2017.00535</pub-id>
</citation>
</ref>
<ref id="B50">
<label>50</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Marquardt</surname> <given-names>N</given-names>
</name>
<name>
<surname>Kekalainen</surname> <given-names>E</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>P</given-names>
</name>
<name>
<surname>Lourda</surname> <given-names>M</given-names>
</name>
<name>
<surname>Wilson</surname> <given-names>JN</given-names>
</name>
<name>
<surname>Scharenberg</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>Unique transcriptional and protein-expression signature in human lung tissue-resident NK cells</article-title>. <source>Nat Commun</source>. (<year>2019</year>) <volume>10</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41467-019-11632-9</pub-id>
</citation>
</ref>
<ref id="B51">
<label>51</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Satija</surname> <given-names>R</given-names>
</name>
<name>
<surname>Shalek</surname> <given-names>AK</given-names>
</name>
</person-group>. <article-title>Heterogeneity in immune responses: from populations to single cells</article-title>. <source>Trends Immunol</source>. (<year>2014</year>) <volume>35</volume>:<page-range>219&#x2013;29</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.it.2014.03.004</pub-id>
</citation>
</ref>
<ref id="B52">
<label>52</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cong</surname> <given-names>J</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>H</given-names>
</name>
</person-group>. <article-title>Natural killer cells in the lungs</article-title>. <source>Front Immunol</source>. (<year>2019</year>) <volume>10</volume>:<elocation-id>1416</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fimmu.2019.01416</pub-id>
</citation>
</ref>
<ref id="B53">
<label>53</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schmidt</surname> <given-names>L</given-names>
</name>
<name>
<surname>Eskiocak</surname> <given-names>B</given-names>
</name>
<name>
<surname>Kohn</surname> <given-names>R</given-names>
</name>
<name>
<surname>Dang</surname> <given-names>C</given-names>
</name>
<name>
<surname>Joshi</surname> <given-names>NS</given-names>
</name>
<name>
<surname>DuPage</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>Enhanced adaptive immune responses in lung adenocarcinoma through natural killer cell stimulation</article-title>. <source>Proc Natl Acad Sci U S A</source>. (<year>2019</year>) <volume>116</volume>:<page-range>17460&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1073/pnas.1904253116</pub-id>
</citation>
</ref>
<ref id="B54">
<label>54</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hsu</surname> <given-names>J</given-names>
</name>
<name>
<surname>Hodgins</surname> <given-names>JJ</given-names>
</name>
<name>
<surname>Marathe</surname> <given-names>M</given-names>
</name>
<name>
<surname>Nicolai</surname> <given-names>CJ</given-names>
</name>
<name>
<surname>Bourgeois-Daigneault</surname> <given-names>M-C</given-names>
</name>
<name>
<surname>Trevino</surname> <given-names>TN</given-names>
</name>
<etal/>
</person-group>. <article-title>Contribution of NK cells to immunotherapy mediated by PD-1/PD-L1 blockade</article-title>. <source>J Clin Invest</source>. (<year>2018</year>) <volume>128</volume>:<page-range>4654&#x2013;68</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1172/JCI99317</pub-id>
</citation>
</ref>
<ref id="B55">
<label>55</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huntington</surname> <given-names>ND</given-names>
</name>
<name>
<surname>Cursons</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>amp]]amp; J. Rautela. The cancer&#x2013;natural killer cell immunity cycle</article-title>. <source>Nat Rev Cancer</source>. (<year>2020</year>) <volume>20</volume>:<page-range>437&#x2013;45</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41568-020-0272-z</pub-id>
</citation>
</ref>
<ref id="B56">
<label>56</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Davis-Marcisak</surname> <given-names>EF</given-names>
</name>
<name>
<surname>Fitzgerald</surname> <given-names>AA</given-names>
</name>
<name>
<surname>Kessler</surname> <given-names>MD</given-names>
</name>
<name>
<surname>Kessler</surname> <given-names>MD</given-names>
</name>
<name>
<surname>Danilova</surname> <given-names>L</given-names>
</name>
<name>
<surname>Jaffee</surname> <given-names>EM</given-names>
</name>
<name>
<surname>Zaidi</surname> <given-names>N</given-names>
</name>
<etal/>
</person-group>. <article-title>Transfer learning between preclinical models and human tumors identifies a conserved NK cell activation signature in anti-CTLA-4 responsive tumors</article-title>. <source>Genome Med</source>. (<year>2021</year>) <volume>13</volume>:<fpage>129</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13073-021-00944-5</pub-id>
</citation>
</ref>
<ref id="B57">
<label>57</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>W</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Li</surname> <given-names>F</given-names>
</name>
</person-group>. <article-title>Natural killer cell dysfunction in cancer and new strategies to utilize NK cell potential for cancer immunotherapy</article-title>. <source>Mol Immunol</source>. (<year>2022</year>) <volume>144</volume>:<fpage>58</fpage>&#x2013;<lpage>70</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.molimm.2022.02.015</pub-id>
</citation>
</ref>
<ref id="B58">
<label>58</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Danaher</surname> <given-names>P</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Nelson</surname> <given-names>B</given-names>
</name>
<name>
<surname>Griswold</surname> <given-names>M</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Piazza</surname> <given-names>E</given-names>
</name>
<etal/>
</person-group>. <article-title>Advances in mixed cell deconvolution enable quantification of cell types in spatial transcriptomic data</article-title>. <source>Nat Commun</source>. (<year>2022</year>) <volume>13</volume>:<fpage>385</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41467-022-28020-5</pub-id>
</citation>
</ref>
<ref id="B59">
<label>59</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Judge</surname> <given-names>S</given-names>
</name>
<name>
<surname>Murphy</surname> <given-names>W</given-names>
</name>
<name>
<surname>RJ</surname> <given-names>C</given-names>
</name>
</person-group>. <article-title>Characterizing the dysfunctional NK cell: assessing the clinical relevance of exhaustion, anergy, and senescence</article-title>. <source>Front Cell Infect Microbiol</source>. (<year>2020</year>) <volume>10</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fcimb.2020.00049</pub-id>
</citation>
</ref>
<ref id="B60">
<label>60</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Isaacson</surname> <given-names>B</given-names>
</name>
<name>
<surname>Mandelboim</surname> <given-names>O</given-names>
</name>
</person-group>. <article-title>Sweet killers: NK cells need glycolysis to kill tumors</article-title>. <source>Cell Metab</source>. (<year>2018</year>) <volume>28</volume>:<page-range>183&#x2013;4</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cmet.2018.07.008</pub-id>
</citation>
</ref>
<ref id="B61">
<label>61</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Leek</surname> <given-names>JT</given-names>
</name>
<name>
<surname>Johnson</surname> <given-names>WE</given-names>
</name>
<name>
<surname>Parker</surname> <given-names>HS</given-names>
</name>
<name>
<surname>Jaffe</surname> <given-names>AE</given-names>
</name>
<name>
<surname>Storey</surname> <given-names>JD</given-names>
</name>
</person-group>. <article-title>The sva package for removing batch effects and other unwanted variation in high-throughput experiments</article-title>. <source>Bioinformatics</source>. (<year>2024</year>) <volume>28</volume>:<page-range>882&#x2013;3</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/bts034</pub-id>
</citation>
</ref>
<ref id="B62">
<label>62</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Luo</surname> <given-names>W</given-names>
</name>
<name>
<surname>Brouwer</surname> <given-names>C</given-names>
</name>
</person-group>. <article-title>Pathview: an R/Bioconductor package for pathway-based data integration and visualization</article-title>. <source>Bioinformatics</source>. (<year>2013</year>) <volume>29</volume>:<page-range>1830&#x2013;1</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btt285</pub-id>
</citation>
</ref>
<ref id="B63">
<label>63</label>
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Blighe</surname> <given-names>K</given-names>
</name>
<name>
<surname>Lun</surname> <given-names>A</given-names>
</name>
</person-group>. <source>PCAtools: PCAtools: Everything Principal Components Analysis. R package version 2.16.0</source> (<year>2024</year>). Available online at: <uri xlink:href="https://github.com/kevinblighe/PCAtools">https://github.com/kevinblighe/PCAtools</uri>. (accessed <access-date>August 10, 2024</access-date>)</citation>
</ref>
</ref-list>
</back>
</article>