<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Immunol.</journal-id>
<journal-title>Frontiers in Immunology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Immunol.</abbrev-journal-title>
<issn pub-type="epub">1664-3224</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fimmu.2022.853213</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Immunology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Joint Analysis of Microbial and Immune Cell Abundance in Liver Cancer Tissue Using a Gene Expression Profile Deconvolution Algorithm Combined With Foreign Read Remapping</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Ai</surname><given-names>Dongmei</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>*</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/662747"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xing</surname><given-names>Yonglian</given-names>
</name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1634684"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname><given-names>Qingchuan</given-names>
</name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1729280"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname><given-names>Yishu</given-names>
</name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1128134"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname><given-names>Xiuqin</given-names>
</name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1729284"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname><given-names>Gang</given-names>
</name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/718732"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Xia</surname><given-names>Li C.</given-names>
</name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>*</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/218631"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Basic Experimental Center of Natural Science, University of Science and Technology Beijing</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>School of Mathematics and Physics, University of Science and Technology Beijing</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>National Engineering Laboratory for Agri-Product Quality Traceability, Beijing Technology and Business University</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>School of Mathematics, South China University of Technology</institution>, <addr-line>Guangzhou</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Hongda Liu, Nanjing Medical University, China</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Mahdi Hasanipanah, Duy Tan University, Vietnam; Mingcong Deng, Tokyo University of Agriculture and Technology, Japan</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Dongmei Ai, <email xlink:href="mailto:aidongmei@ustb.edu.cn">aidongmei@ustb.edu.cn</email>; Li C. Xia, <email xlink:href="mailto:lcxia@scut.edu.cn">lcxia@scut.edu.cn</email></p>
</fn>
<fn fn-type="other" id="fn002">
<p>This article was submitted to Cancer Immunity and Immunotherapy, a section of the journal Frontiers in Immunology</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>14</day>
<month>04</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>853213</elocation-id>
<history>
<date date-type="received">
<day>12</day>
<month>01</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>07</day>
<month>03</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Ai, Xing, Zhang, Wang, Liu, Liu and Xia</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Ai, Xing, Zhang, Wang, Liu, Liu and Xia</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Recent transcriptomics and metagenomics studies showed that tissue-infiltrating immune cells and bacteria interact with cancer cells to shape oncogenesis. This interaction and its effects remain to be elucidated. However, it is technically difficult to co-quantify immune cells and bacteria in their respective microenvironments. To address this challenge, we herein report the development of a complete a bioinformatics pipeline, which accurately estimates the number of infiltrating immune cells using a novel Particle Swarming Optimized Support Vector Regression (PSO-SVR) algorithm, and the number of infiltrating bacterial using foreign read remapping and the GRAMMy algorithm. It also performs systematic differential abundance analyses between tumor-normal pairs. We applied the pipeline to a collection of paired liver cancer tumor and normal samples, and we identified bacteria and immune cell species that were significantly different between tissues in terms of health status. Our analysis showed that this dual model of microbial and immune cell abundance had a better differentiation (84%) between healthy and diseased tissue. <italic>Caldatribacterium</italic> sp., <italic>Acidaminococcaceae</italic> sp., <italic>Planctopirus</italic> sp., <italic>Desulfobulbaceae</italic> sp.,<italic>Nocardia farcinica</italic> as well as regulatory T cells (Tregs), resting mast cells, monocytes, M2 macrophases, neutrophils were identified as significantly different (Mann Whitney Test, FDR&lt; 0.05). Our open-source software is freely available from GitHub at <uri xlink:href="https://github.com/gutmicrobes/PSO-SVR.git">https://github.com/gutmicrobes/PSO-SVR.git</uri>.</p>
</abstract>
<kwd-group>
<kwd>tumor microenvironment</kwd>
<kwd>RNA-seq</kwd>
<kwd>gene expression profiling</kwd>
<kwd>support vector regression</kwd>
<kwd>particle swarm algorithm</kwd>
</kwd-group>
<contract-num rid="cn001">61873027</contract-num>
<contract-sponsor id="cn001">National Natural Science Foundation of China<named-content content-type="fundref-id">10.13039/501100001809</named-content>
</contract-sponsor>
<counts>
<fig-count count="5"/>
<table-count count="4"/>
<equation-count count="10"/>
<ref-count count="44"/>
<page-count count="12"/>
<word-count count="6586"/>
</counts>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<title>Introduction</title>
<p>Contrary to the intuition of most people, bacteria are present in almost every part of the human body, with over 1,000 species in the gut alone. Our commensal bacteria maintain a dynamic balance with each other, and their imbalance can lead to a variety of diseases in the human body, including many cancers (<xref ref-type="bibr" rid="B1">1</xref>). Indeed, approximately 20% of all lethal cancers in humans are induced by or associated with microorganisms (<xref ref-type="bibr" rid="B2">2</xref>). As a component of the tumor microenvironment (TME), bacteria can actively promote tumor development, as well as autoimmunity, contributing to mortality (<xref ref-type="bibr" rid="B3">3</xref>, <xref ref-type="bibr" rid="B4">4</xref>).</p>
<p>Liver cancer is a malignant tumor and approximately 780,000 people were diagnosed with liver cancer worldwide yearly (<xref ref-type="bibr" rid="B5">5</xref>). Previous studies analyzed microbes inferred from the whole-genome sequencing data of liver cancer biopsies. They revealed a strong association between the occurrence of liver cancer and certain bacterial flora. For examples: Moffatt et&#xa0;al. found that the bacterium <italic>Porphyromonas gingivalis</italic> promotes hepatocellular carcinoma by affecting host cell signaling and thus cytokine response, cell cycle, and apoptosis (<xref ref-type="bibr" rid="B6">6</xref>); Garner et al. found that methoxysterigmatocystin, O-methylsterigmatocystin, and other metabolites induced DNA repair-deficient bacterial lesions and thus initiates hepatocellular carcinogenesis (<xref ref-type="bibr" rid="B7">7</xref>); Mangul et&#xa0;al. found that <italic>Escherichia coli</italic>, <italic>Streptococcus faecalis</italic>, and <italic>Clostridium parvum</italic> could act together to significantly promote liver tumorigenesis. However, such activity could be inhibited by the addition of intestinal bacteria (e.g., <italic>Bifidobacterium longum</italic> and <italic>Lactobacillus acidophilus</italic>) and rectal fungi (<xref ref-type="bibr" rid="B8">8</xref>).</p>
<p>However, these studies were mostly focused on the effects of intestinal bacteria upon hepatocellular carcinoma (HCC) while few have examined the bacteria present within hepatocellular carcinoma tissues. Tumor-infiltrating bacteria could be analyzed using sequencing reads of non-human source, however, these unmapped foreign reads are often overlooked. In fact, it is possible to accurately estimate microbial abundance in tissue biopsies using foreign reads remapping (<xref ref-type="bibr" rid="B9">9</xref>). In this study, we applied our previously developed GRAMMy (<xref ref-type="bibr" rid="B10">10</xref>) tool to identify the bacteria that are associated with liver cancer by adding a series of analyses on unmapped foreign reads filtered from RNA-seq data so as to estimate the relative abundance of infiltrating bacteria present within the tissue.</p>
<p>Studies have also shown that the infiltration of various immune cell populations, including monocytes/macrophages, natural killer cells (NK), NKT cells and T cells, is the main pathogenic feature for oncogenesis or other lesions in liver (<xref ref-type="bibr" rid="B11">11</xref>). Rohr-Udilova et&#xa0;al. observed considerable differences in the composition of immune cells between HCC and healthy liver. Pushpa Hegde et&#xa0;al. found a decrease in circulating mucosal-associated invariant T cells in patients with alcoholic or nonalcoholic fatty liver disease-related cirrhosis (<xref ref-type="bibr" rid="B12">12</xref>). Functional immune level changes were detected in a group of healthy people and liver transplant recipients (<xref ref-type="bibr" rid="B13">13</xref>). Microbial infection is very likely to occur in the clinical treatment of liver diseases, especially in the treatment of liver transplantation; bacterial infection being the most common (incidence of 31.45%) (<xref ref-type="bibr" rid="B14">14</xref>). Monitoring the immune status of transplant recipients is essential to predict the risk of infection.</p>
<p>Thus, researchers have just begun to understand the regulatory role of bacteria in the development of cancer, as well as oncogenesis at the interface of bacteria/immune cell interaction. Immune cells alone are a key component of the TME and play a critical role in cancer development and immunization. Therefore, in order to understand the dynamics of the TME, it is equally important to understand the presumed synergistic dynamics between bacteria and tissue-infiltrating immune cells. Computational deconvolution methods have become a convenient choice to assess tissue-infiltrating immune cells by formulating the problem as a system of equations used to describe the gene expression of a sample as a weighted sum of the expression profiles of mixed cell types. That is, once given the&#xa0;immune cell-type characteristic matrix and overall gene expression, by solving the deconvolution problem, immune cell&#xa0;types and levels can be reasonably co-quantified with bacteria using mRNA-seq data without resorting to additional experiments.</p>
<p>The deconvolution problem could be solved in several ways, such as by Support Vector Regression (SVR) (e.g., CIBERSORT) (<xref ref-type="bibr" rid="B15">15</xref>), linear least squares regression (e.g., TIMER) (<xref ref-type="bibr" rid="B16">16</xref>) and constrained least squares regression (e.g., EPIC and MCP-counter) (<xref ref-type="bibr" rid="B17">17</xref>, <xref ref-type="bibr" rid="B18">18</xref>). CIBERSORT uses v-SVR linear regression to solve the linear equation model based on microarray data (RNA-seq data can also be used). The coefficient of the regression model represents the relative proportion of 22 immune cell types. TIMER calculates the abundance of six immune cells including CD4 T cells, CD8 T cells and B cells based on the constrained least square method, which better solved the multicollinearity problem caused by high-dimensional features. EPIC uses least square regression to infer the mRNA number of six immune cells and other cell types, and then converts the mRNA number into the relative proportion of related immune cells. Finally, through the verification of tumor active genes and clinical data of many patients, it is found that epic results are clinically applicable. Finally, the murine Microenvironment Cell Population counter (MCP-counter) was based on highly specific transcriptomic markers, allows a robust quantification of the count number of eight immune cell types and two stromal cell populations in heterogeneous tissues based on transcriptomic data, which represent the cell content.</p>
<p>At present, the main method of optimizing the SVR deconvolution problem uses an heuristic algorithm. Particle swarm optimization (PSO) is simple and easy to operate algorithm, and the optimization search process of PSO take into account the local search ability and global search ability, which can greatly improve the accuracy of SVR solution. At present, PSO- SVR algorithm has been applied in many fields. For example, Mingcong Deng et&#xa0;al. applied PSO- SVR algorithm to robotics in 2014 to predict the ball receiving time of a robot player (<xref ref-type="bibr" rid="B19">19</xref>). In 2017, the same team used PSO to optimize the parameters of the generalized Gaussian kernel model and confirmed that the algorithm also shows good performance in a pneumatic bending rubber actuator control system (<xref ref-type="bibr" rid="B20">20</xref>). However, few teams have applied PSO- SVR to the field of tumor immune cell infiltration. This study aims to further study the universality of this algorithm and its advantages or limitations compared with other algorithms in this field through the results of liver cancer samples under this algorithm.</p>
<p>Mohammadi et&#xa0;al. evaluated several methods for solving the deconvolution problem and found that combining the loss function with regularization can improve the solution performance in the presence of highly correlated cell types in the mixture (<xref ref-type="bibr" rid="B21">21</xref>). However, the effect of the size of the regularization parameters on the effect of immune cell counting is unknown and the SVR model accuracy is affected by the initial parameters, such as the kernel function coefficients and penalty factors. To increase the robustness of the algorithm and reduce the influence of the initial self-defined parameters on the proportional counting results of immune cells, we introduced the particle swarm optimized support vector regression (PSO-SVR) algorithm, which uses PSO -a powerful parameter iterative optimization solution algorithm to improve the accuracy of the SVR solution. The algorithm was also used to compare with CIBERSORTX (<xref ref-type="bibr" rid="B22">22</xref>), EPIC, and MCP-counter on three real data sets, which confirmed its better accuracy.</p>
<p>Finally, we used the PSO-SVR algorithm and foreign read remapping to analyze a collection of paired liver cancer tumor and normal samples, identified bacteria and immune cell species that are significantly different between samples, and evaluated their joint effects by predicting the tumoral or normal pathological status of their originating tissue. The joint model identified B cells, T cells and <italic>Caldatribacterium</italic> sp., <italic>Magnetobacteriaceae</italic> sp. as pathological markers and together they were powerfully predictive for the pathologic status. Based on results from the three cases, classification accuracy when using only a single input feature is lower - only bacteria: 0.70, only immune cell feature: 0.74, and both features: 0.84. Such preliminary results are useful in resolving standing low response issue of immune checkpoint inhibitor-based immunotherapy (<xref ref-type="bibr" rid="B23">23</xref>, <xref ref-type="bibr" rid="B24">24</xref>), in that microbial markers are potentially actionable targets (<xref ref-type="bibr" rid="B25">25</xref>, <xref ref-type="bibr" rid="B26">26</xref>). As the joint immunomodulatory effects of the microbiome and immune cells are further elucidated for other cancer types, we could expect to find more novel biomarkers that could be intervened upon to improve normal tissue&#x2019;s immune response against tumor.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<title>Materials and Methods</title>
<sec id="s2_1">
<title>The Liver Cancer mRNA Sequence Data Set</title>
<p>The mRNA-seq raw sequence data in BAM format of liver cancer patients&#x2019; normal and tumor tissue biopsies for this study were downloaded from the Seven Bridges Cancer Genomics Cloud (CGC <uri xlink:href="https://www.cancergenomicscloud.org/">https://www.cancergenomicscloud.org/</uri>). The data set included 98 samples (49 pairs of RNA-Seq sequencing samples of primary liver cancer tumor tissue and adjacent normal tissue). The read sequences that did not map to the reference genome GRCH38 were extracted from the BAM files. These foreign reads were then mapped to the RefSeq (NCBI Reference Sequence Database) (<uri xlink:href="https://www.ncbi.nlm.nih.gov/refseq/">https://www.ncbi.nlm.nih.gov/refseq/</uri>) -&#xa0;a large collection of bacterial genomes. The mapping results were then input to the GRAMMy pipeline for the relative abundance estimation of infiltrating bacteria. The obtained expression and microbial profiles were used for downstream analysis. The data processing flow is shown in <xref ref-type="fig" rid="f1"><bold>Figure&#xa0;1</bold></xref>.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Flow chart of RNA-Seq sequencing data processing.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-13-853213-g001.tif"/>
</fig>
<p>We first filtered the reads that were not matched to the human genome reference GRCH38 from the RNA-Seq samples in BAM format with SAMtools, which included reads that were not mapped at both ends and reads that were not mapped at one end (mapped at one end but not at the other). Apply Fastp to remove low-quality reads and partial reads, and excise poor quality bases. Use the BWA tool to remap the unmapped reads to a comprehensive microbial reference library RefSeq, containing the full genome sequence of 23,790 bacteria and archaea. We then applied GRAMMy algorithm to estimate the bacteria relative abundance.</p>
<p>The mRNA analysis pipeline begins with the Alignment Workflow, which is performed using a two-pass method with STAR. STAR aligns each read group separately and then merges the resulting alignments into one. Following alignment, BAM files are processed through the RNA Expression Workflow to determine RNA expression levels. The reads mapped to each gene are enumerated using HT-Seq-Count. Expression values are provided in a tab-delimited format.</p>
</sec>
<sec id="s2_2">
<title>Validation Data Sets for ImmuneCell Infiltration</title>
<p>Three real data sets were used to validate the PSO-SVR algorithm as compared to the CIBERSORTX, EPIC, and MCP-counter algorithms. The first validation data set from the CIBERSORT article  consists of the gene expression profiles of 20 peripheral blood mono nuclear cells (PBMC) samples and the ratio of immune cells determined through flow cytometry for the same samples. The second data set comes from the National Center for Biotechnology Information Search database (NCBI database) under the GEO Datasets GSE64385. It consists of the gene expression profile of 10 colorectal cancer (CRC) samples and the composition of immune cells as determined by immunohistochemistry of the same samples. The third data set from CIBERSORTX consists of the gene expression profiles of 19 melanoma samples and the composition of immune cells obtained through single-cell sequencing technique of the same samples. The full description of these data sets is shown in <xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Table&#xa0;1</bold></xref>.</p>
</sec>
<sec id="s2_3">
<title>Extracting Unmapped Reads</title>
<p>We first filtered the reads that did not matched to the human genome reference GRCH38 from the RNA-Seq samples in BAM format using SAMtools (<xref ref-type="bibr" rid="B27">27</xref>). The filtered reads included those that were not mapped at both ends and those that were not mapped at one end (mapped at one end but not the other). The process was as follows:</p>
<p>Extracting unmapped single reads: we used the command &#x201c;samtools view -u -f 4 -F 264&#x201d; to extract single-end unmapped reads in the format of FASTQ (<xref ref-type="bibr" rid="B28">28</xref>).</p>
<p>Extracting unmapped paired-end reads: we used the command &#x201c;samtools view -u -f 12 -F 256&#x201d; to extract paired-end unmapped reads with the format FASTQ.</p>
</sec>
<sec id="s2_4">
<title>Sequencing Quality Control</title>
<p>Sequencing quality issues like low-confidence bases and sequence-specific bias complicate mRNA-seq analyses (<xref ref-type="bibr" rid="B29">29</xref>). We applied comprehensive quality control (QC) before analyses (<xref ref-type="bibr" rid="B30">30</xref>). We applied Fastp (<xref ref-type="bibr" rid="B31">31</xref>) to remove low-quality reads and partial reads, and to excise poor quality bases.</p>
<p>For the extracted unmapped paired-end reads, we used the command &#x201c;fastp -q 0 -u 100 -n 10 -l 36 -A -G -M 0 -i&#x201d; to delete sequences with base quality lower than 40% of Q15, sequences with N greater than 5, sequences with length less than 36, and broke up sequences, respectively. For the extracted unmapped single reads, we used the command &#x201c;fastp -q 0 -u 100 -n 10 -l 36 -A -G -M 0 -i&#x201d; to perform the same QC process on single-ended sequenced sequences. To merge the extracted single-end and paired-end unmapped reads, we used the SMS2 (<xref ref-type="bibr" rid="B32">32</xref>) software to add a reverse complementary sequence to the unmapped single reads. We then converted the resulting FASTQ formatted reads into the FASTA (<xref ref-type="bibr" rid="B33">33</xref>) format through SeqKit (<xref ref-type="bibr" rid="B34">34</xref>).</p>
<p>We then used the BWA tool (<xref ref-type="bibr" rid="B35">35</xref>) to remap the unmapped reads to a comprehensive microbial reference library RefSeq, containing the full genome sequence of 23,790 bacteria and archaea. We then applied the GRAMMy algorithm to estimate the relative abundance of bacteria to mitigate the problem of ambiguous mapping of short reads to relative reference sequences.</p>
</sec>
<sec id="s2_5">
<title>Statistical Analysis</title>
<p>The Mann Whitney U test, also known as &#x201c;Mann Whitney rank sum test&#x201d;, was proposed by H.B.mann and D.Rwhitney in 1947. It assumes that any two samples are from two populations that are exactly the same except the population mean, in order to test for significant difference between the mean of the two populations.</p>
<p>Assuming that the mean values of two populations exist, they are recorded as <italic>&#x3bc;</italic><sub>1</sub>,<italic>&#x3bc;</italic><sub>2</sub> respectively. With only one translation difference between <italic>f</italic><sub>1</sub> and <italic>f</italic><sub>2</sub> at most, we get: <italic>&#x3bc;</italic><sub>1</sub> = <italic>&#x3bc;</italic><sub>2</sub> &#x2013; &#x3b1;. The assumptions to be tested are as follows:</p>
<disp-formula>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo> <mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&lt;</mml:mo>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&gt;</mml:mo>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow> </mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The steps of Mann Whitney U test are:</p>
<list list-type="order">
<list-item>
<p>Randomly select two independent random samples with capacity of <italic>N<sub>A</sub>
</italic> and <italic>N<sub>B</sub>
</italic> from two populations A and B, arrange (<italic>N<sub>A</sub>
</italic> + <italic>N<sub>B</sub>
</italic>) observations in order of size. If the same observations exist, the average of their bit order is used.</p>
</list-item>
<list-item>
<p>Calculate the grade and <italic>T<sub>A</sub>
</italic> and <italic>T<sub>B</sub>
</italic> of two samples.</p>
</list-item>
<list-item>
<p>The formula of Mann Whitney U test can be given according to <italic>T<sub>A</sub>
</italic> and <italic>T<sub>B</sub>
</italic>. The two calculated U values are not equal, but their sum is always equal to <italic>N<sub>A</sub>N<sub>B</sub>
</italic>, that is, <italic>U<sub>A</sub>
</italic> + <italic>U<sub>B</sub>
</italic> = <italic>N<sub>A</sub>N<sub>B</sub>
</italic>. If <italic>N<sub>A</sub>
</italic> &lt; 20 and <italic>N<sub>B</sub>
</italic> &lt; 20, the test statistics are:</p>
</list-item>
</list>
<disp-formula>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:msub>
<mml:mi>U</mml:mi>
<mml:mi>A</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>A</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>B</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>A</mml:mi>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>A</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">/</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>A</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:msub>
<mml:mi>U</mml:mi>
<mml:mi>B</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>A</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>B</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>B</mml:mi>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>B</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">/</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>B</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In the test, because the critical value table of Mann Whitney U test only gives a smaller critical value, the smaller U value in <italic>U<sub>A</sub>
</italic> and <italic>U<sub>B</sub>
</italic> is used as the test statistic.</p>
<p>4. Select the smaller U value to compare with the critical value of U. if <italic>U</italic> &gt; <italic>U<sub>&#x3b1;</sub>
</italic>(<italic>&#x3b1;</italic> = 0.05, accept the original assumption <italic>H</italic><sub>0</sub>. If <italic>U</italic> &lt; <italic>U<sub>&#x3b1;</sub>
</italic>(<italic>&#x3b1;</italic> = 0.05, Then reject <italic>H</italic><sub>0</sub> and accept <italic>H</italic><sub>1</sub>. The acceptance domain is the same as Wilcoxon test. U test can also be divided into small samples and large samples. In case of small samples, the critical values of U have been compiled into a table. In large samples, the distribution of U tends to be normal, so it can be treated by normal approximation.</p>    <p>The loss function is an index to measure the performance of the prediction model in predicting the expected results. The commonly used loss functions are mean squared error (MSE) and root mean squared error (RMSE). Since MSE squares the error (y &#x2013; y^predicted = e), if e &gt; 1, the value of the error will increase a lot. If there is an outlier in our data, the value of e will be very high and will be much greater than &#x2502;e&#x2502;. This will make the model with MSE loss give a higher weight to outliers. In order to minimize this outlier data point, we use the RMSE value, namely root mean square error, but at the expense of the prediction effect of other normal data points, which will eventually reduce the overall performance of the model.</p>
<disp-formula>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mstyle scriptlevel="+1">
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mn>2</mml:mn>
</mml:mfrac>
</mml:mstyle>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Correlation analysis refers to the analysis of two or more variable elements with correlation, so as to measure the correlation degree of two variable factors Correlation analysis can be carried out only when there is a certain connection or probability between the elements of correlation.</p>
<p>(1) Pearson correlation coefficient</p>    <p>Given two continuous variables x and y, the Pearson correlation coefficient is defined as:</p>
<disp-formula>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:mi>&#x3c1;</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3a3;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:msubsup>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:msubsup>
<mml:mi>&#x3a3;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:msubsup>
<mml:mi>&#x3a3;</mml:mi>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mstyle scriptlevel="+1">
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mn>2</mml:mn>
</mml:mfrac>
</mml:mstyle>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im1">
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im2">
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> are the mean values of the variables x and y respectively.</p>
<p>&#x3c1; close to 0 indicates that there is no correlation between the two variables; whereas close 1 or - 1 indicates that the two variables are strongly correlated.</p>
<p>(2) Spearman correlation coefficient</p>    <p>Spearman correlation coefficient is defined as Pearson correlation coefficient &#x3c1; between hierarchical variables. For samples with a sample size of N, N original data are converted into hierarchical data. Compared with Pearson correlation coefficient, Spearman correlation coefficient is insensitive to data errors and extreme values, which is defined as.</p>
<disp-formula>
<mml:math display="block" id="M11">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c1;</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3a3;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:msubsup>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>R</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>S</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:msubsup>
<mml:mi>&#x3a3;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>R</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:msubsup>
<mml:mi>&#x3a3;</mml:mi>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>S</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mstyle scriptlevel="+1">
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mn>2</mml:mn>
</mml:mfrac>
</mml:mstyle>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where R and S are the grades of observed values i respectively, <inline-formula>
<mml:math display="inline" id="im3">
<mml:mover accent="true">
<mml:mi>R</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im4">
<mml:mover accent="true">
<mml:mi>S</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> are the average grades of variables x and y respectively, and N is the total number of observed values.</p>    <p>Logistic regression, also known as log probability regression, is a machine learning method used to solve the binary classification problem, which is used to estimate the possibility of something. It does not need to scale the input features, and the interpretability of the model is very good. The influence of different features on the final result can be seen from the weight of features. We fit the following regularization model to binary features as:</p>
<disp-formula>    <mml:math display="block" id="M13">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo> <mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>n</mml:mi>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>T</mml:mi>
</mml:msubsup>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mstyle>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>T</mml:mi>
</mml:msubsup>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow> <mml:mo>]</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mi>&#x3bc;</mml:mi>
<mml:mrow>
<mml:mo>[</mml:mo> <mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mo>|</mml:mo> <mml:mi>&#x3b2;</mml:mi> <mml:mo>|</mml:mo>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:mfrac>
<mml:mo>+</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mo>|</mml:mo> <mml:mi>&#x3b2;</mml:mi> <mml:mo>|</mml:mo>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow> <mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where &#x3b2; is the regression coefficient. Parameter &#x3b1; is the balanced lasso (L1) and ridge (L2) regularization, and &#x3bb; determines their weights.</p>
<p>Therefore, we use the logistic regression classifier to classify 98 samples. Their corresponding bacterial relative abundance data and immune cell proportion data were used as input characteristics, and the sample status (normal/tumor) was used as the prediction variable 0 or 1. 75% of the data were used for training and 25% for testing. The association between hepatocarcinogenesis and tumor bacteria and invasive immune cells was analyzed through the classification results under different input characteristics.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<title>Results and Discussion</title>
<sec id="s3_1">
<title>The Particle Swarm Optimized - Support Vector Regression Algorithm</title>
<p>Support Vector Regression (SVR) is a commonly used technique for non-linear unsupervised learning. The generalization ability and prediction accuracy of SVR depend on the choice of kernel function coefficients and penalty factors. Due to the large feature size, small sample size and unknown sample distribution, in order to avoid over fitting as much as possible, we used a simple and effective linear kernel function in order to avoid overfitting as much as possible. We introduced the particle swarm optimization (PSO) technique, which uses a powerful algorithm to invoke an iterative approach to solve parameter optimization. It can improve the accuracy of SVR results, and increase the robustness of the results through multiple iterations. The overall flowchart is shown in <xref ref-type="fig" rid="f2"><bold>Figure&#xa0;2</bold></xref>.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>The flow chart of PSO-SVR algorithm. SVR model was embedded into the PSO algorithm to calculate the optimal parameters; then SVR model with optimized parameters was applied to the immune cell ratios.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-13-853213-g002.tif"/>
</fig>    <p>We briefly introduce the SVR model for immune cell infiltration estimation. Based on a large amount of tissue gene expression profile data combined with <italic>a priori</italic> knowledge of purified leukocyte subpopulation expression profile data (i.e., a &#x201c;signature matrix&#x201d; representing the expression profile data of each gene in different immune cells), the proportion of immune cells in tumor biopsies can be accurately estimated. The idea was first used in CIBERSORT. The gene expression profile data were first transformed into a linear combination of marker genes for immune cells to solve for f (representing the proportion of immune cells) in a linear combination equation. The linear combination equation was expressed as:</p>
<disp-formula>
<mml:math display="block" id="M15">
<mml:mrow>
<mml:msub>
<mml:mtext>X</mml:mtext>
<mml:mrow>
<mml:mtext>n</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>L</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mtext>S</mml:mtext>
<mml:mrow>
<mml:mtext>n</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>m</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mtext>f</mml:mtext>
<mml:mrow>
<mml:mtext>m</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>L</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where X<sub>n&#xd7;L</sub> denotes the expression profile data of n genes from L samples for deconvolution. f<sub>m&#xd7;L</sub> is the proportion of immune cells to be sought, specifically indicating the proportion of each immune cell in L samples among the m immune cell types. S<sub>n&#xd7;m</sub> is the signature matrix, indicating the expression profile data of n genes in each immune cell for m immune cells. The SVR algorithm solves the linear combinatorial equation using a deconvolution scheme.</p>    <p>The accuracy of the immune cell ratio calculated through the above SVR model is influenced by the model parameters such as the penalty factor C, sensitivity &#x3f5;, and kernel function coefficient &#x3d5;. These parameters are crucial to an SVR model&#x2019;s accuracy and generalizability. However, in practice, these parameters are largely manually chosen without justification. Here we propose to optimize (&#x3f5;, C, &#x3d5;) parameters through iterations using the particle swarm algorithm. In each iteration, the swarming particles update their velocity and position vectors by tracking two &#x201c;optimal values&#x201d; (p<sub>ibest</sub>, g<sub>best</sub>), where p<sub>ibest</sub> denotes the individual optimal value of the particle and g<sub>best</sub> is the global optimal value. In the computation, the expression data of each gene (i.e., each row vector in the <italic>X<sub>n</sub>
</italic><sub>&#xd7;</sub><italic><sub>L</sub>
</italic> matrix) was treated as a particle in the swarming iteration.</p>
<disp-formula>
<mml:math display="block" id="M17">
<mml:mrow>
<mml:msub>
<mml:mtext>v</mml:mtext>
<mml:mrow>
<mml:mtext>i</mml:mtext>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mtext>qv</mml:mtext>
</mml:mrow>
<mml:mtext>i</mml:mtext>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mtext>c</mml:mtext>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:msub>
<mml:mtext>r</mml:mtext>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>p</mml:mtext>
<mml:mrow>
<mml:mtext>ibest</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">x</mml:mi>
<mml:mtext>i</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mtext>c</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:msub>
<mml:mtext>r</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>g</mml:mtext>
<mml:mrow>
<mml:mtext>best</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">x</mml:mi>
<mml:mtext>i</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula>
<mml:math display="block" id="M19">
<mml:mrow>
<mml:msub>
<mml:mtext>x</mml:mtext>
<mml:mrow>
<mml:mtext>i</mml:mtext>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">x</mml:mi>
<mml:mtext>i</mml:mtext>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mtext>v</mml:mtext>
<mml:mrow>
<mml:mtext>i</mml:mtext>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where v<sub>i</sub> and x<sub>i</sub> represent the velocity vector and position vector of the i<sup>th</sup> particle respectively, each gene is regarded as a particle respectively, and n represents the size of the population, specifically the number of genes; q is a non-negative inertia factor. The larger the value of q is, the stronger the global optimization ability is and the weaker the local optimization ability is; c<sub>1</sub> and c<sub>2</sub> are learning factors, general c<sub>1</sub> = c<sub>1</sub> = <bold>2</bold>; r<sub>1</sub> and r<sub>1</sub> both represent random coefficients belonging to [0,1]; p<sub>ibest</sub>  represents the individual optimal value of the ith particle, and g<sub>best</sub>  is the global optimal value. We provide more detailed parameter explanation and algorithm flow in the <xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Materials</bold></xref>. <xref ref-type="table" rid="T1"><bold>Table&#xa0;1</bold></xref> provides an explanation of key parameters.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>PSO-SVR model parameter.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Parameter</th>
<th valign="top" align="center">Meaning</th>
<th valign="top" align="center">Reference range</th>
<th valign="top" align="center">Value</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">k(x,y)</td>
<td valign="top" align="left">kernel function</td>
<td valign="top" align="center">linear</td>
<td valign="top" align="center">k (x,y)=x&#xb7;y</td>
</tr>
<tr>
<td valign="top" align="left">C</td>
<td valign="top" align="left">Penalty factor</td>
<td valign="top" align="center">[1,108]</td>
<td valign="top" align="center">3</td>
</tr>
<tr>
<td valign="top" align="left">&#x3f5;</td>
<td valign="top" align="left">sensitivity</td>
<td valign="top" align="center">[0,0.2]</td>
<td valign="top" align="center">1.203564&#xd7;10<sup>-5</sup>
</td>
</tr>
<tr>
<td valign="top" align="left">&#x3a6;</td>
<td valign="top" align="left">Kernel function coefficient</td>
<td valign="top" align="center">[0.01,2.0]</td>
<td valign="top" align="center">0.04545455</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>We apply this iterative optimization process of particle swarm to solve the SVR model, so that its optimal input parameters are optimized according to the swarm fitness metric. Upon initialization, we specify the approximate ranges of the parameters &#x3f5;, C, and &#x3d5; as: &#x3f5; = (0, 0.2), C = (1, 100), &#x3d5; = (0.01, 2.0), and q<sub>max</sub> and q<sub>min</sub> values are 0.9 and 0.4 for the model. Upon convergence, we find the optimal parameter solution (&#x3f5;, C, &#x3d5;). SVR models using optimal parameters are validated using benchmark data sets. We apply model with these parameters to the expression data of liver cancer patients&#x2019; tissues to solve for immune cell infiltration ratios.</p>
<p>The major innovation of PSO-SVR over previous SVR deconvolution methods is that it applies the machine learning technique &#x2013; particle swarm algorithm, to do SVR iterative optimization. In the process, the swarming algorithm searches for SVR hyperplane formed by support vectors that capture as many data points as possible while satisfying the given constraints and avoiding overfitting, using a linear &#x201c;&#x3c9;-insensitive&#x201d; loss function. The function penalizes only those data points outside a specific error radius. The support vectors were selected from the signature matrix of genes, and the standard reference expression profile (signature matrix) was selected from a composition of 22 immune cells including CD4 T, and CD8 T immune cells in CIBERSORTX, but other signature gene sets can be applied as well.</p>
</sec>
<sec id="s3_2">
<title>Benchmark of the PSO-SVR Algorithm</title>
<p>We gathered three data sets (see <xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Table&#xa0;1</bold></xref>), in which the immune cell proportions were determined through orthogonal flow cytometry (PBMC-FC), immunohistochemistry (CRC-IC), and single-cell RNA sequencing technologies (Melanoma-scRNA). Flow cytometry was an established technology to identify and determine different cell types in heterogeneous cell populations (<xref ref-type="bibr" rid="B36">36</xref>). immunohistochemistry is also well-established, using chemical reaction to label antibodies and to identify and quantify antigens within tissue cells. The genetic heterogeneity of cells of the same tissue can be also analyzed by scRNA-Seq at the level of individual cells to cluster and compute the composition immune cells (<xref ref-type="bibr" rid="B37">37</xref>). In order to verify the accuracy of PSO-SVR algorithm, we used data sets based on these technologies as the orthogonal control, and studied the deviation between PSO-SVR results and the control.</p>
<p>As shown in <xref ref-type="fig" rid="f3"><bold>Figure&#xa0;3A</bold></xref>, for the PBMC-FC data, that the PSO-SVR estimated B cell, CD4 T cells, CD8 T cells and Monocytes levels were very close to the flow cytometry results. Overall, the four types of immune cells present a good correlation. The CD8 T cells showed relatively less consistency as compared to the other two types, maybe because CD8 T cells accounted for a relatively small number of T lymphocytes in PBMC (5%&#x2013;20%) thus is subject to less accurate estimation. As shown in <xref ref-type="fig" rid="f3"><bold>Figure&#xa0;3B</bold></xref>, the PSO-SVR estimates of B&#xa0;cells, monocytes, NK cells, and T cells were also in good concordance with the immunohistochemistry results. As shown in <xref ref-type="fig" rid="f3"><bold>Figure&#xa0;3C</bold></xref>, PSO-SVR estimated immune cell fractions, such as CD4 T and CD8 T cells, also showed a good correlation with the scRNA-seq levels.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Comparison between PSO-SVR and orthogonal technologies. Scatter plots of immune cell fractions with regression of <bold>(A)</bold> PBMC-FC; <bold>(B)</bold> CRC-IC; and <bold>(C)</bold> Melanoma-scRNA data; r for Pearson&#x2019;s correlation, rs for Spearman&#x2019;s correlation.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-13-853213-g003.tif"/>
</fig>
<p>To further evaluate the PSO-SVR estimates using the PBMC-FC, CRC-IC and Melanoma-scRNA data sets, we used three metrics, namely root mean square error (RMSE), Pearson correlation coefficient, and Spearman&#x2019;s correlation coefficient. The individual indicators are shown in <xref ref-type="table" rid="T2"><bold>Table&#xa0;2</bold></xref>.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Immune cell error table of three data sets.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left"/>
<th valign="top" align="center">Cell type</th>
<th valign="top" align="center">Root mean square error (RMSE)</th>
<th valign="top" align="center">Pearson (r)</th>
<th valign="top" align="center">Spearman (rs)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" rowspan="4" align="left"><bold>PBMC-FC</bold>
</td>
<td valign="top" align="left">B cells</td>
<td valign="top" align="center">0.028763</td>
<td valign="top" align="center">0.69534</td>
<td valign="top" align="center">0.67669</td>
</tr>
<tr>
<td valign="top" align="left">Monocytes</td>
<td valign="top" align="center">0.056215</td>
<td valign="top" align="center">0.92192</td>
<td valign="top" align="center">0.93684</td>
</tr>
<tr>
<td valign="top" align="left">CD4 T cells</td>
<td valign="top" align="center">0.073478</td>
<td valign="top" align="center">0.84827</td>
<td valign="top" align="center">0.88270</td>
</tr>
<tr>
<td valign="top" align="left">CD8T cells</td>
<td valign="top" align="center">0.110915</td>
<td valign="top" align="center">0.84207</td>
<td valign="top" align="center">0.82105</td>
</tr>
<tr>
<td valign="top" rowspan="4" align="left"><bold>CRC-IC</bold>
</td>
<td valign="top" align="left">B cells</td>
<td valign="top" align="center">0.057609</td>
<td valign="top" align="center">0.39181</td>
<td valign="top" align="center">0.51515</td>
</tr>
<tr>
<td valign="top" align="left">Monocytes</td>
<td valign="top" align="center">0.018876</td>
<td valign="top" align="center">0.87587</td>
<td valign="top" align="center">0.78181</td>
</tr>
<tr>
<td valign="top" align="left">NK cells</td>
<td valign="top" align="center">0.116719</td>
<td valign="top" align="center">0.23098</td>
<td valign="top" align="center">0.30909</td>
</tr>
<tr>
<td valign="top" align="left">T cells</td>
<td valign="top" align="center">0.060646</td>
<td valign="top" align="center">0.18626</td>
<td valign="top" align="center">0.16363</td>
</tr>
<tr>
<td valign="top" rowspan="5" align="left"><bold>Melanoma-scRNA</bold>
</td>
<td valign="top" align="left">B cells</td>
<td valign="top" align="center">0.019590</td>
<td valign="top" align="center">0.99128</td>
<td valign="top" align="center">0.85381</td>
</tr>
<tr>
<td valign="top" align="left">Macrophages</td>
<td valign="top" align="center">0.028086</td>
<td valign="top" align="center">0.70316</td>
<td valign="top" align="center">0.67192</td>
</tr>
<tr>
<td valign="top" align="left">NK cells</td>
<td valign="top" align="center">0.021296</td>
<td valign="top" align="center">0.92560</td>
<td valign="top" align="center">0.86315</td>
</tr>
<tr>
<td valign="top" align="left">CD4 T cells</td>
<td valign="top" align="center">0.034945</td>
<td valign="top" align="center">0.95508</td>
<td valign="top" align="center">0.92847</td>
</tr>
<tr>
<td valign="top" align="left">CD8 T cells</td>
<td valign="top" align="center">0.050518</td>
<td valign="top" align="center">0.97476</td>
<td valign="top" align="center">0.97543</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>We observe in <xref ref-type="table" rid="T2"><bold>Table&#xa0;2</bold></xref> that the error was generally larger for B cells and that both Pearson and Spearman&#x2019;s correlation coefficients were relatively low. For CD4 T and CD8 T cells, the difference between the results presented in the PBMC-FC and Melanoma-scRNA is very small, and the RMSE and correlation coefficients for both are close, proving the reliability of the algorithm in this study.</p>
<p>Finally, the results calculated through PSO-SVR were compared with other algorithms such as CIBERSORTX, EPIC, MCP-counter, for calculating the proportion of immune cells in infiltrating tumors. From the results of PBMC-FC and Melanoma-scRNA (refer to <xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Tables&#xa0;2</bold> and <bold>5</bold></xref>), it can be observed that PSO-SVR outperforms CIBERSORTX, EPIC, MCP-counter, in terms of the three indicators, RMSE, Pearson, and Spearman correlation coefficients. However, the results of CRC-IC (refer to <xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Table&#xa0;3</bold></xref>) are not as good as the other three methods, which could be attributed to the small amount of CRC-IC data. Overall, our algorithm was highly accurate.</p>
</sec>
<sec id="s3_3">
<title>Joint Microbial- and Immune-Effect Analysis of Hepatocellular Carcinoma Samples</title>
<sec id="s3_3_1">
<title>Differential Presence of Infiltrating Immune Cells Between Tumor and Normal Tissues</title>
<p>We analyzed the differences in the relative proportions of immune cells in tumor samples and normal solid tissue samples. The non-parametric Mann-Whitney-Wilcoxon test was conducted in R software and then corrected for p-values with the Benjamini-Hochberg correction (FDR). Immune cells with significant differences were identified as shown in <xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Table&#xa0;5</bold></xref> (FDR &lt; 0.05).</p>
<p>As it can be seen in <xref ref-type="fig" rid="f4"><bold>Figure&#xa0;4</bold></xref> and <xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Table&#xa0;6</bold></xref>, the relative proportions of regulatory T cells [confidence level=95%, FDR= 1.58E-07] and 59.87% significantly higher in tumor tissues, and the relative proportions of Monocytes and Neutrophils are 150.07% and 363.69% significantly lower [FDR= 6.77E-07, 1.37E-05], as compared to normal solid tissue samples. These results suggested that regulatory T cells, Monocytes and Neutrophils were attracted toward tumor tissues who may act as host defense against invasive tumor growth. These findings were consistent with external knowledge and evidence.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Differetial analysis of immune cell infiltration in normal and tumor tissues. &#x201c;**&#x201d; indicates FDR &lt; 0.01; &#x201c;***&#x201d; indicates FDR &lt; 0.001.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-13-853213-g004.tif"/>
</fig>
<p>As known, neutrophils provide the first line of defense for the innate immune system by phagocytosing, killing, and digesting bacteria and fungi (<xref ref-type="bibr" rid="B38">38</xref>). CD8 T cells are an important component of the immune system, and inducing an effective memory T cell response is a major target for vaccines against chronic infections and tumors (<xref ref-type="bibr" rid="B39">39</xref>). Studies have shown that the presence of Tregs in tumors of patients with hepatocellular carcinoma are suppressor cells and their increased levels are associated with immunosuppression and evasion in patients with cancer, where the inappropriate immune responses can be prevented by suppressing immune effector cells. In addition, the frequency of Tregs in lymphoid tissue, peripheral blood, and in the TME is greater than that in normal tissue (<xref ref-type="bibr" rid="B40">40</xref>).</p>
</sec>
<sec id="s3_3_2">
<title>Differential Presence of Infiltrating Bacteria Between Tumor and Normal Tissues</title>
<p>In <xref ref-type="table" rid="T3"><bold>Table&#xa0;3</bold></xref> and <xref ref-type="fig" rid="f5"><bold>Figure&#xa0;5A</bold></xref>, we showed the most significant differentially present bacteria between tumor and normal tissues. These include <italic>Caldatribacterium</italic> sp. [&#x2191;74.94%, FDR=3.15&#xd7;10<sup>-11</sup>], <italic>Acidaminococcaceae</italic> sp. [&#x2191;73.58%, FDR=1.79&#xd7;10<sup>-9</sup>], among others. We used the nonparametric Mann-Whitney U test. The p-values were corrected for multiple testing through FDR.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Bacteria with significant variability between normal and primary tumor tissues.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Species</th>
<th valign="top" align="center">FDR</th>
<th valign="top" align="center">Difference %</th>
<th valign="top" align="center">State (&#x2191;&#x2193;)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><italic>Caldatribacterium</italic> sp.</td>
<td valign="top" align="center">3.15 &#xd7; 10<sup>-11</sup>
</td>
<td valign="top" align="center">74.94%</td>
<td valign="top" align="center">&#x2191;</td>
</tr>
<tr>
<td valign="top" align="left"><italic>Acidaminococcaceae</italic> sp.</td>
<td valign="top" align="center">1.79 &#xd7; 10<sup>-9</sup>
</td>
<td valign="top" align="center">73.58%</td>
<td valign="top" align="center">&#x2191;</td>
</tr>
<tr>
<td valign="top" align="left"><italic>Planctopirus</italic> sp.</td>
<td valign="top" align="center">2.50 &#xd7; 10<sup>-9</sup>
</td>
<td valign="top" align="center">73.66%</td>
<td valign="top" align="center">&#x2191;</td>
</tr>
<tr>
<td valign="top" align="left"><italic>Desulfobulbaceae</italic> sp.</td>
<td valign="top" align="center">2.50 &#xd7; 10<sup>-9</sup>
</td>
<td valign="top" align="center">71.10%</td>
<td valign="top" align="center">&#x2191;</td>
</tr>
<tr>
<td valign="top" align="left"><italic>Nocardia farcinica</italic>
</td>
<td valign="top" align="center">8.02 &#xd7; 10<sup>-9</sup>
</td>
<td valign="top" align="center">71.20%</td>
<td valign="top" align="center">&#x2191;</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Difference and diversity of bacteria in liver cancer samples. <bold>(A)</bold> Differentiation of bacteria between liver tumor tissue and normal tissue. &#x201c;**&#x201d; indicates FDR &lt; 0.05; &#x201c;***&#x201d; indicates FDR &lt; 0.001. <bold>(B)</bold> Analysis of alpha diversity of bacteria in tumor and normal tissues. The green color on the left indicates normal samples, the red part on the right indicates tumor samples, and the middle BASE part indicates the interquartile range.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-13-853213-g005.tif"/>
</fig>
<p>Noticeably, <italic>Caldatribacterium</italic> sp., <italic>Acidaminococcaceae</italic> sp. (amino acid cocci), and <italic>Planctopirus</italic> sp. are significantly more abundant in infiltrating tumor tissues as compared to normal tissues and are likely associated with the pathological development of hepatocellular carcinoma.</p>
<p>These results were substantiated by external molecular biology and clinical evidence from previous studies. For examples it was found that the increase in <italic>Caldatribacterium</italic> sp. leads to DNA damage in hepatic stellate cells, which then promotes the development of hepatocellular carcinoma through metabolites or toxins from intestinal bacteria (<xref ref-type="bibr" rid="B41">41</xref>). Studies also showed the presence of <italic>Planctopirus</italic> sp. bacterial infection in serum and tissues derived from patients with chronic liver disease &#x2013; a common presage of hepatocellular carcinoma; the enrichment of <italic>Planctopirus</italic> sp. bacteria in the serum of patients with chronic liver disease is significantly higher than that of healthy individuals, suggesting that it may be associated with the development of chronic hepatitis B and primary liver cancer (<xref ref-type="bibr" rid="B42">42</xref>). The other bacteria identified in our study were also supported in the literature, where multiple species including <italic>Gemmiger formicilis</italic>, <italic>Marithrix</italic> sp., and <italic>Haemophilus</italic> sp. were found to be over-represented in in gut microbiome in patients with HCC (<xref ref-type="bibr" rid="B43">43</xref>).</p>
<p>We also found that the microbial diversity (Shannon Index) in the primary tumor tissue is significantly higher than that in normal tissue samples (see <xref ref-type="fig" rid="f5"><bold>Figure&#xa0;5B</bold></xref>). This is somewhat contrary to the discoveries of higher gut microbiome diversity associated with healthy controls as known in many other related cancers (<xref ref-type="bibr" rid="B44">44</xref>).</p>
</sec>
</sec>
<sec id="s3_4">
<title>Joint Effects of Infiltrating Immune Cells and Bacteria On Hepatocellular Carcinoma</title>
<p>In order to further analyze the impact of tumor infiltrating immune cells and bacteria on the occurrence and development of liver cancer, we used the downloaded gene expression data and transcriptome data of 98 liver cancer samples, and calculated the corresponding bacterial abundance data by using the estimation method of microbial relative abundance in this paper. The PSO-SVR algorithm described in Section 3.1 was used to calculate the relative proportion of immune cells in liver cancer samples.</p>
<p>Then the logistic regression classification method was applied to three cases: case 1, only bacterial abundance data were used as the input feature; In case 2, only the immune cell proportion data were used as the input feature; and in case 3, both bacterial abundance and immune cell ratio were used as input characteristics. From the classification results of the three cases (see <xref ref-type="table" rid="T4"><bold>Table&#xa0;4</bold></xref>), the classification accuracy of using only a single input feature is lower - only bacteria: 0.70, only immune cell feature: 0.74, and both features: 0.84. No matter which data feature is added, the classification accuracy can be improved. After adding the immune cell proportion data, the classification accuracy is improved by 20% compared with case 1; After adding bacterial abundance data, the accuracy of classifier is improved by 13.51% compared with case 2. This shows that the bacteria and infiltrating immune cells in liver cancer tissue contain key characteristics for cancer diagnosis, which can be used as a reference index for clinical cancer diagnosis.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Classification effect of liver cancer samples under different input features.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left"/>
<th valign="top" align="center">Case1: Bacteria</th>
<th valign="top" align="center">Case2: Immune cell</th>
<th valign="top" align="center">Case3: Bacteria-Cell</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" rowspan="3" align="left"><bold>CIBERSORT</bold>
</td>
<td valign="top" align="center">0.68</td>
<td valign="top" align="center">0.64</td>
<td valign="top" align="center">0.80</td>
</tr>
<tr>
<td valign="top" align="center">0.75</td>
<td valign="top" align="center">0.71</td>
<td valign="top" align="center">0.88</td>
</tr>
<tr>
<td valign="top" align="center">0.67</td>
<td valign="top" align="center">0.67</td>
<td valign="top" align="center">0.88</td>
</tr>
<tr>
<td valign="top" align="left"><bold>Average accuracy</bold>
</td>
<td valign="top" align="center">0.70</td>
<td valign="top" align="center">0.67</td>
<td valign="top" align="center">0.85</td>
</tr>
<tr>
<td valign="top" rowspan="3" align="left"><bold>PSO-SVR</bold>
</td>
<td valign="top" align="center">0.68</td>
<td valign="top" align="center">0.76</td>
<td valign="top" align="center">0.80</td>
</tr>
<tr>
<td valign="top" align="center">0.75</td>
<td valign="top" align="center">0.79</td>
<td valign="top" align="center">0.75</td>
</tr>
<tr>
<td valign="top" align="center">0.67</td>
<td valign="top" align="center">0.67</td>
<td valign="top" align="center">0.96</td>
</tr>
<tr>
<td valign="top" align="left"><bold>Average accuracy</bold>
</td>
<td valign="top" align="center">0.70</td>
<td valign="top" align="center">0.74</td>
<td valign="top" align="center">0.84</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s4" sec-type="conclusions">
<title>Conclusions</title>
<p>We performed a joint analysis of microbial and immune cell abundance in liver cancer tissue using a gene expression profile deconvolution algorithm combined with foreign read remapping. First, we filtered out reads from the RNA-Seq data that did not map to the human reference genome. Second, we assembled the single-ended unmapped reads with a reverse complementary sequence to the double-ended unmapped reads, and the processed reads were then mapped to the microbial reference library. Finally, the GRAMMy algorithm was introduced to calculate the relative abundance of microorganisms. This algorithm overcomes the problem of mapping one read to different microbial reference sequences owing to small reads and enables a more accurate calculation of relative abundance.</p>
<p>This complete procedure was then applied to RNA-seq samples of 98 patients with hepatocellular carcinoma, and we found that microorganisms including <italic>Caldatribacterium</italic> sp. and <italic>Planctopirus</italic> sp. differed significantly between normal and tumor tissues, which was also reported in the literature.</p>
<p>To study the degree of immune cell response to microorganisms and the effect on liver cancer in the human microenvironment, we introduced SVR and particle swarm algorithm to estimate the relative proportion of infiltrating immune cells based on the deconvolution model. The results of our study were validated by actual data and then compared with some immune cell counting algorithms, such as CIBESORTX, EPIC, and MCP-counter, and were found to be relatively more reliable.</p>
<p>The PSO-SVR algorithm was applied to the gene expression profile data of liver cancer samples. The differential analysis revealed significant differences in regulatory T cells, monocytes, and neutrophils between normal and tumor tissues. Finally, using the classification regression algorithm within machine learning, we found that adding microbial characteristics can improve the accuracy of liver cancer prediction.</p>
<p>This study has some limitations that result from the limited sample size. For example, a small number of samples affects statistical power, which we anticipate correcting in further studies. Also,  liver cancer has different cancer subtypes, which are not subdivided in this study; The model does not take gene variation into account; otherwise, there can be new information to improve the accuracy.</p>
</sec>
<sec id="s5" sec-type="data-availability">
<title>Data Availability Statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <uri xlink:href="https://portal.gdc.cancer.gov/repository">https://portal.gdc.cancer.gov/repository</uri> and <uri xlink:href="http://www.cancergenomicscloud.org/">http://www.cancergenomicscloud.org/</uri>.</p>
</sec>
<sec id="s6" sec-type="author-contributions">
<title>Author Contributions</title>
<p>DA, LCX and YX conceived and designed the study. YX and GL performed the analyses and summarized the data. LCX and DA supervised the study. DA, YX and LCX wrote the manuscript with inputs from GL, XL, and YW. All authors have read and agreed to the published version of the manuscript.</p>
</sec>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>This work was supported by grants from the National Natural Science Foundation of China (61873027), open project of the National Engineering Laboratory for Agri-product Quality Traceability (No.AQT-2020-YB6), and GuangDong Basic and Applied Basic Research Foundation  (2022A1515-011426).</p>
</sec>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s9" sec-type="disclaimer">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<sec id="s10" sec-type="supplementary-material">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fimmu.2022.853213/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fimmu.2022.853213/full#supplementary-material</ext-link>.</p>
<supplementary-material xlink:href="Table_1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>O'Hara</surname> <given-names>AM</given-names>
</name>
<name>
<surname>Shanahan</surname> <given-names>F</given-names>
</name>
</person-group>. <article-title>The Gut Flora as a Forgotten Organ</article-title>. <source>EMBO Rep</source> (<year>2006</year>) <volume>7</volume>(<issue>7</issue>):<page-range>688&#x2013;93</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/sj.embor.7400731</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Elinav</surname> <given-names>E</given-names>
</name>
<name>
<surname>Nowarski</surname> <given-names>R</given-names>
</name>
<name>
<surname>Thaiss</surname> <given-names>CA</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>B</given-names>
</name>
<name>
<surname>Jin</surname> <given-names>C</given-names>
</name>
<name>
<surname>Flavell</surname> <given-names>RA</given-names>
</name>
</person-group>. <article-title>Inflammation-Induced Cancer: Crosstalk Between Tumours, Immune Cells and Microorganisms</article-title>. <source>Nat Rev Cancer</source> (<year>2013</year>) <volume>13</volume>(<issue>11</issue>):<page-range>759&#x2013;71</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nrc3611</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arleevskaya</surname> <given-names>MI</given-names>
</name>
<name>
<surname>Aminov</surname> <given-names>R</given-names>
</name>
<name>
<surname>Brooks</surname> <given-names>WH</given-names>
</name>
<name>
<surname>Manukyan</surname> <given-names>G</given-names>
</name>
<name>
<surname>Renaudineau</surname> <given-names>Y</given-names>
</name>
</person-group>. <article-title>Editorial: Shaping of Human Immune System and Metabolic Processes by Viruses and Microorganisms</article-title>. <source>Front Microbiol</source> (<year>2019</year>) <volume>10</volume>:<elocation-id>816</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fmicb.2019.00816</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>An</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>W</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>T</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>B</given-names>
</name>
<name>
<surname>Cao</surname> <given-names>H</given-names>
</name>
</person-group>. <article-title>The Intratumoural Microbiota in Cancer: New Insights From Inside</article-title>. <source>Biochim Biophys Acta Rev Cancer</source> (<year>2021</year>) <volume>1876</volume>(<issue>2</issue>):<elocation-id>188626</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.bbcan.2021.188626</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mizutani</surname> <given-names>T</given-names>
</name>
<name>
<surname>Mitsuoka</surname> <given-names>T</given-names>
</name>
</person-group>. <article-title>Inhibitory Effect of Some Intestinal Bacteria on Liver Tumorigenesis in Gnotobiotic C3H/He Male Mice</article-title>. <source>Cancer Lett</source> (<year>1980</year>) <volume>11</volume>(<issue>2</issue>):<fpage>89</fpage>&#x2013;<lpage>95</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/0304-3835(80)90098-1</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moffatt</surname> <given-names>CE</given-names>
</name>
<name>
<surname>Lamont</surname> <given-names>RJ</given-names>
</name>
</person-group>. <article-title>Porphyromonas Gingivalis Induction of MicroRNA-203 Expression Controls Suppressor of Cytokine Signaling 3 in Gingival Epithelial Cells</article-title>. <source>Infect Immun</source> (<year>2011</year>) <volume>79</volume>(<issue>7</issue>):<page-range>2632&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1128/Iai.00082-11</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Garner</surname> <given-names>RC</given-names>
</name>
<name>
<surname>Wright</surname> <given-names>CM</given-names>
</name>
</person-group>. <article-title>Induction of Mutations in DNA-Repair Deficient Bacteria by a Liver Microsomal Metabolite of Aflatoxin B1</article-title>. <source>Br J Cancer</source> (<year>1973</year>) <volume>28</volume>(<issue>6</issue>):<page-range>544&#x2013;51</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/bjc.1973.184</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mangul</surname> <given-names>S</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>HT</given-names>
</name>
<name>
<surname>Strauli</surname> <given-names>N</given-names>
</name>
<name>
<surname>Gruhl</surname> <given-names>F</given-names>
</name>
<name>
<surname>Porath</surname> <given-names>HT</given-names>
</name>
<name>
<surname>Hsieh</surname> <given-names>K</given-names>
</name>
<etal/>
</person-group>. <article-title>ROP: Dumpster Diving in RNA-Sequencing to Find the Source of 1 Trillion Reads Across Diverse Adult Human Tissues</article-title>. <source>Genome Biol</source> (<year>2018</year>) <volume>19</volume>(<issue>1</issue>):<fpage>36</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13059-018-1403-7</pub-id>
</citation>
</ref>
<ref id="B9">
<label>9</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zheng-Bradley</surname> <given-names>X</given-names>
</name>
<name>
<surname>Streeter</surname> <given-names>I</given-names>
</name>
<name>
<surname>Fairley</surname> <given-names>S</given-names>
</name>
<name>
<surname>Richardson</surname> <given-names>D</given-names>
</name>
<name>
<surname>Clarke</surname> <given-names>L</given-names>
</name>
<name>
<surname>Flicek</surname> <given-names>P</given-names>
</name>
<etal/>
</person-group>. <article-title>Alignment of 1000 Genomes Project Reads to Reference Assembly Grch38</article-title>. <source>Gigascience</source> (<year>2017</year>) <volume>6</volume>(<issue>7</issue>):<fpage>1</fpage>&#x2013;<lpage>8</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/gigascience/gix038</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xia</surname> <given-names>LC</given-names>
</name>
<name>
<surname>Cram</surname> <given-names>JA</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>T</given-names>
</name>
<name>
<surname>Fuhrman</surname> <given-names>JA</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>FZ</given-names>
</name>
</person-group>. <article-title>Accurate Genome Relative Abundance Estimation Based on Shotgun Metagenomic Reads</article-title>. <source>PloS One</source> (<year>2011</year>) <volume>6</volume>(<issue>12</issue>):<fpage>e27992</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0027992</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karlmark</surname> <given-names>KR</given-names>
</name>
<name>
<surname>Wasmuth</surname> <given-names>HE</given-names>
</name>
<name>
<surname>Trautwein</surname> <given-names>C</given-names>
</name>
<name>
<surname>Tacke</surname> <given-names>F</given-names>
</name>
</person-group>. <article-title>Chemokine-Directed Immune Cell Infiltration in Acute and Chronic Liver Disease</article-title>. <source>Expert Rev Gastroenterol Hepatol</source> (<year>2008</year>) <volume>2</volume>(<issue>2</issue>):<page-range>233&#x2013;42</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1586/17474124.2.2.233</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hegde</surname> <given-names>P</given-names>
</name>
<name>
<surname>Weiss</surname> <given-names>E</given-names>
</name>
<name>
<surname>Paradis</surname> <given-names>V</given-names>
</name>
<name>
<surname>Wan</surname> <given-names>J</given-names>
</name>
<name>
<surname>Mabire</surname> <given-names>M</given-names>
</name>
<name>
<surname>Sukriti</surname> <given-names>S</given-names>
</name>
<etal/>
</person-group>. <article-title>Mucosal-Associated Invariant T Cells are a Profibrogenic Immune Cell Population in the Liver</article-title>. <source>Nat Commun</source> (<year>2018</year>) <volume>9</volume>(<issue>1</issue>):<fpage>2146</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41467-018-04450-y</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xue</surname> <given-names>F</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Han</surname> <given-names>L</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>N</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>T</given-names>
</name>
<etal/>
</person-group>. <article-title>Immune Cell Functional Assay in Monitoring of Adult Liver Transplantation Recipients With Infection</article-title>. <source>Transplantation</source> (<year>2010</year>) <volume>89</volume>(<issue>5</issue>):<page-range>620&#x2013;6</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1097/tp.0b013e3181c690fa</pub-id>
</citation>
</ref>
<ref id="B14">
<label>14</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname> <given-names>T</given-names>
</name>
<name>
<surname>Xue</surname> <given-names>F</given-names>
</name>
<name>
<surname>Han</surname> <given-names>LZ</given-names>
</name>
<name>
<surname>Xi</surname> <given-names>ZF</given-names>
</name>
<name>
<surname>Li</surname> <given-names>QG</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>N</given-names>
</name>
<etal/>
</person-group>. <article-title>Invasive Fungal Infection After Liver Transplantation: Risk Factors and Significance of Immune Cell Function Monitoring</article-title>. <source>J Dig Dis</source> (<year>2011</year>) <volume>12</volume>(<issue>6</issue>):<page-range>467&#x2013;75</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/j.1751-2980.2011.00542.x</pub-id>
</citation>
</ref>
<ref id="B15">
<label>15</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Newman</surname> <given-names>AM</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>CL</given-names>
</name>
<name>
<surname>Green</surname> <given-names>MR</given-names>
</name>
<name>
<surname>Gentles</surname> <given-names>AJ</given-names>
</name>
<name>
<surname>Feng</surname> <given-names>WG</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>Y</given-names>
</name>
<etal/>
</person-group>. <article-title>Robust Enumeration of Cell Subsets From Tissue Expression Profiles</article-title>. <source>Nat Methods</source> (<year>2015</year>) <volume>12</volume>(<issue>5</issue>):<page-range>453&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/Nmeth.3337</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>B</given-names>
</name>
<name>
<surname>Severson</surname> <given-names>E</given-names>
</name>
<name>
<surname>Pignon</surname> <given-names>JC</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>HQ</given-names>
</name>
<name>
<surname>Li</surname> <given-names>TW</given-names>
</name>
<name>
<surname>Novak</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>Comprehensive Analyses of Tumor Immunity: Implications for Cancer Immunotherapy</article-title>. <source>Genome Biol</source> (<year>2016</year>) <volume>17</volume>(<issue>1</issue>):<fpage>174</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13059-016-1028-7</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Becht</surname> <given-names>E</given-names>
</name>
<name>
<surname>Giraldo</surname> <given-names>NA</given-names>
</name>
<name>
<surname>Lacroix</surname> <given-names>L</given-names>
</name>
<name>
<surname>Buttard</surname> <given-names>B</given-names>
</name>
<name>
<surname>Elarouci</surname> <given-names>N</given-names>
</name>
<name>
<surname>Petitprez</surname> <given-names>F</given-names>
</name>
<etal/>
</person-group>. <article-title>Estimating the Population Abundance of Tissue-Infiltrating Immune and Stromal Cell Populations Using Gene Expression</article-title>. <source>Genome Biol</source> (<year>2016</year>) <volume>17</volume>(<issue>1</issue>):<fpage>218</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13059-016-1070-5</pub-id>
</citation>
</ref>
<ref id="B18">
<label>18</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Racle</surname> <given-names>J</given-names>
</name>
<name>
<surname>de Jonge</surname> <given-names>K</given-names>
</name>
<name>
<surname>Baumgaertner</surname> <given-names>P</given-names>
</name>
<name>
<surname>Speiser</surname> <given-names>DE</given-names>
</name>
<name>
<surname>Gfeller</surname> <given-names>D</given-names>
</name>
</person-group>. <article-title>Simultaneous Enumeration of Cancer and Immune Cell Types From Bulk Tumor Gene Expression Data</article-title>. <source>Elife</source> (<year>2017</year>) <volume>6</volume>:<fpage>e26476</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.7554/eLife.26476</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Deng</surname> <given-names>M</given-names>
</name>
<name>
<surname>Matsumoto</surname> <given-names>N</given-names>
</name>
<name>
<surname>Kitayama</surname> <given-names>M</given-names>
</name>
<name>
<surname>Inoue</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Online Estimation of Arriving Time for Robot to Soccer Ball in RoboCup Soccer Using PSO-SVR</article-title>. In: <source>Proceedings of the 2014 International Conference on Advanced Mechatronic Systems</source>. (<publisher-name>Institute of Electrical &amp; Electronics Engineers</publisher-name>) (<year>2014</year>). p. <page-range>543&#x2013;6</page-range>. doi: <pub-id pub-id-type="doi">10.1109/ICAMechS.2014.6911605</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fujita</surname> <given-names>K</given-names>
</name>
<name>
<surname>Deng</surname> <given-names>M</given-names>
</name>
<name>
<surname>Wakimoto</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>A Miniature Pneumatic Bending Rubber Actuator Controlled by Using the PSO-SVR-Based Motion Estimation Method With the Generalized Gaussian Kernel</article-title>. <source>Actuators</source> (<year>2017</year>) <volume>6</volume>:<elocation-id>6</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/act6010006</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mohammadi</surname> <given-names>S</given-names>
</name>
<name>
<surname>Zuckerman</surname> <given-names>N</given-names>
</name>
<name>
<surname>Goldsmith</surname> <given-names>A</given-names>
</name>
<name>
<surname>Grama</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>A Critical Survey of Deconvolution Methods for Separating Cell Types in Complex Tissues</article-title>. <source>Proc IEEE</source> (<year>2017</year>) <volume>105</volume>(<issue>2</issue>):<page-range>340&#x2013;66</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/Jproc.2016.2607121</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Newman</surname> <given-names>AM</given-names>
</name>
<name>
<surname>Steen</surname> <given-names>CB</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>CL</given-names>
</name>
<name>
<surname>Gentles</surname> <given-names>AJ</given-names>
</name>
<name>
<surname>Chaudhuri</surname> <given-names>AA</given-names>
</name>
<name>
<surname>Scherer</surname> <given-names>F</given-names>
</name>
<etal/>
</person-group>. <article-title>Determining Cell Type Abundance and Expression From Bulk Tissues With Digital Cytometry</article-title>. <source>Nat Biotechnol</source> (<year>2019</year>) <volume>37</volume>(<issue>7</issue>):<page-range>773&#x2013;82</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41587-019-0114-2</pub-id>
</citation>
</ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gibney</surname> <given-names>GT</given-names>
</name>
<name>
<surname>Weiner</surname> <given-names>LM</given-names>
</name>
<name>
<surname>Atkins</surname> <given-names>MB</given-names>
</name>
</person-group>. <article-title>Predictive Biomarkers for Checkpoint Inhibitor-Based Immunotherapy</article-title>. <source>Lancet Oncol</source> (<year>2016</year>) <volume>17</volume>(<issue>12</issue>):<page-range>E542&#x2013;51</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S1470-2045(16)30406-5</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hogan</surname> <given-names>SA</given-names>
</name>
<name>
<surname>Levesque</surname> <given-names>MP</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>PF</given-names>
</name>
</person-group>. <article-title>Melanoma Immunotherapy: Next-Generation Biomarkers</article-title>. <source>Front Oncol</source> (<year>2018</year>) <volume>8</volume>:<elocation-id>178</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fonc.2018.00178</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gopalakrishnan</surname> <given-names>V</given-names>
</name>
<name>
<surname>Spencer</surname> <given-names>CN</given-names>
</name>
<name>
<surname>Nezi</surname> <given-names>L</given-names>
</name>
<name>
<surname>Reuben</surname> <given-names>A</given-names>
</name>
<name>
<surname>Andrews</surname> <given-names>MC</given-names>
</name>
<name>
<surname>Karpinets</surname> <given-names>TV</given-names>
</name>
<etal/>
</person-group>. <article-title>Gut Microbiome Modulates Response to Anti-PD-1 Immunotherapy in Melanoma Patients</article-title>. <source>Science</source> (<year>2018</year>) <volume>359</volume>(<issue>6371</issue>):<fpage>97</fpage>&#x2013;<lpage>103</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1126/science.aan4236</pub-id>
</citation>
</ref>
<ref id="B26">
<label>26</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Matson</surname> <given-names>V</given-names>
</name>
<name>
<surname>Fessler</surname> <given-names>J</given-names>
</name>
<name>
<surname>Bao</surname> <given-names>R</given-names>
</name>
<name>
<surname>Chongsuwat</surname> <given-names>T</given-names>
</name>
<name>
<surname>Zha</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Alegre</surname> <given-names>ML</given-names>
</name>
<etal/>
</person-group>. <article-title>The Commensal Microbiome Is Associated With Anti-PD-1 Efficacy in Metastatic Melanoma Patients</article-title>. <source>Science</source> (<year>2018</year>) <volume>359</volume>(<issue>6371</issue>):<page-range>104&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1126/science.aao3290</pub-id>
</citation>
</ref>
<ref id="B27">
<label>27</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>H</given-names>
</name>
<name>
<surname>Handsaker</surname> <given-names>B</given-names>
</name>
<name>
<surname>Wysoker</surname> <given-names>A</given-names>
</name>
<name>
<surname>Fennell</surname> <given-names>T</given-names>
</name>
<name>
<surname>Ruan</surname> <given-names>J</given-names>
</name>
<name>
<surname>Homer</surname> <given-names>N</given-names>
</name>
<etal/>
</person-group>. <article-title>The Sequence Alignment/Map Format and SAMtools</article-title>. <source>Bioinformatics</source> (<year>2009</year>) <volume>25</volume>(<issue>16</issue>):<page-range>2078&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btp352</pub-id>
</citation>
</ref>
<ref id="B28">
<label>28</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peter</surname> <given-names>JA</given-names>
</name>
<name>
<surname>Cock</surname> <given-names>CJF</given-names>
</name>
<name>
<surname>Naohisa</surname> <given-names>G</given-names>
</name>
<name>
<surname>Heuer</surname> <given-names>ML</given-names>
</name>
<name>
<surname>Rice</surname> <given-names>PM</given-names>
</name>
</person-group>. <article-title>The Sanger FASTQ File Format for Sequences With Quality Scores, and the Solexa/Illumina FASTQ Variants</article-title>. <source>Nucleic Acids Res</source> (<year>2010</year>) <volume>38</volume>(<issue>6</issue>):<page-range>1767&#x2013;71</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkp1137</pub-id>
</citation>
</ref>
<ref id="B29">
<label>29</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yingjun</surname> <given-names>YDG</given-names>
</name>
</person-group>. <article-title>Next Generation Sequencing in Quality Control and Analysis of Data</article-title>. <source>Lab Med</source> (<year>2017</year>) <volume>32</volume>(<issue>4</issue>):<page-range>255&#x2013;61</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.3969/j.issn.1673-8640.2017.04.003</pub-id>
</citation>
</ref>
<ref id="B30">
<label>30</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>C</given-names>
</name>
<name>
<surname>Cleveland</surname> <given-names>K</given-names>
</name>
<name>
<surname>Schnoll-Sussman</surname> <given-names>F</given-names>
</name>
<name>
<surname>McClure</surname> <given-names>B</given-names>
</name>
<name>
<surname>Bigg</surname> <given-names>M</given-names>
</name>
<name>
<surname>Thakkar</surname> <given-names>P</given-names>
</name>
<etal/>
</person-group>. <article-title>Identification of Low Abundance Microbiome in Clinical Samples Using Whole Genome Sequencing</article-title>. <source>Genome Biol</source> (<year>2015</year>) <volume>16</volume>:<fpage>265</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13059-015-0821-z</pub-id>
</citation>
</ref>
<ref id="B31">
<label>31</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>SF</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>YQ</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>YR</given-names>
</name>
<name>
<surname>Gu</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Fastp: An Ultra-Fast All-in-One FASTQ Preprocessor</article-title>. <source>Bioinformatics</source> (<year>2018</year>) <volume>34</volume>(<issue>17</issue>):<page-range>884&#x2013;90</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/bty560</pub-id>
</citation>
</ref>
<ref id="B32">
<label>32</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tafesse</surname> <given-names>FG</given-names>
</name>
<name>
<surname>Huitema</surname> <given-names>K</given-names>
</name>
<name>
<surname>Hermansson</surname> <given-names>M</given-names>
</name>
<name>
<surname>van der Poel</surname> <given-names>S</given-names>
</name>
<name>
<surname>van den Dikkenberg</surname> <given-names>J</given-names>
</name>
<name>
<surname>Uphoff</surname> <given-names>A</given-names>
</name>
<etal/>
</person-group>. <article-title>Both Sphingomyelin Synthases SMS1 and SMS2 are Required for Sphingomyelin Homeostasis and Growth in Human HeLa Cells</article-title>. <source>J Biol Chem</source> (<year>2007</year>) <volume>282</volume>(<issue>24</issue>):<page-range>17537&#x2013;47</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1074/jbc.M702423200</pub-id>
</citation>
</ref>
<ref id="B33">
<label>33</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pearson</surname> <given-names>WR</given-names>
</name>
</person-group>. <article-title>Rapid and Sensitive Sequence Comparison With FASTP and FASTA</article-title>. <source>Methods Enzymol</source> (<year>1990</year>) <volume>183</volume>:<fpage>63</fpage>&#x2013;<lpage>98</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/0076-6879(90)83007-v</pub-id>
</citation>
</ref>
<ref id="B34">
<label>34</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shen</surname> <given-names>W</given-names>
</name>
<name>
<surname>Le</surname> <given-names>S</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>FQ</given-names>
</name>
</person-group>. <article-title>SeqKit: A Cross-Platform and Ultrafast Toolkit for FASTA/Q File Manipulation</article-title>. <source>PloS One</source> (<year>2016</year>) <volume>11</volume>(<issue>10</issue>):<fpage>e0163962</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0163962</pub-id>
</citation>
</ref>
<ref id="B35">
<label>35</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>H</given-names>
</name>
<name>
<surname>Durbin</surname> <given-names>R</given-names>
</name>
</person-group>. <article-title>Fast and Accurate Short Read Alignment With Burrows-Wheeler Transform</article-title>. <source>Bioinformatics</source> (<year>2009</year>) <volume>25</volume>(<issue>14</issue>):<page-range>1754&#x2013;60</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btp324</pub-id>
</citation>
</ref>
<ref id="B36">
<label>36</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Al-Rubeai</surname> <given-names>M</given-names>
</name>
<name>
<surname>Emery</surname> <given-names>AN</given-names>
</name>
</person-group>. <article-title>Flow Cytometry in Animal Cell Culture</article-title>. <source>Bio/technology</source> (<year>1993</year>) <volume>11</volume>(<issue>5</issue>):<page-range>572&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nbt0593-572</pub-id>
</citation>
</ref>
<ref id="B37">
<label>37</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Navin</surname> <given-names>N</given-names>
</name>
<name>
<surname>Kendall</surname> <given-names>J</given-names>
</name>
<name>
<surname>Troge</surname> <given-names>J</given-names>
</name>
<name>
<surname>Andrews</surname> <given-names>P</given-names>
</name>
<name>
<surname>Rodgers</surname> <given-names>L</given-names>
</name>
<name>
<surname>McIndoo</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>Tumour Evolution Inferred by Single-Cell Sequencing</article-title>. <source>Nature</source> (<year>2011</year>) <volume>472</volume>(<issue>7341</issue>):<fpage>90</fpage>&#x2013;<lpage>U119</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nature09807</pub-id>
</citation>
</ref>
<ref id="B38">
<label>38</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Segal</surname> <given-names>AW</given-names>
</name>
</person-group>. <article-title>How Neutrophils Kill Microbes</article-title>. <source>Annu Rev Immunol</source> (<year>2005</year>) <volume>23</volume>:<fpage>197</fpage>&#x2013;<lpage>223</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1146/annurev.immunol.23.021704.115653</pub-id>
</citation>
</ref>
<ref id="B39">
<label>39</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Araki</surname> <given-names>K</given-names>
</name>
<name>
<surname>Turner</surname> <given-names>AP</given-names>
</name>
<name>
<surname>Shaffer</surname> <given-names>VO</given-names>
</name>
<name>
<surname>Gangappa</surname> <given-names>S</given-names>
</name>
<name>
<surname>Keller</surname> <given-names>SA</given-names>
</name>
<name>
<surname>Bachmann</surname> <given-names>MF</given-names>
</name>
<etal/>
</person-group>. <article-title>mTOR Regulates Memory CD8 T-Cell Differentiation</article-title>. <source>Nature</source> (<year>2009</year>) <volume>460</volume>(<issue>7251</issue>):<fpage>108</fpage>&#x2013;<lpage>U124</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nature08155</pub-id>
</citation>
</ref>
<ref id="B40">
<label>40</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Johdi</surname> <given-names>NA</given-names>
</name>
<name>
<surname>Ait-Tahar</surname> <given-names>K</given-names>
</name>
<name>
<surname>Sagap</surname> <given-names>I</given-names>
</name>
<name>
<surname>Jamal</surname> <given-names>R</given-names>
</name>
</person-group>. <article-title>Molecular Signatures of Human Regulatory T Cells in Colorectal Cancer and Polyps</article-title>. <source>Front Immunol</source> (<year>2017</year>) <volume>8</volume>:<elocation-id>620</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fimmu.2017.00620</pub-id>
</citation>
</ref>
<ref id="B41">
<label>41</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dodsworth</surname> <given-names>JA</given-names>
</name>
<name>
<surname>Blainey</surname> <given-names>PC</given-names>
</name>
<name>
<surname>Murugapiran</surname> <given-names>SK</given-names>
</name>
<name>
<surname>Swingley</surname> <given-names>WD</given-names>
</name>
<name>
<surname>Ross</surname> <given-names>CA</given-names>
</name>
<name>
<surname>Tringe</surname> <given-names>SG</given-names>
</name>
<etal/>
</person-group>. <article-title>Single-Cell and Metagenomic Analyses Indicate a Fermentative and Saccharolytic Lifestyle for Members of the OP9 Lineage</article-title>. <source>Nat Commun</source> (<year>2013</year>) <volume>4</volume>:<fpage>1854</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/ncomms2884</pub-id>
</citation>
</ref>
<ref id="B42">
<label>42</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kohn</surname> <given-names>T</given-names>
</name>
<name>
<surname>Wiegand</surname> <given-names>S</given-names>
</name>
<name>
<surname>Boedeker</surname> <given-names>C</given-names>
</name>
<name>
<surname>Rast</surname> <given-names>P</given-names>
</name>
<name>
<surname>Heuer</surname> <given-names>A</given-names>
</name>
<name>
<surname>Jetten</surname> <given-names>MSM</given-names>
</name>
<etal/>
</person-group>. <article-title>Planctopirus Ephydatiae, a Novel Planctomycete Isolated From a Freshwater Sponge</article-title>. <source>Syst Appl Microbiol</source> (<year>2020</year>) <volume>43</volume>(<issue>1</issue>):<elocation-id>126022</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.syapm.2019.126022</pub-id>
</citation>
</ref>
<ref id="B43">
<label>43</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ren</surname> <given-names>ZG</given-names>
</name>
<name>
<surname>Li</surname> <given-names>A</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>JW</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>L</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>ZJ</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>HF</given-names>
</name>
<etal/>
</person-group>. <article-title>Gut Microbiome Analysis as a Tool Towards Targeted Non-Invasive Biomarkers for Early Hepatocellular Carcinoma</article-title>. <source>Gut</source> (<year>2019</year>) <volume>68</volume>(<issue>6</issue>):<page-range>1014&#x2013;23</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1136/gutjnl-2017-315084</pub-id>
</citation>
</ref>
<ref id="B44">
<label>44</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yelena</surname> <given-names>L</given-names>
</name>
<name>
<surname>Amnon</surname> <given-names>A</given-names>
</name>
<name>
<surname>Rita</surname> <given-names>N</given-names>
</name>
<name>
<surname>Atara</surname> <given-names>U-Y</given-names>
</name>
<name>
<surname>Ella</surname> <given-names>V</given-names>
</name>
<name>
<surname>Oranit</surname> <given-names>C-E</given-names>
</name>
<etal/>
</person-group>. <article-title>Alterations in the Gut Microbiome in the Progression of Cirrhosis to Hepatocellular Carcinoma</article-title>. <source>mSystems</source> (<year>2020</year>) <volume>5</volume>(<issue>3</issue>):<page-range>e00153&#x2013;20</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1128/mSystems.00153-20</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>