<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="methods-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Microbiol.</journal-id>
<journal-title>Frontiers in Microbiology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Microbiol.</abbrev-journal-title>
<issn pub-type="epub">1664-302X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmicb.2024.1394204</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Microbiology</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A GLM-based zero-inflated generalized Poisson factor model for analyzing microbiome data</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Chi</surname> <given-names>Jinling</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2668575/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Ye</surname> <given-names>Jimin</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Zhou</surname> <given-names>Ying</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x0002A;</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>School of Mathematics and Statistics, Xidian University</institution>, <addr-line>Xi&#x00027;an</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>School of Mathematical Sciences, Heilongjiang University</institution>, <addr-line>Harbin</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Hyun-Seob Song, University of Nebraska-Lincoln, United States</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Mohamed R. Abonazel, Cairo University, Egypt</p>
<p>Aditya Mishra, Flatiron Institute, United States</p>
<p>Pierluigi Polese, University of Udine, Italy</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Jimin Ye <email>jmye&#x00040;mail.xidian.edu.cn</email></corresp>
<corresp id="c002">Ying Zhou <email>zhouying&#x00040;hlju.edu.cn</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>30</day>
<month>05</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1394204</elocation-id>
<history>
<date date-type="received">
<day>01</day>
<month>03</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>20</day>
<month>05</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2024 Chi, Ye and Zhou.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Chi, Ye and Zhou</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<sec>
<title>Motivation</title>
<p>High-throughput sequencing technology facilitates the quantitative analysis of microbial communities, improving the capacity to investigate the associations between the human microbiome and diseases. Our primary motivating application is to explore the association between gut microbes and obesity. The complex characteristics of microbiome data, including high dimensionality, zero inflation, and over-dispersion, pose new statistical challenges for downstream analysis.</p>
</sec>
<sec>
<title>Results</title>
<p>We propose a GLM-based zero-inflated generalized Poisson factor analysis (GZIGPFA) model to analyze microbiome data with complex characteristics. The GZIGPFA model is based on a zero-inflated generalized Poisson (ZIGP) distribution for modeling microbiome count data. A link function between the generalized Poisson rate and the probability of excess zeros is established within the generalized linear model (GLM) framework. The latent parameters of the GZIGPFA model constitute a low-rank matrix comprising a low-dimensional score matrix and a loading matrix. An alternating maximum likelihood algorithm is employed to estimate the unknown parameters, and cross-validation is utilized to determine the rank of the model in this study. The proposed GZIGPFA model demonstrates superior performance and advantages through comprehensive simulation studies and real data applications.</p>
</sec></abstract>
<kwd-group>
<kwd>factor analysis</kwd>
<kwd>GLM</kwd>
<kwd>microbiome data</kwd>
<kwd>zero inflation</kwd>
<kwd>ZIGP model</kwd>
</kwd-group>
<counts>
<fig-count count="5"/>
<table-count count="2"/>
<equation-count count="9"/>
<ref-count count="70"/>
<page-count count="15"/>
<word-count count="9642"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Systems Microbiology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>The human microbiome is the collection of all microorganisms that live in and associate with the human body, including bacteria, archaeobacteria, protists, and viruses, distributed in the nasal cavity, oral cavity, skin, gastrointestinal tract, and genitourinary tract. The growing significance of the microbiome in ecosystems is increasingly recognized. In particular, the relationship between gut microorganisms and human health has garnered widespread scientific interest. Over time, an increasing number of studies have demonstrated that dysbiosis of the gut microbial community is associated with complex diseases, such as human gastrointestinal disorders (Willing et al., <xref ref-type="bibr" rid="B63">2010</xref>; Machiels et al., <xref ref-type="bibr" rid="B33">2013</xref>; Knights et al., <xref ref-type="bibr" rid="B23">2014</xref>), metabolic traits, diabetes (Turnbaugh et al., <xref ref-type="bibr" rid="B57">2006</xref>; Wen et al., <xref ref-type="bibr" rid="B61">2008</xref>; Vijay-Kumar et al., <xref ref-type="bibr" rid="B59">2010</xref>), obesity (McKnite et al., <xref ref-type="bibr" rid="B39">2012</xref>; Carlisle et al., <xref ref-type="bibr" rid="B5">2013</xref>; Parks et al., <xref ref-type="bibr" rid="B42">2013</xref>), and inflammatory bowel disease (Frank et al., <xref ref-type="bibr" rid="B17">2007</xref>). These investigations significantly contribute on exploring the causes and treatments of diseases. Furthermore, complex interactions between hosts and microbiota are also observed in various ecosystems. For example, in marine ecosystems, microbial communities associated with seaweed play an vital role in the development, reproduction, function, and defense of seaweeds (Egan et al., <xref ref-type="bibr" rid="B13">2013</xref>; Singh and Reddy, <xref ref-type="bibr" rid="B51">2016</xref>). Therefore, it becomes crucial to quantify the abundance of microbial taxa and investigate the association between microbiota and diseases or traits.</p>
<p>The development of high-throughput sequencing (HTS) technology has been widely employed in microbial research, enabling researchers to identify the composition and abundance of microbial species directly (Kuczynski et al., <xref ref-type="bibr" rid="B25">2011</xref>). Microbiome data are typically generated by extracting samples from the specified environment, followed by sequencing the 16S rRNA genes of the DNA extracts using high-throughput sequencing technology. The obtained sequence reads are compared with the reference 16S rRNA database and assigned to Operational Taxonomic Units (OTUs) based on a sequence similarity threshold (e.g., 97%; Tyler et al., <xref ref-type="bibr" rid="B58">2014</xref>). High-throughput sequencing data provide valuable insights for investigating the relationship between the microbiome and the host environment or clinical factors. As a motivating application, we consider the gut microbiome data in Sun et al. (<xref ref-type="bibr" rid="B54">2019</xref>), which explores the association between gut microbes and obesity. The authors sequenced 16S rDNA genes of 48 individuals and obtained a dataset with 895 OTUs, where the number of variables (i.e., OTUs) vastly exceeds the number of observations (i.e., the number of samples). Moreover, we found that &#x0007E;45% of the OTU counts were zero, and the variance of the data significantly exceeded the mean. These characteristics are manifestations of high dimensionality, zero inflation, and over-dispersion, which may distort downstream analysis. However, many microbiome datasets exhibit the same problems as the motivating data, posing challenges for statistical analysis. Firstly, most microbiome data are non-negative counts with a large number of zeros (i.e., zero-inflated; Xu et al., <xref ref-type="bibr" rid="B65">2015</xref>; Kaul et al., <xref ref-type="bibr" rid="B22">2017</xref>). Some of these observed zeros result from insufficient sequencing depth (i.e., library size, which is the total number of reads obtained by per sample from equipment) or other technical reasons that result in some taxa not being detected, and others are the fact that some taxa are very rare and not present in most samples (Silverman et al., <xref ref-type="bibr" rid="B49">2020</xref>). Traditional statistical methods may not accurately estimate the parameters of the data distribution due to the preponderance of zeros, leading to biased results (Campbell, <xref ref-type="bibr" rid="B4">2021</xref>). Secondly, microbial abundance data only represent relative information in observed samples and cannot describe the abundance in the entire ecosystem (Mandal et al., <xref ref-type="bibr" rid="B34">2015</xref>; Gloor et al., <xref ref-type="bibr" rid="B18">2017</xref>). Moreover, the sequencing depth varies among samples, and even the variation between samples is magnitude (Sims et al., <xref ref-type="bibr" rid="B50">2014</xref>). Finally, microbiome data are typically over-dispersed and high-dimensional (Kurtz et al., <xref ref-type="bibr" rid="B26">2015</xref>; Xu et al., <xref ref-type="bibr" rid="B65">2015</xref>; Armstrong et al., <xref ref-type="bibr" rid="B2">2022</xref>). The number of taxa in the OTUs table may significantly exceed the number of observed samples, which is a sign of high dimensionality. The high dimensionality of the data may strain computational resources and increase the risk of overfitting. Meanwhile, the standard model may underestimate the true variation within the data when over-dispersion exists, leading to inaccurate estimation and hypothesis testing (Robinson et al., <xref ref-type="bibr" rid="B48">2009</xref>; Love et al., <xref ref-type="bibr" rid="B32">2014</xref>). Detecting associations between microbes and diseases remains challenging because of the complex features of microbiome data and the limitations of current statistical methods. Therefore, it is necessary to develop novel statistical analysis methods for the characteristics of microbiome data.</p>
<p>Zero-inflation and over-dispersion of count data have received widespread attention from scholars recently. Wagh and Kamalja (<xref ref-type="bibr" rid="B60">2017</xref>) briefly reviews different zero-inflated models for handling count data and the performance of their parameter estimation, which provides suggestions for selecting parameter estimation methods for zero-inflated models. Motivated by zero-inflation and over-dispersion problems, a zero-inflated negative binomial (ZINB) mixed regression approach is proposed to analyze the data on the length of stay for pancreas disorder (Yau et al., <xref ref-type="bibr" rid="B67">2003</xref>). However, in a few cases, the parameter estimation algorithm for the ZINB regression model fails to converge (Lambert, <xref ref-type="bibr" rid="B27">1992</xref>). A zero-inflated generalized Poisson (ZIGP) regression model has been proposed to model domestic violence data with too many zeros (Famoye and Singh, <xref ref-type="bibr" rid="B15">2006</xref>). It is a strong competitor to the Poisson and negative binomial regression model when the count data is over-dispersed. In addition, zero-inflated generalized Poisson and zero-inflated negative binomial regression models were used in QTL mapping studies for the count traits with excess zeros (Cui and Yang, <xref ref-type="bibr" rid="B10">2009</xref>; Moghimbeigi, <xref ref-type="bibr" rid="B41">2015</xref>; Chi et al., <xref ref-type="bibr" rid="B7">2020</xref>). More recently, Tirozzi et al. (<xref ref-type="bibr" rid="B56">2022</xref>) used zero-inflation models to assess long-term population trends and elucidate the effects of environmental bias, over-dispersion, and zero-inflation on the population trend estimates. These studies provide some inspiration for analyzing microbiome data with complex characteristics.</p>
<p>In recent years, extensive research has been conducted by scholars to address the challenges associated with microbiome data, including zero inflation, high dimensionality, and over-dispersion (Zhang et al., <xref ref-type="bibr" rid="B69">2018</xref>; Xu et al., <xref ref-type="bibr" rid="B66">2020</xref>; Jiang et al., <xref ref-type="bibr" rid="B20">2023</xref>). Two typical methods have been proposed to address the zero-inflated structure of sequencing data. One method is replacing the zeros with small non-zero positive number (pseudo count; Chen and Li, <xref ref-type="bibr" rid="B6">2013</xref>; Lin et al., <xref ref-type="bibr" rid="B31">2014</xref>). However, the effects of creating pseudo count has not been evaluated thoroughly when the data contain excessive zeros. Besides, the choice of pseudo count may impact subsequent analysis (Costea et al., <xref ref-type="bibr" rid="B9">2014</xref>), and this approach is not statistically rigorous. Moreover, the idea of multiplicative replacement has been proposed. The non-parametric replacement method can be used to adjust the data through multiplicative modification under simple conditions (a small number of zeros; Mart&#x000ED;n-Fern&#x000E1;ndez et al., <xref ref-type="bibr" rid="B35">2003</xref>). In other cases, more sophisticated model-based methods can be utilized to replace zeros in the data (Mart&#x000ED;n-Fern&#x000E1;ndez et al., <xref ref-type="bibr" rid="B36">2012</xref>). Recently, a Bayesian-multiplicative treatment has been proposed to solve the problem of count zero, which assumes a Dirichlet prior for the proportions and replaces the zeros with posterior Bayesian estimates (Mart&#x000ED;n-Fern&#x000E1;ndez et al., <xref ref-type="bibr" rid="B37">2014</xref>). The other standard and widely used method is to construct a two-part model with a point probability mass at zero along with another parametric distribution, such as zero-inflated Gaussian model (Xu et al., <xref ref-type="bibr" rid="B65">2015</xref>), zero-inflated lognormal model (Sohn et al., <xref ref-type="bibr" rid="B52">2015</xref>), zero-inflated Poisson model (Xu et al., <xref ref-type="bibr" rid="B66">2020</xref>), zero-inflated negative binomial model (Jiang et al., <xref ref-type="bibr" rid="B21">2019</xref>; Zhang and Yi, <xref ref-type="bibr" rid="B70">2020</xref>), and many others (Peng et al., <xref ref-type="bibr" rid="B45">2016</xref>; Tang and Chen, <xref ref-type="bibr" rid="B55">2018</xref>; Zeng et al., <xref ref-type="bibr" rid="B68">2022</xref>; Jiang et al., <xref ref-type="bibr" rid="B20">2023</xref>). The advantage of this method is that an appropriate model can be selected according to the nature of the data. For example, the zero-inflated negative binomial model can effectively address the issue of zero-inflated and over-dispersion in the data because the negative binomial provides a standard statistical model for over-dispersed data.</p>
<p>The other well-known challenge for analyzing microbial data is the high dimensionality of the data. Generally, the number of taxa usually far exceeds the observed samples in the OTUs table, which is a symbol of high dimensionality (Armstrong et al., <xref ref-type="bibr" rid="B2">2022</xref>). Therefore, dimensionality reduction technology is used to map high-dimensional data into a potential low-dimensional space while retaining the primary information in the data intact to facilitate the subsequent analysis, which is a desirable preprocessing step (Fan et al., <xref ref-type="bibr" rid="B16">2015</xref>; Jasner et al., <xref ref-type="bibr" rid="B19">2021</xref>).</p>
<p>Factor analysis, an extensively employed technique, serves as a prominent method for dimensionality reduction of high-dimensional data. Pierson and Yau (<xref ref-type="bibr" rid="B46">2015</xref>) proposed a zero-inflated factor analysis (ZIFA) model to explicitly consider excess zeros in Single-cell RNA-seq data. However, the ZIFA model preprocesses count data via a normal transformation, which may overlook its inherent count nature and potentially result in information loss during the preprocessing step. Lee et al. (<xref ref-type="bibr" rid="B28">2013</xref>) developed a Poisson factor model with offsets to explicitly incorporate the special features that count nature and heterogeneous library size (the total reads per sample). Subsequently, the negative binomial factor regression model was proposed to reduce the dimensionality of microbial abundance data, and then model the associations of microbial abundance and host-associated features by including only a subset of the predictors for a few latent factors (Mishra and M&#x000FC;ller, <xref ref-type="bibr" rid="B40">2022</xref>). However, these two methods (Poisson factor model and negative binomial factor model) are only suitable for data that does not contain excessive zeros, which fail to consider zero inflation. Sohn and Li (<xref ref-type="bibr" rid="B53">2017</xref>) proposed a GLM-based ordination method for microbiome samples (GOMMS), which employs a zero-inflated quasi-Poisson factor model to dimensionality reduction and overcome the challenge of zero-inflation. However, this method assumes that each taxa has a fixed probability of zero, which is generally not easily satisfied. More recently, Xu et al. (<xref ref-type="bibr" rid="B66">2020</xref>) proposed a factor analysis model based on the zero-inflated Poisson distribution (ZIPFA), which can more flexibly adapt to some characteristics of microbial data, such as count value, excessive zeros, and high dimensionality. A significant critique of Poisson models is the failure to accommodate over-dispersion, which has been widely observed for microbiome data. Following this line of research, we combined the zero-inflated generalized Poisson distribution with factor analysis under the framework of the generalized linear model to propose a GLM-based zero-inflated generalized Poisson factor analysis (GZIGPFA) model, which provides a valuable dimensionality reduction tool for microbiome data. The GZIGPFA model can also address the issues of over-dispersion and handle the zero-inflated structure. Furthermore, our method models absolute abundance directly, avoiding the information loss attributable to data transformation.</p>
<p>The rest of this paper is organized below. Section 2 presents a new GZIGPFA model for handling microbiome data and introduces methods for parameter estimation and rank selection. A simulation and comparison study are conducted in Section 3 to demonstrate the performance of the proposed method. In Section 4, we apply our method to the gut microbial data to explore the association between gut microbes and obesity. In Section 5, a conclusion of this paper is drawn with a discussion of extensions and areas for subsequent work.</p>
</sec>
<sec id="s2">
<title>2 Method</title>
<p>For <italic>i</italic> &#x0003D; 1, 2, &#x02026;, <italic>n</italic> and <italic>j</italic> &#x0003D; 1, 2, &#x02026;, <italic>m</italic>, let <italic>y</italic><sub><italic>ij</italic></sub> denote the count of the <italic>j</italic>-th taxon from the <italic>i</italic>-th individual, then, an <italic>n</italic>&#x000D7;<italic>m</italic> microbial abundance matrix can be expressed as <italic><bold>Y</bold></italic> &#x0003D; (<sub><italic>y</italic><sub><italic>ij</italic></sub>)<italic>n</italic>&#x000D7;<italic>m</italic></sub>. Denote the <italic>i</italic>th row of matrix <italic><bold>Y</bold></italic> as <italic><bold>y</bold></italic><sub>(<italic>i</italic>)</sub> &#x0003D; (<italic>y</italic><sub><italic>i</italic>1</sub>, &#x02026;, <italic>y</italic><sub><italic>im</italic></sub>), refer to as the <italic>i</italic>-th sample of sequencing data.</p>
<sec>
<title>2.1 Zero-inflated generalized Poisson factor model</title>
<p>The microbiome dataset typically presents as a highly skewed non-negative count matrix with numerous zeros, often characterized by over-dispersion. Therefore, we build statistical models to address these issues for microbiome data.</p>
<p>The presence of zeros in microbiome data may be true absences or undetected taxa. Considering the over-dispersion characteristics of the data, we assume that the sequencing count <italic>y</italic><sub><italic>ij</italic></sub> follows the zero-inflated generalized Poisson (ZIGP) distribution (Famoye and Singh, <xref ref-type="bibr" rid="B15">2006</xref>):</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>~</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mtable columnalign='left'><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo></mml:mrow></mml:mtd><mml:mtd columnalign='left'><mml:mrow><mml:mtext>with&#x000A0;probability&#x000A0;</mml:mtext><mml:msub><mml:mi>&#x003D5;</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:mi>G</mml:mi><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:msub><mml:mo>&#x003BB;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo></mml:mrow></mml:mtd><mml:mtd columnalign='left'><mml:mrow><mml:mtext>with&#x000A0;probability&#x000A0;</mml:mtext><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:msub><mml:mi>&#x003D5;</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>where <italic>&#x003D5;</italic><sub><italic>ij</italic></sub> is the zero-inflation parameter describing the probability of excess zero; <italic>GP</italic>(<italic>T</italic><sub><italic>i</italic></sub>&#x003BB;<sub><italic>ij</italic></sub>, &#x003B1;) is the generalized Poisson distribution (Consul and Famoye, <xref ref-type="bibr" rid="B8">1992</xref>; Famoye, <xref ref-type="bibr" rid="B14">1993</xref>), with the probability function</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>&#x003BB;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>!</mml:mo></mml:mrow></mml:mfrac><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>&#x003BB;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>&#x003BB;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mtext>exp</mml:mtext></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mo>-</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>&#x003BB;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>&#x003BB;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003BB;<sub><italic>ij</italic></sub> and &#x003B1; are the mean and dispersion parameters of the generalized Poisson part, respectively; <italic>T</italic><sub><italic>i</italic></sub> is the relative library size of the <italic>i</italic>-th sample, which is utilized to regulate &#x003BB;<sub><italic>ij</italic></sub>. Generally, there are many representations of <italic>T</italic><sub><italic>i</italic></sub> (Anders and Huber, <xref ref-type="bibr" rid="B1">2010</xref>; Eddy, <xref ref-type="bibr" rid="B12">2011</xref>; Badri et al., <xref ref-type="bibr" rid="B3">2020</xref>; Mishra and M&#x000FC;ller, <xref ref-type="bibr" rid="B40">2022</xref>). In this paper, we take</p>
<disp-formula id="E3"><mml:math id="M3"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>m</mml:mi></mml:munderover><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle><mml:mo>/</mml:mo><mml:mtext>median</mml:mtext><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>m</mml:mi></mml:munderover><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>m</mml:mi></mml:munderover><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula>
<p>Next, the link between the zero-inflation probability <italic>&#x003D5;</italic><sub><italic>ij</italic></sub> and the mean parameter &#x003BB;<sub><italic>ij</italic></sub> is established according to Lambert (<xref ref-type="bibr" rid="B27">1992</xref>). Typically, an increase in the number of zeros in the data results in a smaller overall mean. Therefore, a negative relationship between <italic>&#x003D5;</italic><sub><italic>ij</italic></sub> and &#x003BB;<sub><italic>ij</italic></sub> is established, i.e.,</p>
<disp-formula id="E4"><label>(3)</label><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">logit</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mi>&#x003C4;</mml:mi><mml:mo class="qopname">log</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mo>&#x003BB;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003C4; is the shape parameter; logit(<italic>&#x003D5;</italic><sub><italic>ij</italic></sub>) and log(&#x003BB;<sub><italic>ij</italic></sub>) are the link functions for the probability of zero-inflation and the mean of generalized Poisson in the generalized linear model (GLM), respectively. Let <inline-formula><mml:math id="M5"><mml:mstyle mathvariant="bold"><mml:mi>&#x0039B;</mml:mi></mml:mstyle><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mo>&#x003BB;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> and <inline-formula><mml:math id="M6"><mml:mstyle mathvariant="bold"><mml:mi>&#x003A6;</mml:mi></mml:mstyle><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> be the matrix forms of &#x003BB;<sub><italic>ij</italic></sub> and <italic>&#x003D5;</italic><sub><italic>ij</italic></sub>, respectively. Therefore, the matrix form of <xref ref-type="disp-formula" rid="E4">Equation (3)</xref> can be expressed as logit(<bold>&#x003A6;</bold>) &#x0003D; &#x02212;&#x003C4;log(<bold>&#x0039B;</bold>).</p>
<p>The ZIGP model (<xref ref-type="disp-formula" rid="E1">Equation 1</xref>) described above can accommodate simultaneously zero inflation and over-dispersion count data. Furthermore, upon review of the existing literature about microbiome data analysis, the ZIGP model represents the inaugural utilization of the zero-inflated generalized Poisson model in microbiome datasets. In the following, we intend to solve the prevalent issue of high dimensionality in the microbiome data with a factor analysis model. Therefore, we propose a GLM-based zero-inflated generalized Poisson factor analysis (GZIGPFA) model to provide a suitable model for zero-inflated, over-dispersed, and high-dimensional microbiome data.</p>
<p>Assume that matrix log(<bold>&#x0039B;</bold>) has a low-rank structure log(<bold>&#x0039B;</bold>) &#x0003D; <italic><bold>FL</bold></italic><sup><italic>T</italic></sup> with rank <italic>K</italic> (Lee et al., <xref ref-type="bibr" rid="B28">2013</xref>), where <italic><bold>F</bold></italic>&#x02208;&#x0211D;<sup><italic>n</italic>&#x000D7;<italic>K</italic></sup> is the factor score matrix and <inline-formula><mml:math id="M7"><mml:mstyle mathvariant="bold-italic"><mml:mi>F</mml:mi></mml:mstyle><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>f</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>f</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> with <italic><bold>f</bold></italic><bold><sub>(<italic>i</italic>)</sub></bold> &#x0003D; (<italic>f</italic><sub><italic>i</italic>1</sub>, &#x02026;, <italic>f</italic><sub><italic>iK</italic></sub>), <italic>i</italic> &#x0003D; 1, 2, &#x02026;, <italic>n</italic>; <italic><bold>L</bold></italic>&#x02208;&#x0211D;<sup><italic>m</italic>&#x000D7;<italic>K</italic></sup> is the loading matrix and <inline-formula><mml:math id="M8"><mml:mstyle mathvariant="bold-italic"><mml:mi>L</mml:mi></mml:mstyle><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>l</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>l</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> with <italic><bold>l</bold></italic><sub>(<italic>j</italic>)</sub> &#x0003D; (<italic>l</italic><sub><italic>j</italic>1</sub>, <italic>l</italic><sub><italic>j</italic>2</sub>, &#x02026;, <italic>l</italic><sub><italic>jK</italic></sub>), <italic>j</italic> &#x0003D; 1, 2, &#x02026;, <italic>m</italic>. Then, we consider the following zero-inflated generalized Poisson factor model:</p>
<disp-formula id="E5"><label>(4)</label><mml:math id="M9"><mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mtable columnalign='left'><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>~</mml:mo><mml:mtext>ZIGP</mml:mtext><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:msub><mml:mo>&#x003BB;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>&#x003D5;</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:mtext>logit</mml:mtext><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>&#x003D5;</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>&#x003C4;</mml:mi><mml:mi>log</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mo>&#x003BB;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:mi>log</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mo>&#x003BB;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mi>l</mml:mi><mml:mrow><mml:mi>j</mml:mi><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mi>l</mml:mi><mml:mrow><mml:mi>j</mml:mi><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>+</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>K</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>l</mml:mi><mml:mrow><mml:mi>j</mml:mi><mml:mi>K</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>where <italic>f</italic><sub><italic>ik</italic></sub> is an element of the matrix <italic><bold>F</bold></italic>, denoting the <italic>k</italic>th factor score for the <italic>i</italic>-th sample; <italic>l</italic><sub><italic>jk</italic></sub> is an element of matrix <italic><bold>L</bold></italic>, denoting the loading of the <italic>j</italic>th taxon on the <italic>k</italic>th factor, where <italic>i</italic> &#x0003D; 1, 2, &#x02026;, <italic>n</italic>, <italic>j</italic> &#x0003D; 1, 2, &#x02026;, <italic>m</italic>, and <italic>k</italic> &#x0003D; 1, 2, &#x02026;, <italic>K</italic>. In this model, the logarithm is the canonical link function in the generalized linear model (GLM) framework (McCullagh and Nelder, <xref ref-type="bibr" rid="B38">1989</xref>).</p>
<p>After the rank <italic>K</italic> is determined and the unknown parameters &#x003B1;, &#x003C4;, <italic><bold>F</bold></italic>, <italic><bold>L</bold></italic> in the model (<xref ref-type="disp-formula" rid="E5">Equation 4</xref>) are estimated, we reduce the dimensionality of the microbiome dataset from <italic>m</italic> to <italic>K</italic>. The score matrix <italic><bold>F</bold></italic> possesses an equivalent sample size to the original microbiome dataset <italic><bold>Y</bold></italic> but only has <italic>K</italic> variables. In subsequent work, it is easier to perform association analysis between disease phenotypes and the low-dimensional score matrix, providing a brief tool for investigating the relationship between microbiome and disease.</p>
</sec>
<sec>
<title>2.2 An alternating maximum likelihood algorithm</title>
<p>To estimate the unknown parameters &#x003B1;, &#x003C4;, <italic><bold>F</bold></italic>, <italic><bold>L</bold></italic> in model (<xref ref-type="disp-formula" rid="E5">Equation 4</xref>), we adopt a method that maximizes the ZIGP likelihood function:</p>
<disp-formula id="E6"><label>(5)</label><mml:math id="M10"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mi>L</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003B1;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x003C4;</mml:mi><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold-italic"><mml:mi>F</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold-italic"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x0220F;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x0220F;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>&#x003BB;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>p</italic>(<italic>y</italic><sub><italic>ij</italic></sub>; <italic>T</italic><sub><italic>i</italic></sub>&#x003BB;<sub><italic>ij</italic></sub>, &#x003B1;) is the probability function of GP distribution (<xref ref-type="disp-formula" rid="E2">Equation 2</xref>); <inline-formula><mml:math id="M11"><mml:mo class="qopname">log</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mo>&#x003BB;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo class="qopname">&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:munderover><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and logit(<italic>&#x003D5;</italic><sub><italic>ij</italic></sub>) &#x0003D; &#x02212;&#x003C4;log(&#x003BB;<sub><italic>ij</italic></sub>). In <xref ref-type="disp-formula" rid="E6">Equation (5)</xref>, &#x003B1;, &#x003C4;, <italic><bold>F</bold></italic>, <italic><bold>L</bold></italic> are the unknown parameters, and it is challenging to maximize the likelihood directly. Therefore, we consider an alternating maximum likelihood algorithm in the GLM framework to estimate the parameters.</p>
<p>In order to obtain the initial <italic><bold>F</bold></italic> and <italic><bold>L</bold></italic>, we apply the singular value decomposition (SVD) to the log-transformed matrix <inline-formula><mml:math id="M12"><mml:mover accent="false"><mml:mrow><mml:mi>Y</mml:mi></mml:mrow><mml:mo>&#x0007E;</mml:mo></mml:mover></mml:math></inline-formula> and obtain the singular, i.e., <inline-formula><mml:math id="M13"><mml:mo class="qopname">log</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>Y</mml:mi></mml:mrow><mml:mo class="qopname">&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>U</mml:mi><mml:mo>&#x003A3;</mml:mo><mml:msup><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>. Set <italic><bold>L</bold></italic><sup><italic>old</italic></sup> &#x0003D; <italic>V</italic><sup><italic>T</italic></sup> and <inline-formula><mml:math id="M14"><mml:msup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>F</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>l</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>&#x003A3;</mml:mo></mml:mrow><mml:mrow><mml:mn>11</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo>,</mml:mo><mml:mn>2</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>&#x003A3;</mml:mo></mml:mrow><mml:mrow><mml:mn>22</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo>,</mml:mo><mml:mi>K</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>&#x003A3;</mml:mo></mml:mrow><mml:mrow><mml:mi>K</mml:mi><mml:mi>K</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>, where &#x003A3;<sub><italic>kk</italic></sub>, <italic>k</italic> &#x0003D; 1, &#x02026;, <italic>K</italic> is the <italic>k</italic>th diagonal element of &#x003A3;.</p>
<p><italic><bold>Step 1</bold></italic>: Assuming that factor score matrix <italic><bold>F</bold></italic> is known as <italic><bold>F</bold></italic><sup><italic>old</italic></sup>, a ZIGP regression model is fitted with the <italic>j</italic>th column of matrix <italic><bold>Y</bold></italic> (denoted by <italic><bold>y</bold></italic><sub><italic>j</italic></sub>) as the response and <italic><bold>F</bold></italic><sup><italic>old</italic></sup> as a covariate matrix, the regression model can be written as</p>
<disp-formula id="E7"><mml:math id="M15"><mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mtable columnalign='left'><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:msub><mml:mstyle mathvariant='bold-italic' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mi>j</mml:mi></mml:msub><mml:mo>~</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mtable columnalign='left'><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo></mml:mrow></mml:mtd><mml:mtd columnalign='left'><mml:mrow><mml:mtext>with&#x000A0;probability&#x000A0;</mml:mtext><mml:msub><mml:mi>&#x003D5;</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:mi>G</mml:mi><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold-italic' mathsize='normal'><mml:mi>T</mml:mi></mml:mstyle><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003BB;</mml:mi></mml:mstyle><mml:mi>j</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo></mml:mrow></mml:mtd><mml:mtd columnalign='left'><mml:mrow><mml:mtext>with&#x000A0;probability&#x000A0;</mml:mtext><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:msub><mml:mi>&#x003D5;</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mrow></mml:mrow></mml:mtd></mml:mtr><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:mi>log</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003BB;</mml:mi></mml:mstyle><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:msup><mml:mstyle mathvariant='bold-italic' mathsize='normal'><mml:mi>F</mml:mi></mml:mstyle><mml:mrow><mml:mi>o</mml:mi><mml:mi>l</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:msup><mml:msubsup><mml:mstyle mathvariant='bold-italic' mathsize='normal'><mml:mi>l</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mi>T</mml:mi></mml:msubsup><mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:mtext>logit</mml:mtext><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>&#x003D5;</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>&#x003C4;</mml:mi><mml:mi>log</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003BB;</mml:mi></mml:mstyle><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>where the vector <italic><bold>T</bold></italic> &#x0003D; (<italic>T</italic><sub>1</sub>, <italic>T</italic><sub>2</sub>, &#x02026;, <italic>T</italic><sub><italic>n</italic></sub>) is the relative library size vector; the regression coefficient vector <italic><bold>l</bold></italic><sub>(<italic>j</italic>)</sub> &#x0003D; (<italic>l</italic><sub><italic>j</italic>1</sub>, <italic>l</italic><sub><italic>j</italic>2</sub>, &#x02026;, <italic>l</italic><sub><italic>jK</italic></sub>) is the <italic>j</italic>th row of the factor loading matrix <inline-formula><mml:math id="M16"><mml:msup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mi>e</mml:mi><mml:mi>w</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>l</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>l</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>l</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>; the vectors <bold>&#x003BB;</bold><sub><italic>j</italic></sub> and <bold>&#x003D5;</bold><sub><italic>j</italic></sub> are the <italic>j</italic>th column of the matrices <bold>&#x0039B;</bold> and <bold>&#x003A6;</bold>, respectively.</p>
<p>To estimate the unknown parameter vector <inline-formula><mml:math id="M17"><mml:mstyle mathvariant="bold-italic"><mml:mi>&#x003B8;</mml:mi></mml:mstyle><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003C4;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>l</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> of the regression model, we should maximize the likelihood function. However, the explicit solution of each parameter cannot be obtained by directly using the maximum likelihood estimation method. Therefore, we perform parameter estimation of the regression model with the EM algorithm. The detailed procedure of the EM algorithm is given in <xref ref-type="supplementary-material" rid="SM1">Appendix A</xref>.</p>
<p>Since matrix <italic><bold>Y</bold></italic> has <italic>m</italic> columns, we need to fit <italic>m</italic> GLMs to obtain <italic>m</italic> rows of <italic><bold>L</bold></italic><sup><italic>new</italic></sup>. However, the proposed model assumes that the &#x003C4; and &#x003B1; remain the same across all <italic>m</italic> different GLMs. To accommodate this, we combine <italic><bold>y</bold></italic><sub>1</sub>, &#x02026;, <italic><bold>y</bold></italic><sub><italic>m</italic></sub> into a column vector and solve all <italic>m</italic> models simultaneously to obtain the globally optimal &#x003C4; and &#x003B1; values.</p>
<p>After estimating &#x003C4;, &#x003B1; and <italic><bold>L</bold></italic>, we continue to update <italic><bold>F</bold></italic>. The process is similar to Step 1.</p>
<p><italic><bold>Step 2</bold></italic>: Fit a ZIGP regression model with the <italic>i</italic>th row of matrix <italic><bold>Y</bold></italic> (denoted by <italic><bold>y</bold></italic><sub>(<italic>i</italic>)</sub>) as the response and the estimated loading matrix <italic><bold>L</bold></italic> &#x0003D; <italic><bold>L</bold></italic><sup><italic>new</italic></sup> from the previous step as a covariate, the regression model can be written as</p>
<disp-formula id="E8"><mml:math id="M18"><mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mtable columnalign='left'><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:msub><mml:mstyle mathvariant='bold-italic' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msub><mml:mo>~</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mtable columnalign='left'><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo></mml:mrow></mml:mtd><mml:mtd columnalign='left'><mml:mrow><mml:mtext>with&#x000A0;probability&#x000A0;</mml:mtext><mml:msub><mml:mi>&#x003D5;</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:mi>G</mml:mi><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003BB;</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo></mml:mrow></mml:mtd><mml:mtd columnalign='left'><mml:mrow><mml:mtext>with&#x000A0;probability&#x000A0;</mml:mtext><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:msub><mml:mi>&#x003D5;</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mrow></mml:mrow></mml:mtd></mml:mtr><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:mi>log</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003BB;</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:msub><mml:mstyle mathvariant='bold-italic' mathsize='normal'><mml:mi>f</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msub><mml:msup><mml:mstyle mathvariant='bold-italic' mathsize='normal'><mml:mi>L</mml:mi></mml:mstyle><mml:mrow><mml:mi>n</mml:mi><mml:mi>e</mml:mi><mml:mi>w</mml:mi><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:mtext>logit</mml:mtext><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>&#x003D5;</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>&#x003C4;</mml:mi><mml:mi>log</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003BB;</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>where <italic>T</italic><sub><italic>i</italic></sub> is the relative library size of the <italic>i</italic>th sample; the regression coefficient vector <italic><bold>f</bold></italic><sub>(<italic>i</italic>)</sub> &#x0003D; (<italic>f</italic><sub><italic>i</italic>1</sub>, <italic>f</italic><sub><italic>i</italic>2</sub>, &#x02026;, <italic>f</italic><sub><italic>iK</italic></sub>) is the <italic>i</italic>th row of the factor score matrix <inline-formula><mml:math id="M19"><mml:msup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>F</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mi>e</mml:mi><mml:mi>w</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>f</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>f</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>f</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>; the vectors <bold>&#x003BB;</bold><sub>(<italic>i</italic>)</sub> and <bold>&#x003D5;</bold><sub>(<italic>i</italic>)</sub> are the <italic>i</italic>th row of the matrices <bold>&#x0039B;</bold> and <bold>&#x003A6;</bold>, respectively. Next, the parameters &#x003B1;, &#x003C4;, and <italic><bold>f</bold></italic><sub>(<italic>i</italic>)</sub> in the regression model are estimated by the EM algorithm. The specific process is similar to Step 1.</p>
<p>Since matrix <italic><bold>Y</bold></italic> has <italic>n</italic> rows, we need to fit <italic>n</italic> GLMs to obtain <italic>n</italic> rows of <italic><bold>F</bold></italic><sup><italic>new</italic></sup>. However, the proposed model assumes that the &#x003C4; and &#x003B1; remain the same across all <italic>n</italic> different GLMs. Therefore, similar to step 1, we combine <italic><bold>y</bold></italic><sub>(1)</sub>, &#x02026;, <italic><bold>y</bold></italic><sub>(<italic>n</italic>)</sub> into a column vector and solve all <italic>n</italic> models simultaneously to obtain the globally optimal &#x003C4; and &#x003B1; values.</p>
<p><italic><bold>Step 3</bold></italic>: Apply the singular value decomposition (SVD) method to the <italic><bold>F</bold></italic><sup><italic>new</italic></sup><italic><bold>L</bold></italic><sup><italic>newT</italic></sup> to obtain a new <italic><bold>F</bold></italic><sup><italic>old</italic></sup>, and repeat the above alternating algorithm until convergence.</p>
<p>When the percentage of total likelihood difference between two iterations is less than a certain small value, the algorithm terminates; otherwise, we continue to update <italic><bold>F</bold></italic>, <italic><bold>L</bold></italic>, &#x003C4; and &#x003B1; until convergence. In the ZIGP regression step, we will use the EM algorithm to estimate the parameters. Therefore, the likelihood increases due to the nature of the EM algorithm used in regression estimation (Dempster et al., <xref ref-type="bibr" rid="B11">1977</xref>; Wu, <xref ref-type="bibr" rid="B64">1983</xref>). The likelihood remains the same in the SVD step. Overall, the algorithm is guaranteed to converge. In the Step 3, we apply SVD to <italic><bold>F</bold></italic><sup><italic>new</italic></sup><italic><bold>L</bold></italic><sup><italic>newT</italic></sup>, which ensures the uniqueness and the orthogonality of the updated components. We briefly summarize the alternating maximum likelihood algorithm under the GLM framework in the &#x0201C;<xref ref-type="fig" rid="F6">Algorithm 1</xref>&#x0201D; box.</p>
<fig id="F6" position="float">
<label>Algorithm 1</label>
<caption><p>GZIGPFA algorithm.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-15-1394204-g0006.tif"/>
</fig>
</sec>
<sec>
<title>2.3 Rank estimation</title>
<p>We use the <italic>N</italic>-fold cross-validation suggested by Li et al. (<xref ref-type="bibr" rid="B30">2018</xref>) to determine the optimal number of factors, i.e., the rank <italic>K</italic> of model (<xref ref-type="disp-formula" rid="E5">Equation 4</xref>). The idea is to randomly divide the entries of a data matrix into <italic>N</italic> non-overlapping parts. We systematically exclude one block at a time and utilize the remaining data to estimate the unknown parameters with varying ranks. Subsequently, we compute the likelihood of the model using the data of the excluded block. Finally, we sum up the likelihood of all <italic>N</italic> folds to obtain the total cross-validation (CV) likelihood of the model with rank <italic>k</italic> and calculate the CV likelihood for every rank <italic>k</italic>. The rank that provides the maximum CV likelihood is chosen as the optimal rank. The procedure of rank selection is briefly summarized in the &#x0201C;<xref ref-type="fig" rid="F7">Algorithm 2</xref>&#x0201D; box.</p>
<fig id="F7" position="float">
<label>Algorithm 2</label>
<caption><p><italic>N</italic>-fold cross-validation for rank estimation.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-15-1394204-g0007.tif"/>
</fig>
</sec>
</sec>
<sec id="s3">
<title>3 Simulation studies</title>
<p>The performance of our proposed GZIGPFA method is demonstrated through a simulation study. We compare the GZIGPFA method with four other methods:</p>
<list list-type="bullet">
<list-item><p>ZIPFA (Zero-inflated Poisson factor analysis): This method uses a zero-inflated Poisson factor analysis model for reducing the dimension of the microbiome data while accommodating the zero-inflated nature of the data (Xu et al., <xref ref-type="bibr" rid="B66">2020</xref>).</p></list-item>
<list-item><p>log-PCA (log-principal component analysis): The data is preprocessed by replacing all zeros with a small value and then taking a logarithm of the transformed data. After that, the data is processed by performing a principal component analysis (PCA).</p></list-item>
<list-item><p>PSVDOS (Poisson Singular Value Decomposition with Offset): The method is an efficient algorithm for estimating the Poisson factor model, which addresses the issue of sample normalization through the use of unknown offset parameters (Lee et al., <xref ref-type="bibr" rid="B28">2013</xref>).</p></list-item>
<list-item><p>GOMMS (GLM-based ordination method for microbiome samples): This method uses a zero-inflated quasi-Poisson factor model, which accounts for characteristics of microbiome data (e.g., highly skewed non-negative counts with excessive zeros) while reducing dimensionality (Sohn and Li, <xref ref-type="bibr" rid="B53">2017</xref>).</p></list-item>
</list>
<sec>
<title>3.1 Simulation design</title>
<p>We followed the design of Xu et al. (<xref ref-type="bibr" rid="B66">2020</xref>) to simulate microbiome datasets. A sequence data of <italic>n</italic> samples and <italic>m</italic> taxa is generated according to model (<xref ref-type="disp-formula" rid="E5">Equation 4</xref>). We simulate <italic>n</italic> = 200 different samples measured on <italic>m</italic> = 100 taxa. The rate matrix <bold>&#x0039B;</bold> follows: log(<bold>&#x0039B;</bold>) &#x0003D; <italic><bold>FL</bold></italic><sup><italic>T</italic></sup>, where the <italic><bold>F</bold></italic>&#x02208;&#x0211D;<sup><italic>n</italic>&#x000D7;3</sup> is a left singular vector matrix, and <italic><bold>L</bold></italic>&#x02208;&#x0211D;<sup><italic>m</italic>&#x000D7;3</sup> is a right singular vector matrix. To generate matrix <italic><bold>F</bold></italic>, we create a 200-by-3 matrix <italic><bold>F</bold></italic> such that:</p>
<p>Column 1: <italic>F</italic>(36:80, 1) &#x0003D; 2, <italic>F</italic>(81:140, 1) &#x0003D; 1.7,</p>
<p>Column 2: <italic>F</italic>(1:35, 2) &#x0003D; 1.8, <italic>F</italic>(36:80, 2) &#x0003D; 0.9,</p>
<p>Column 3: <italic>F</italic>(1:35, 3) &#x0003D; 1.7, <italic>F</italic>(36:200, 2) &#x0003D; 0,</p>
<p>with all the other entries being 0, and then jitter all the entries by adding random noises generated from <italic>N</italic>(0, 0.06<sup>2</sup>). Similarly, to generate matrix <italic><bold>L</bold></italic>, we create a 100-by-3 matrix <italic><bold>L</bold></italic> such that:</p>
<p>Column 1: <italic>L</italic>(1:60, 1) &#x0003D; 0, <italic>L</italic>(61:100, 1) &#x0003D; 1.7,</p>
<p>Column 2: <italic>L</italic>(36:60, 2) &#x0003D; 1.7, <italic>L</italic>(61:100, 2) &#x0003D; 1,</p>
<p>Column 3: <italic>L</italic>(1:25, 3) &#x0003D; 1.7, <italic>L</italic>(26:100, 2) &#x0003D; 0.9,</p>
<p>with all the other entries being 0, and then jitter all the entries by adding random noises generated from <italic>N</italic>(0, 0.05<sup>2</sup>). <xref ref-type="fig" rid="F1">Figure 1A</xref> displays the heatmap of the true log(<bold>&#x0039B;</bold>) matrix, and the three columns of <italic><bold>F</bold></italic> and <italic><bold>L</bold></italic> are shown in the columns of <xref ref-type="fig" rid="F1">Figures 1B</xref>, <xref ref-type="fig" rid="F1">C</xref>, respectively. Each row in <xref ref-type="fig" rid="F1">Figure 1B</xref> corresponds to one sample, and <xref ref-type="fig" rid="F1">Figure 1C</xref> shows the heatmap of the right singular vector matrix <italic><bold>L</bold></italic>, in which each row indicates one taxon profile.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Plots of simulation parameters. <bold>(A)</bold> The heatmap of true log(<bold>&#x0039B;</bold>) matrix. The rows represent samples, and the columns represent taxa. <bold>(B)</bold> True left singular vector matrix <italic><bold>F</bold></italic>. The rows correspond to the samples, and the columns denote factors. <bold>(C)</bold> True right singular vector matrix <italic><bold>L</bold></italic>. The rows correspond to the taxa, and the columns denote factors.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-15-1394204-g0001.tif"/>
</fig>
<p>After the matrices <italic><bold>F</bold></italic> and <italic><bold>L</bold></italic> are generated, <bold>&#x0039B;</bold> can be obtained according to <bold>&#x0039B;</bold> &#x0003D; exp(<italic><bold>FL</bold></italic><sup><italic>T</italic></sup>). Next, a zero-inflated sequencing matrix <italic><bold>Y</bold></italic> was generated from the following ZIGP model,</p>
<disp-formula id="E9"><mml:math id="M27"><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mo>&#x003BB;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>G</mml:mi><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>&#x003BB;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:math></disp-formula>
<p>where &#x003BB;<sub><italic>ij</italic></sub> is an element of the matrix <bold>&#x0039B;</bold>; the scaling parameter <italic>T</italic><sub><italic>i</italic></sub> and the dispersion parameter &#x003B1; were set to 1 and 0.2, respectively; the probability of excess zero <italic>&#x003D5;</italic><sub><italic>ij</italic></sub> is obtained by establishing the link between <italic>&#x003D5;</italic><sub><italic>ij</italic></sub> and &#x003BB;<sub><italic>ij</italic></sub>. Firstly, we consider the scenario with the relationship between <italic>&#x003D5;</italic><sub><italic>ij</italic></sub> and &#x003BB;<sub><italic>ij</italic></sub> established in Section 2, that is,</p>
<list list-type="bullet">
<list-item><p><italic>Scenario</italic> 1: logit(<italic>&#x003D5;</italic><sub><italic>ij</italic></sub>) &#x0003D; &#x02212;&#x003C4; log(&#x003BB;<sub><italic>ij</italic></sub>).</p></list-item>
</list>
<p>Furthermore, to more comprehensively evaluate the robustness of the proposed method, we further considered generating sequencing data under several misspecified scenarios. First, consider two common links to <italic>&#x003D5;</italic><sub><italic>ij</italic></sub> and &#x003BB;<sub><italic>ij</italic></sub>, which are mentioned in Lambert (<xref ref-type="bibr" rid="B27">1992</xref>) besides Scenario 1:</p>
<list list-type="bullet">
<list-item><p><italic>Scenario</italic> 2: log{&#x02212; log(<italic>&#x003D5;</italic><sub><italic>ij</italic></sub>)} &#x0003D; &#x003C4; log(&#x003BB;<sub><italic>ij</italic></sub>).</p></list-item>
<list-item><p><italic>Scenario</italic> 3: log{&#x02212; log(1 &#x02212; <italic>&#x003D5;</italic><sub><italic>ij</italic></sub>)} &#x0003D; &#x003C4; log(&#x003BB;<sub><italic>ij</italic></sub>).</p></list-item>
</list>
<p>In addition, we set up a scenario along the lines in Sohn and Li (<xref ref-type="bibr" rid="B53">2017</xref>), namely, each taxon has a fixed probability <italic>&#x003D5;</italic><sub><italic>j</italic></sub> independent of the &#x003BB;<sub><italic>ij</italic></sub>:</p>
<list list-type="bullet">
<list-item><p><italic>Scenario</italic> 4: <italic>&#x003D5;</italic><sub><italic>j</italic></sub> &#x0007E; Uniform(&#x003C4; &#x02212; 0.10, &#x003C4; &#x0002B; 0.10).</p></list-item>
</list>
<p>Finally, considering that actual microbiome data may come from different distributions, we set up two misspecified scenarios for generating microbiome data from other distributions, that is, the data come from the ZIP and ZINB distributions, and the relationship between the <italic>&#x003D5;</italic><sub><italic>ij</italic></sub> and &#x003BB;<sub><italic>ij</italic></sub> follows the setup in Scenario 1:</p>
<list list-type="bullet">
<list-item><p><italic>Scenario</italic> 5: <italic>y</italic><sub><italic>ij</italic></sub> &#x0007E; ZIP distribution, and logit(<italic>&#x003D5;</italic><sub><italic>ij</italic></sub>) &#x0003D; &#x02212;&#x003C4; log(&#x003BB;<sub><italic>ij</italic></sub>).</p></list-item>
<list-item><p><italic>Scenario</italic> 6: <italic>y</italic><sub><italic>ij</italic></sub> &#x0007E; ZINB distribution, and logit(<italic>&#x003D5;</italic><sub><italic>ij</italic></sub>) &#x0003D; &#x02212;&#x003C4; log(&#x003BB;<sub><italic>ij</italic></sub>).</p></list-item>
</list>
<p>We evaluated the simulation results for all the scenarios above at light zero inflation (20%) and higher zero inflation (40%), respectively.</p>
</sec>
<sec>
<title>3.2 Simulation results</title>
<p>First, the performance of the proposed method for rank estimation in all scenarios is examined. A 10-fold cross-validation is performed on the data generated in each scenario separately to compute the cross-validation likelihood for different ranks. The rank estimation results of the GZIGPFA method for all simulation scenarios are displayed in <xref ref-type="fig" rid="F2">Figure 2</xref>. It can be seen from <xref ref-type="fig" rid="F2">Figure 2</xref> that the proposed method provides the maximum CV likelihood with rank 3 in all simulation scenarios. <xref ref-type="fig" rid="F2">Figure 2</xref> shows that the proposed method is accurate in the rank estimation under the given model [Model (<xref ref-type="disp-formula" rid="E5">Equation 4</xref>)] and performs well under the misspecified scenarios, indicating the robustness of the GZIGPFA method.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Cross-validation to choose the rank in the GZIGPFA model. <bold>(A)</bold> The CV likelihood of the GZIGPFA model under scenarios 14 with a 20% zero-inflated probability. <bold>(B)</bold> The CV likelihood of the GZIGPFA model under scenarios 14 with a 40% zero-inflated proportion. <bold>(C)</bold> The CV likelihood of the GZIGPFA model under scenarios 5 and 6 with 20% and 40% zero-inflated proportions. The proposed method provides maximum CV likelihoods with rank 3 under all simulation scenarios.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-15-1394204-g0002.tif"/>
</fig>
<p>Next, a comprehensive comparison of the GZIGPFA method with other approaches (ZIPFA, log-PCA, PSVDOS, and GOMMS) is presented to illustrate the superior performance of the proposed method in depth. For each simulation scenario, 200 replicates are performed. The Frobenius norm of the error matrix (denoted as loss value) is utilized to evaluate the effectiveness of each method. The loss values of several methods in all simulation scenarios are listed in <xref ref-type="table" rid="T1">Table 1</xref>. <xref ref-type="table" rid="T1">Table 1</xref> shows that the GZIGPFA method has a small loss in most simulation scenarios, indicating that the proposed method is effective. In Scenarios (1)&#x02013;(4), the performance of all four methods is significantly better than the GOMMS method. In addition, we find that the convergence effect of the GOMMS method is poor when there are more zeros in the data. In scenario (5), ZIPFA outperforms GZIGPFA when the zero percentage is high (40%) because this scenario essentially favors ZIPFA by using the ZIP model. In Scenario (6), the data is generated by the ZINB model, and the PSVDOS method performs the worst among the five methods because this method cannot consider over-dispersed and zero-inflated data. In addition, the GOMMS method performs second only to GZIGPFA and ZIPFA in Scenario (6) because GOMMS is based on the zero-inflated quasi-Poisson, which is intrinsically closer to ZINB. Overall, our method performs better than competing methods, even in the misspecified scenarios.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>The mean of loss values and standard errors (in the parenthesis) for the five methods under different scenarios.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Scenario</bold></th>
<th valign="top" align="center"><bold>Zero (%)</bold></th>
<th valign="top" align="center"><bold>GZIGPFA</bold></th>
<th valign="top" align="center"><bold>ZIPFA</bold></th>
<th valign="top" align="center"><bold>log-PCA</bold></th>
<th valign="top" align="center"><bold>PSVDOS</bold></th>
<th valign="top" align="center"><bold>GOMMS</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Scenario (1)</td>
<td valign="top" align="center">20%</td>
<td valign="top" align="center"><bold>6.9878</bold> (1.4968)</td>
<td valign="top" align="center">10.7800 (1.8965)</td>
<td valign="top" align="center">10.2970 (0.1332)</td>
<td valign="top" align="center">26.4999 (0.0594)</td>
<td valign="top" align="center">92.5108 (207.740)</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">40%</td>
<td valign="top" align="center"><bold>12.7542</bold> (7.1569)</td>
<td valign="top" align="center">13.2471 (2.9255)</td>
<td valign="top" align="center">16.7528 (0.1516)</td>
<td valign="top" align="center">26.6175 (0.0843)</td>
<td valign="top" align="center">102.689 (238.263)</td>
</tr>
<tr>
<td valign="top" align="left">Scenario (2)</td>
<td valign="top" align="center">20%</td>
<td valign="top" align="center"><bold>8.9129</bold> (3.2038)</td>
<td valign="top" align="center">10.8945 (2.0912)</td>
<td valign="top" align="center">10.7148 (0.1366)</td>
<td valign="top" align="center">26.5064 (0.0581)</td>
<td valign="top" align="center">87.5358 (166.145)</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">40%</td>
<td valign="top" align="center"><bold>10.2343</bold> (4.8788)</td>
<td valign="top" align="center">13.5586 (3.3809)</td>
<td valign="top" align="center">17.7686 (0.1339)</td>
<td valign="top" align="center">26.7447 (0.1101)</td>
<td valign="top" align="center">91.5380 (144.401)</td>
</tr>
<tr>
<td valign="top" align="left">Scenario (3)</td>
<td valign="top" align="center">20%</td>
<td valign="top" align="center"><bold>7.1095</bold> (1.2251)</td>
<td valign="top" align="center">10.8162 (1.8302)</td>
<td valign="top" align="center">10.0293 (0.1294)</td>
<td valign="top" align="center">26.5015 (0.0644)</td>
<td valign="top" align="center">99.1790 (183.318)</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">40%</td>
<td valign="top" align="center">21.5684 (8.9917)</td>
<td valign="top" align="center"><bold>12.1517</bold> (2.5438)</td>
<td valign="top" align="center">15.8540 (0.1635)</td>
<td valign="top" align="center">26.5817 (0.0819)</td>
<td valign="top" align="center">114.108 (243.236)</td>
</tr>
<tr>
<td valign="top" align="left">Scenario (4)</td>
<td valign="top" align="center">20%</td>
<td valign="top" align="center"><bold>10.2995</bold> (0.9843)</td>
<td valign="top" align="center">11.5437 (2.1585)</td>
<td valign="top" align="center">12.0547 (0.2186)</td>
<td valign="top" align="center">26.5496 (0.0694)</td>
<td valign="top" align="center">78.4083 (114.947)</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">40%</td>
<td valign="top" align="center"><bold>9.8594</bold> (4.7380)</td>
<td valign="top" align="center">13.0524 (3.0187)</td>
<td valign="top" align="center">17.5302 (0.2297)</td>
<td valign="top" align="center">26.7168 (0.0957)</td>
<td valign="top" align="center">100.821 (241.874)</td>
</tr>
<tr>
<td valign="top" align="left">Scenario (5)</td>
<td valign="top" align="center">20%</td>
<td valign="top" align="center"><bold>1.9069</bold> (0.0882)</td>
<td valign="top" align="center">2.0074 (0.0686)</td>
<td valign="top" align="center">5.7243 (0.1271)</td>
<td valign="top" align="center">26.2521 (0.0052)</td>
<td valign="top" align="center">4.7897 (0.0036)</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">40%</td>
<td valign="top" align="center">3.9383 (0.2266)</td>
<td valign="top" align="center"><bold>3.2860</bold> (0.2841)</td>
<td valign="top" align="center">13.0891 (0.1697)</td>
<td valign="top" align="center">27.2165 (0.2778)</td>
<td valign="top" align="center">10.5189 (0.0140)</td>
</tr>
<tr>
<td valign="top" align="left">Scenario (6)</td>
<td valign="top" align="center">20%</td>
<td valign="top" align="center"><bold>2.9830</bold> (0.1168)</td>
<td valign="top" align="center">3.3560 (0.2495)</td>
<td valign="top" align="center">6.4711 (0.1243)</td>
<td valign="top" align="center">26.2505 (0.0030)</td>
<td valign="top" align="center">6.0003 (0.0013)</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">40%</td>
<td valign="top" align="center"><bold>4.4550</bold> (0.3200)</td>
<td valign="top" align="center">4.5013 (0.5164)</td>
<td valign="top" align="center">13.7113 (0.1683)</td>
<td valign="top" align="center">26.6065 (0.3625)</td>
<td valign="top" align="center">10.0941 (0.0107)</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The best results in each setting are in boldface.</p>
</table-wrap-foot>
</table-wrap>
<p>Finally, we show the heatmaps of the true log(<bold>&#x0039B;</bold>) and the estimated log(<bold>&#x0039B;</bold>) of several methods in <xref ref-type="fig" rid="F3">Figure 3</xref> to visualize the performance of the GZIGPFA method, and we also display the clustering effects of several methods at the taxa (top) and sample (left side) levels. Since the GOMMS method performs poorly in <xref ref-type="table" rid="T1">Table 1</xref>, only the estimation and clustering results of the four methods GZIGPFA, ZIPFA, log-PCA, and PSVDOS are presented in <xref ref-type="fig" rid="F3">Figure 3</xref>. Panel (a) in <xref ref-type="fig" rid="F3">Figures 3A</xref>, <xref ref-type="fig" rid="F3">B</xref> displays the true log(<bold>&#x0039B;</bold>) used in the simulation. The phylogenetic tree on the left side of the heatmap shows the clustering of the sample, which falls into four clusters. Similarly, the phylogenetic tree above the heatmap shows the clustering of the taxa. The clustering pattern is obtained by applying the complete linkage hierarchical clustering analysis to <italic><bold>F</bold></italic> and <italic><bold>L</bold></italic> (Wilkinson and Friendly, <xref ref-type="bibr" rid="B62">2009</xref>). <xref ref-type="fig" rid="F3">Figures 3A</xref>, <xref ref-type="fig" rid="F3">B</xref> show the estimation and clustering effects of the four methods when the zero-inflated proportion is low (20%) and high (40%) in Scenario (1), respectively. GZIGPFA method (Panel b in <xref ref-type="fig" rid="F3">Figures 3A</xref>, <xref ref-type="fig" rid="F3">B</xref>) offers the best approximation to the true signal, and it gives the accurate clustering result, which is as expected, as the dataset was designed in a way that takes advantage of the unique features of GZIGPFA. The log(<bold>&#x0039B;</bold>) estimated by the log-PCA method (Panel d in <xref ref-type="fig" rid="F3">Figures 3A</xref>, <xref ref-type="fig" rid="F3">B</xref>) is far from the true value (Panel a) and is the worst performer among several methods. Meanwhile, log-PCA fails to capture the right sample clustering when the zero-inflated proportion goes from 20 to 40%. The log-PCA performs poorly overall because it does not consider the underlying distribution and excessive zeros.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>The heatmap of true log(<bold>&#x0039B;</bold>) and the estimated log(<bold>&#x0039B;</bold>) from different methods in Scenario 1. The phylogenetic tree on the top and left show the clustering of taxa and samples, respectively. <bold>(A)</bold> The zero-inflated proportion is 20%. <bold>(B)</bold> The zero-inflated proportion is 40%. The rows and columns of all heatmaps represent samples and taxa, respectively.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-15-1394204-g0003.tif"/>
</fig>
</sec>
</sec>
<sec id="s4">
<title>4 Application to the gut microbiome data</title>
<p>Empirical research has demonstrated that mice and humans harbor similar microbiota at high taxonomic levels (Ley et al., <xref ref-type="bibr" rid="B29">2008</xref>; Krych et al., <xref ref-type="bibr" rid="B24">2013</xref>). Therefore, laboratory mice can be used to simulate the human gut environment for experiments and to explore the mechanisms of host-microbial interactions in a data-driven manner when studying human gut microbes. In this section, the GZIGPFA model is applied to the mouse gut microbial dataset (Sun et al., <xref ref-type="bibr" rid="B54">2019</xref>) to explore the association between gut microbes and obesity. Microbial datasets were extracted from 48 male mice. The mice were divided into the blank group, high-fat control group, and probiotic experimental group, in which the blank group was fed normal chow, the high-fat control group and the probiotic experimental group were fed a high-fat diet for 4 weeks to establish an obesity model for the mice, and the probiotic experimental group was fed high-fat chow plus probiotic capsules starting from the 5th week of the successful modeling, while the high-fat control group continued to be fed high-fat chow. At the end of the 8th week, various indicators of the mice were measured, including weight, body length, total cholesterol, endotoxin, etc. The samples were first amplified with a set of primers targeting the 16S rDNA V4 region. Then, the original data were subjected to operational taxonomic unit (OTU) clustering and species classification analysis based on valid data. According to the results of OTU clustering, species annotation was performed for the representative sequences of each OTU, and the corresponding species information and species-based abundance were obtained. Then, we reduced the dimensionality of the dataset with the proposed GZIGPFA method to extract the common factors and further explore the association between the common factors and obesity.</p>
<p>We selected body weight, total cholesterol, and endotoxin as three responses from the measured indicators of mice, where the weight of mice can intuitively reflect the degree of obesity. Obesity caused by a high-fat diet is often accompanied by hyperlipidemia, and total cholesterol (TC) is widely employed clinically as an indicator for measuring blood lipids. Endotoxin, also known as lipopolysaccharide (LPS), is a critical factor in the systemic inflammatory reaction. When the intestinal microbiota is imbalanced and harmful bacteria increase, the body is susceptible to endotoxemia, and sustained low-level endotoxemia is the leading cause of obesity and metabolic disorders. Therefore, we will focus on the relationship between the gut microbial community and three responses (weight, TC, and LPS).</p>
<p>We applied 10-fold cross-validation on microbial abundance data. <xref ref-type="fig" rid="F4">Figure 4</xref> shows that CV likelihood reaches the maximum point at a rank equal to 3, so we will use three factors in the following analysis. We performed GZIGPFA fitting with a rank of 3 on the microbiome data. The algorithm converged after 6 iterations and obtained score matrix estimates (<italic><bold>F</bold></italic>) and loading matrix estimates (<italic><bold>L</bold></italic>). We can compute log(<bold>&#x0039B;</bold>) &#x0003D; <italic><bold>FL</bold></italic><sup><italic>T</italic></sup> according to the estimates of <italic><bold>F</bold></italic> and <italic><bold>L</bold></italic>, and the zero-inflated probability matrix <bold>&#x003A6;</bold> can be obtained through the relationship between <bold>&#x0039B;</bold> and <bold>&#x003A6;</bold> assumed in Section 2.1 [i.e., logit(<bold>&#x003A6;</bold>) &#x0003D; &#x02212;&#x003C4;log(<bold>&#x0039B;</bold>)]. The total probability of zero for each count is estimated as <inline-formula><mml:math id="M28"><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x003BB;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>/</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x0002B;</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x003BB;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:math></inline-formula>. We reorder the total zero probability matrix and plot the corresponding heatmap (<xref ref-type="fig" rid="F5">Figure 5A</xref>). In <xref ref-type="fig" rid="F5">Figure 5A</xref>, the bottom right indicates the large values of total zero probability (red points), and the small total zero probability values are sorted to the top left (blue points). Meanwhile, the rearrangement of the true data is plotted in <xref ref-type="fig" rid="F5">Figure 5B</xref>, where non-zero values are shown in the top left (blue points) and zeros in the bottom right (white points). We compare the predicted probability of zeros with the distribution of zeros in the real data. The significant similarity between the red part in <xref ref-type="fig" rid="F5">Figure 5A</xref> and the white part in <xref ref-type="fig" rid="F5">Figure 5B</xref> shows that the proposed method captures the structure of excess zeros well.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Cross validation to choose the number of factors in real data analysis.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-15-1394204-g0004.tif"/>
</fig>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Comparison of predicted probability of zeros and real zero distribution in the dataset. <bold>(A)</bold> The heatmap of predicted zero probability. <bold>(B)</bold> The heatmap of the binary real data value. Blue points are non-zero values and white points are zeros. The rows and columns of both heatmaps represent samples and taxa, respectively.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-15-1394204-g0005.tif"/>
</fig>
<p>To determine the association between the three factors obtained through GZIGPFA dimensionality reduction and the three responses (weight, TC, and LPS), a linear model was fitted in which each response was regressed on all three factors, respectively. The <italic>p</italic>-values corresponding to different factors and responses are listed in <xref ref-type="table" rid="T2">Table 2</xref>. In addition, we demonstrate the results of the other comparison methods (ZIPFA, log-PCA, PSVDOS, and GOMMS) introduced in Section 3. It can be observed in <xref ref-type="table" rid="T2">Table 2</xref> that the GZIGPFA and log-PCA methods can identify factors significantly associated with each response, while ZIPFA and PSVDOS failed to provide significant factors for TC and LPS. In addition, GOMMS is also unable to find factors associated with LPS. In particular, all five methods identified factors (factors 2 or 3) associated with weight, indicating that gut microbiome composition may be an essential factor influencing weight. Furthermore, factor 2 was a significant predictor of all responses in our proposed method, suggesting a potential link between obesity diseases and gut microbiome.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>The <italic>P</italic>-values corresponding to different factors and response variables in different models.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Response</bold></th>
<th valign="top" align="center"><bold>Factor</bold></th>
<th valign="top" align="center"><bold>GZIGPFA</bold></th>
<th valign="top" align="center"><bold>ZIPFA</bold></th>
<th valign="top" align="center"><bold>log-PCA</bold></th>
<th valign="top" align="center"><bold>PSVDOS</bold></th>
<th valign="top" align="center"><bold>GOMMS</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td/>
<td valign="top" align="center">Factor 1</td>
<td valign="top" align="center">0.2874</td>
<td valign="top" align="center">0.3727</td>
<td valign="top" align="center">0.3551</td>
<td valign="top" align="center">0.5600</td>
<td valign="top" align="center">0.0810</td>
</tr>
 <tr>
<td valign="top" align="left">Weight</td>
<td valign="top" align="center">Factor 2</td>
<td valign="top" align="center">0.0004<sup>&#x0002A;&#x0002A;&#x0002A;</sup></td>
<td valign="top" align="center">0.4077</td>
<td valign="top" align="center">0.0003<sup>&#x0002A;&#x0002A;&#x0002A;</sup></td>
<td valign="top" align="center">0.0482<sup>&#x0002A;</sup></td>
<td valign="top" align="center">0.9480</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Factor 3</td>
<td valign="top" align="center">0.0281<sup>&#x0002A;</sup></td>
<td valign="top" align="center">0.0008<sup>&#x0002A;&#x0002A;&#x0002A;</sup></td>
<td valign="top" align="center">0.0493<sup>&#x0002A;</sup></td>
<td valign="top" align="center">0.1706</td>
<td valign="top" align="center">0.0260<sup>&#x0002A;</sup></td>
</tr>
<tr>
<td/>
<td valign="top" align="center">Factor 1</td>
<td valign="top" align="center">0.1243</td>
<td valign="top" align="center">0.2850</td>
<td valign="top" align="center">0.5406</td>
<td valign="top" align="center">0.8740</td>
<td valign="top" align="center">0.0684</td>
</tr>
 <tr>
<td valign="top" align="left">LPS</td>
<td valign="top" align="center">Factor 2</td>
<td valign="top" align="center">0.0171<sup>&#x0002A;</sup></td>
<td valign="top" align="center">0.2530</td>
<td valign="top" align="center">0.0092<sup>&#x0002A;&#x0002A;</sup></td>
<td valign="top" align="center">0.1460</td>
<td valign="top" align="center">0.4635</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Factor 3</td>
<td valign="top" align="center">0.7297</td>
<td valign="top" align="center">0.1050</td>
<td valign="top" align="center">0.7876</td>
<td valign="top" align="center">0.660</td>
<td valign="top" align="center">0.2885</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">Factor 1</td>
<td valign="top" align="center">0.4014</td>
<td valign="top" align="center">0.7030</td>
<td valign="top" align="center">0.8232</td>
<td valign="top" align="center">0.3290</td>
<td valign="top" align="center">0.0099<sup>&#x0002A;&#x0002A;</sup></td>
</tr>
 <tr>
<td valign="top" align="left">TC</td>
<td valign="top" align="center">Factor 2</td>
<td valign="top" align="center">0.0076<sup>&#x0002A;&#x0002A;</sup></td>
<td valign="top" align="center">0.5972</td>
<td valign="top" align="center">0.0035<sup>&#x0002A;&#x0002A;</sup></td>
<td valign="top" align="center">0.1430</td>
<td valign="top" align="center">0.9905</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Factor 3</td>
<td valign="top" align="center">0.7657</td>
<td valign="top" align="center">0.0504</td>
<td valign="top" align="center">0.5464</td>
<td valign="top" align="center">0.6070</td>
<td valign="top" align="center">0.1704</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">Factor 1</td>
<td valign="top" align="center">0.6075</td>
<td valign="top" align="center">0.0705</td>
<td valign="top" align="center">0.6760</td>
<td valign="top" align="center">0.7870</td>
<td valign="top" align="center">0.7837</td>
</tr>
 <tr>
<td valign="top" align="left">TNF-&#x003B1;</td>
<td valign="top" align="center">Factor 2</td>
<td valign="top" align="center">0.7743</td>
<td valign="top" align="center">0.0873</td>
<td valign="top" align="center">0.6760</td>
<td valign="top" align="center">0.4050</td>
<td valign="top" align="center">0.5351</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">Factor 3</td>
<td valign="top" align="center">0.9345</td>
<td valign="top" align="center">0.4159</td>
<td valign="top" align="center">0.5480</td>
<td valign="top" align="center">0.3520</td>
<td valign="top" align="center">0.2076</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p><sup>&#x0002A;&#x0002A;&#x0002A;</sup>, <sup>&#x0002A;&#x0002A;</sup>, and <sup>&#x0002A;</sup> denote the significance level takes 0.001, 0.01, 0.05, respectively.</p>
</table-wrap-foot>
</table-wrap>
<p>We consulted the literature to explain the factors obtained by GZIGPFA. By searching for gut microbes and obesity keywords on PubMed, some review articles were screened to identify microbes related to obesity. To identify microbes associated with obesity by searching for gut microbes and obesity keywords on Pubmed and filtering some review articles. After a full-text review of 116 papers, Pinart et al. (<xref ref-type="bibr" rid="B47">2021</xref>) concluded that <italic>Firmicutes</italic> and <italic>Bacteroidetes</italic> are the two microorganisms that mainly affect obesity at the phylum level. Therefore, factors 2 and 3 significantly associated with the obesity phenotype in the association analysis may be summarized as <italic>Firmicutes</italic> and <italic>Bacteroidetes</italic>. In conclusion, the proposed method can help the experimenter to determine the approximate factors affecting the experiment in advance.</p>
<p>Finally, in order to demonstrate that the proposed model does the absence of over-prediction problems, we additionally selected a response for analysis, i.e., tumor necrosis factor-alpha (TNF-&#x003B1;), which is not directly related to obesity. As can be seen from <xref ref-type="table" rid="T2">Table 2</xref>, all these methods did not identify factors significantly related to TNF-&#x003B1;, indicating that there is no over-prediction problem.</p>
</sec>
<sec sec-type="discussion" id="s5">
<title>5 Discussion</title>
<p>Dimensionality reduction is a prevalent preprocessing step in high dimensional microbiome analysis. In this paper, we propose a new GLM-based zero-inflated generalized Poisson factor analysis model to analyze high-dimensional microbiome count data. This method explores the correlation between microbial taxa and response variables, and focuses on selecting a few common factors that summarize the majority of variable information, thus one can mitigate the high dimensionality problem and the computational expenses. The GZIGPFA model can simultaneously consider the zero-inflation, over-dispersion, and high-dimensional characteristics of microbial data. Meanwhile, the model directly models absolute abundance data, avoiding the problem of information loss during data conversion. We establish a link function between generalized Poisson expectation and true zero probability within the GLM framework, and perform parameter estimation using the alternating maximum likelihood algorithm. The rank of the model was determined via cross-validation method. In addition, we performed simulation studies under different scenarios and compared the GZIGPFA method with existing methods to validate the performance of the proposed method. In the analysis of gut microbiome data, the proposed method identified microorganisms significantly associated with obesity.</p>
<p>The novelty of the GZIGPFA method is reflected in the combination of the ZIGP model and the factor analysis model, which provides more possibilities for future microbial-related analysis work. Furthermore, upon review of the existing literature pertaining to microbiome data analysis, our proposed approach represents the inaugural utilization of the zero-inflated generalized Poisson model in microbial datasets, which expands the methodological options of researchers for addressing complex microbiome datasets. In addition to microbiome data, the proposed method can be used for other count data such as micro RNA data, single-cell RNA-seq data, etc. In addition, other suitable models can be extended to the framework of this article to provide more statistical methods for the analysis of high-dimensional microbiome data in the future.</p>
<p>The work presented in this paper remains subject to certain limitations. In this paper, a cross-validation method is used for rank estimation, which is accompanied by a high computational cost, although the results have high accuracy. In future work, the process of rank estimation can be further optimized to improve computational efficiency. In addition, the GZIGPFA model proposed in this article can only extract common factors associated with obesity phenotypes from numerous microbial taxa. The meaning of common factors needs to be determined based on existing prior information, and the interpretation of the actual meaning of each factor is not absolute. We can further extend our approach to provide a more comprehensive tool for the analysis of microorganisms in the future. Finally, although the performance of the log-PCA method in real data analysis closely resembles that of our method, it employs a strategy of replacing zeros in the data with pseudo counts. However, there is no consensus on how to choose the pseudo count, and it has been shown that the choice of pseudo count can affect the conclusions of a microbiome analysis (Costea et al., <xref ref-type="bibr" rid="B9">2014</xref>; Paulson et al., <xref ref-type="bibr" rid="B43">2014</xref>). The gut microbiome data that we used in real data analysis contains &#x0007E;45% zeros, which is moderately zero-inflated. Perhaps the strategy of replacing zeros has less impact on the results, which may be the main reason why we did not show a clear advantage. Once a dataset shows a serious zero-inflated trend, the log-PCA method may become unstable. In the field of microbiology, it is common for microbiome data to be severely zero-inflated (Paulson et al., <xref ref-type="bibr" rid="B44">2013</xref>; Silverman et al., <xref ref-type="bibr" rid="B49">2020</xref>). Due to sharing restrictions on these data, we do not conduct a practical demonstration in this article.</p>
</sec>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The mouse gut microbiome dataset for the real data analysis section was obtained with the support of Professor Qingshen Sun. The datasets presented in this article are not readily available because they they have not been made publicly available by Sun et al. (<xref ref-type="bibr" rid="B54">2019</xref>). Requests to access these datasets should be directed to corresponding author JC, <email>jinlingchi_edu&#x00040;163.com</email>.</p>
</sec>
<sec sec-type="ethics-statement" id="s7">
<title>Ethics statement</title>
<p>The manuscript presents research on animals that do not require ethical approval for their study.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>JC: Software, Visualization, Writing&#x02014;original draft, Writing&#x02014;review &#x00026; editing, Methodology. JY: Funding acquisition, Writing&#x02014;review &#x00026; editing. YZ: Data curation, Funding acquisition, Writing&#x02014;review &#x00026; editing.</p>
</sec>
</body>
<back>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This research was supported by the National Natural Science Foundation of China (Grant No. 12071114) and Natural Science Basic Research Program of Shaanxi (Program No. 2024JC-YBMS-043).</p>
</sec>
<ack><p>The authors would like to thank the Qingshen Sun for providing the gut microbiome data.</p>
</ack>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s11">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fmicb.2024.1394204/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fmicb.2024.1394204/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_2.ZIP" id="SM2" mimetype="application/zip" xmlns:xlink="http://www.w3.org/1999/xlink"/></sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Anders</surname> <given-names>S.</given-names></name> <name><surname>Huber</surname> <given-names>W.</given-names></name></person-group> (<year>2010</year>). <article-title>Differential expression analysis for sequence count data</article-title>. <source>Nat. Prec</source>. <volume>2010</volume>:<fpage>1</fpage>. <pub-id pub-id-type="doi">10.1038/npre.2010.4282.1</pub-id></citation>
</ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Armstrong</surname> <given-names>G.</given-names></name> <name><surname>Rahman</surname> <given-names>G.</given-names></name> <name><surname>Martino</surname> <given-names>C.</given-names></name> <name><surname>McDonald</surname> <given-names>D.</given-names></name> <name><surname>Gonzalez</surname> <given-names>A.</given-names></name> <name><surname>Mishne</surname> <given-names>G.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Applications and comparison of dimensionality reduction methods for microbiome data</article-title>. <source>Front. Bioinformat</source>. <volume>2</volume>:<fpage>821861</fpage>. <pub-id pub-id-type="doi">10.3389/fbinf.2022.821861</pub-id><pub-id pub-id-type="pmid">36304280</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Badri</surname> <given-names>M.</given-names></name> <name><surname>Kurtz</surname> <given-names>Z. D.</given-names></name> <name><surname>Bonneau</surname> <given-names>R.</given-names></name> <name><surname>M&#x000FC;ller</surname> <given-names>C. L.</given-names></name></person-group> (<year>2020</year>). <article-title>Shrinkage improves estimation of microbial associations under different normalization methods</article-title>. <source>NAR Genom. Bioinformat</source>. <volume>2</volume>, <fpage>1</fpage>&#x02013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.1101/406264</pub-id><pub-id pub-id-type="pmid">33575644</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Campbell</surname> <given-names>H.</given-names></name></person-group> (<year>2021</year>). <article-title>The consequences of checking for zero&#x02013;inflation and overdispersion in the analysis of count data</article-title>. <source>Methods Ecol. Evol</source>. <volume>12</volume>, <fpage>665</fpage>&#x02013;<lpage>680</lpage>. <pub-id pub-id-type="doi">10.1111/2041-210x.13559</pub-id></citation>
</ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Carlisle</surname> <given-names>E. M.</given-names></name> <name><surname>Poroyko</surname> <given-names>V.</given-names></name> <name><surname>Caplan</surname> <given-names>M. S.</given-names></name> <name><surname>Alverdy</surname> <given-names>J.</given-names></name> <name><surname>Morowitz</surname> <given-names>M. J.</given-names></name> <name><surname>Liu</surname> <given-names>D.</given-names></name></person-group> (<year>2013</year>). <article-title>Murine gut microbiota and transcriptome are diet dependent</article-title>. <source>Ann. Surg</source>. <volume>257</volume>, <fpage>287</fpage>&#x02013;<lpage>294</lpage>. <pub-id pub-id-type="doi">10.1097/sla.0b013e318262a6a6</pub-id><pub-id pub-id-type="pmid">23001074</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name></person-group> (<year>2013</year>). <article-title>Variable selection for sparse dirichlet-multinomial regression with an application to microbiome data analysis</article-title>. <source>Ann. Appl. Stat</source>. <volume>7</volume>, <fpage>418</fpage>&#x02013;<lpage>442</lpage>. <pub-id pub-id-type="doi">10.1214/12-aoas592</pub-id><pub-id pub-id-type="pmid">24312162</pub-id></citation></ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chi</surname> <given-names>J.</given-names></name> <name><surname>Zhou</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>L.</given-names></name> <name><surname>Zhou</surname> <given-names>Y.</given-names></name></person-group> (<year>2020</year>). <article-title>Bayesian interval mapping of count trait loci based on zero&#x02013;inflated generalized poisson regression model</article-title>. <source>Biometr. J</source>. <volume>62</volume>, <fpage>1428</fpage>&#x02013;<lpage>1442</lpage>. <pub-id pub-id-type="doi">10.1002/bimj.201900274</pub-id><pub-id pub-id-type="pmid">32399977</pub-id></citation></ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Consul</surname> <given-names>P. C.</given-names></name> <name><surname>Famoye</surname> <given-names>F.</given-names></name></person-group> (<year>1992</year>). <article-title>Generalized poisson regression model</article-title>. <source>Commun. Stat</source>. <volume>21</volume>, <fpage>89</fpage>&#x02013;<lpage>109</lpage>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Costea</surname> <given-names>P. I.</given-names></name> <name><surname>Zeller</surname> <given-names>G.</given-names></name> <name><surname>Sunagawa</surname> <given-names>S.</given-names></name> <name><surname>Bork</surname> <given-names>P.</given-names></name></person-group> (<year>2014</year>). <article-title>A fair comparison</article-title>. <source>Nat. Methods</source> <volume>11</volume>:<fpage>359</fpage>. <pub-id pub-id-type="doi">10.1038/nmeth.2897</pub-id><pub-id pub-id-type="pmid">24681719</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cui</surname> <given-names>Y.</given-names></name> <name><surname>Yang</surname> <given-names>W.</given-names></name></person-group> (<year>2009</year>). <article-title>Zero-inflated generalized poisson regression mixture model for mapping quantitative trait loci underlying count trait with many zeros</article-title>. <source>J. Theoret. Biol</source>. <volume>256</volume>, <fpage>276</fpage>&#x02013;<lpage>285</lpage>. <pub-id pub-id-type="doi">10.1016/j.jtbi.2008.10.003</pub-id><pub-id pub-id-type="pmid">18977361</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dempster</surname> <given-names>A. P.</given-names></name> <name><surname>Laird</surname> <given-names>N. M.</given-names></name> <name><surname>Rubin</surname> <given-names>D. B.</given-names></name></person-group> (<year>1977</year>). <article-title>Maximum likelihood from incomplete data via the EM algorithm</article-title>. <source>J. Royal Stat. Soc. Ser. B</source> <volume>39</volume>, <fpage>1</fpage>&#x02013;<lpage>22</lpage>.</citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Eddy</surname> <given-names>S. R.</given-names></name></person-group> (<year>2011</year>). <article-title>Accelerated profile HMM searches</article-title>. <source>PLoS Comput. Biol</source>. <volume>7</volume>:<fpage>e1002195</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1002195</pub-id><pub-id pub-id-type="pmid">22039361</pub-id></citation></ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Egan</surname> <given-names>S.</given-names></name> <name><surname>Harder</surname> <given-names>T.</given-names></name> <name><surname>Burke</surname> <given-names>C.</given-names></name> <name><surname>Steinberg</surname> <given-names>P.</given-names></name> <name><surname>Kjelleberg</surname> <given-names>S.</given-names></name> <name><surname>Thomas</surname> <given-names>T.</given-names></name></person-group> (<year>2013</year>). <article-title>The seaweed holobiont: understanding seaweed&#x02013;bacteria interactions</article-title>. <source>FEMS Microbiol. Rev</source>. <volume>37</volume>, <fpage>462</fpage>&#x02013;<lpage>476</lpage>. <pub-id pub-id-type="doi">10.1111/1574-6976.12011</pub-id><pub-id pub-id-type="pmid">23157386</pub-id></citation></ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Famoye</surname> <given-names>F.</given-names></name></person-group> (<year>1993</year>). <article-title>Restricted generalized poisson regression model</article-title>. <source>Commun. Stat</source>. <volume>22</volume>, <fpage>1335</fpage>&#x02013;<lpage>1354</lpage>.</citation>
</ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Famoye</surname> <given-names>F.</given-names></name> <name><surname>Singh</surname> <given-names>K. P.</given-names></name></person-group> (<year>2006</year>). <article-title>Zero-inflated generalized poisson regression model with an application to domestic violence data</article-title>. <source>J. Data Sci</source>. <volume>4</volume>, <fpage>117</fpage>&#x02013;<lpage>130</lpage>. <pub-id pub-id-type="doi">10.6339/jds.2006.04(1).257</pub-id></citation>
</ref>
<ref id="B16">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Fan</surname> <given-names>Y.</given-names></name> <name><surname>Jiang</surname> <given-names>X.</given-names></name> <name><surname>Hu</surname> <given-names>X.</given-names></name> <name><surname>Song</surname> <given-names>B.</given-names></name> <name><surname>Ling</surname> <given-names>Y.</given-names></name> <name><surname>Wu</surname> <given-names>W.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;A novel dimensionality reduction algorithm based on laplace matrix for microbiome data analysis,&#x0201D;</article-title> in <source>2015 IEEE International Conference on Bioinformatics and Biomedicine (BIBM)</source>. <publisher-loc>Washington, DC</publisher-loc>: <publisher-name>IEEE</publisher-name>.</citation>
</ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Frank</surname> <given-names>D. N.</given-names></name> <name><surname>Amand</surname> <given-names>A. L. S.</given-names></name> <name><surname>Feldman</surname> <given-names>R. A.</given-names></name> <name><surname>Boedeker</surname> <given-names>E. C.</given-names></name> <name><surname>Harpaz</surname> <given-names>N.</given-names></name> <name><surname>Pace</surname> <given-names>N. R.</given-names></name></person-group> (<year>2007</year>). <article-title>Molecular-phylogenetic characterization of microbial community imbalances in human inflammatory bowel diseases</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A</source>. <volume>104</volume>, <fpage>13780</fpage>&#x02013;<lpage>13785</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.0706625104</pub-id><pub-id pub-id-type="pmid">17699621</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gloor</surname> <given-names>G. B.</given-names></name> <name><surname>Macklaim</surname> <given-names>J. M.</given-names></name> <name><surname>Pawlowsky-Glahn</surname> <given-names>V.</given-names></name> <name><surname>Egozcue</surname> <given-names>J. J.</given-names></name></person-group> (<year>2017</year>). <article-title>Microbiome datasets are compositional: and this is not optional</article-title>. <source>Front. Microbiol</source>. <volume>8</volume>:<fpage>2224</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2017.02224</pub-id><pub-id pub-id-type="pmid">29187837</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jasner</surname> <given-names>Y.</given-names></name> <name><surname>Belogolovski</surname> <given-names>A.</given-names></name> <name><surname>Ben-Itzhak</surname> <given-names>M.</given-names></name> <name><surname>Koren</surname> <given-names>O.</given-names></name> <name><surname>Louzoun</surname> <given-names>Y.</given-names></name></person-group> (<year>2021</year>). <article-title>Microbiome preprocessing machine learning pipeline</article-title>. <source>Front. Immunol</source>. <volume>12</volume>:<fpage>677870</fpage>. <pub-id pub-id-type="doi">10.3389/fimmu.2021.677870</pub-id><pub-id pub-id-type="pmid">34220823</pub-id></citation></ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jiang</surname> <given-names>R.</given-names></name> <name><surname>Zhan</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>T.</given-names></name></person-group> (<year>2023</year>). <article-title>A flexible zero-inflated poisson-gamma model with application to microbiome sequence count data</article-title>. <source>J. Am. Stat. Assoc</source>. <volume>118</volume>, <fpage>792</fpage>&#x02013;<lpage>804</lpage>. <pub-id pub-id-type="doi">10.1080/01621459.2022.2151447</pub-id></citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jiang</surname> <given-names>S.</given-names></name> <name><surname>Xiao</surname> <given-names>G.</given-names></name> <name><surname>Koh</surname> <given-names>A. Y.</given-names></name> <name><surname>Kim</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>Q.</given-names></name> <name><surname>Zhan</surname> <given-names>X.</given-names></name></person-group> (<year>2019</year>). <article-title>A bayesian zero-inflated negative binomial regression model for the integrative analysis of microbiome data</article-title>. <source>Biostatistics</source> <volume>22</volume>, <fpage>522</fpage>&#x02013;<lpage>540</lpage>. <pub-id pub-id-type="doi">10.1093/biostatistics/kxz050</pub-id><pub-id pub-id-type="pmid">31844880</pub-id></citation></ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kaul</surname> <given-names>A.</given-names></name> <name><surname>Mandal</surname> <given-names>S.</given-names></name> <name><surname>Davidov</surname> <given-names>O.</given-names></name> <name><surname>Peddada</surname> <given-names>S. D.</given-names></name></person-group> (<year>2017</year>). <article-title>Analysis of microbiome data in the presence of excess zeros</article-title>. <source>Front. Microbiol</source>. <volume>8</volume>:<fpage>2114</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2017.02114</pub-id><pub-id pub-id-type="pmid">29163406</pub-id></citation></ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Knights</surname> <given-names>D.</given-names></name> <name><surname>Silverberg</surname> <given-names>M. S.</given-names></name> <name><surname>Weersma</surname> <given-names>R. K.</given-names></name> <name><surname>Gevers</surname> <given-names>D.</given-names></name> <name><surname>Dijkstra</surname> <given-names>G.</given-names></name> <name><surname>Huang</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>Complex host genetics influence the microbiome in inflammatory bowel disease</article-title>. <source>Genome Med</source>. <volume>6</volume>, <fpage>1</fpage>&#x02013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1186/s13073-014-0107-1</pub-id><pub-id pub-id-type="pmid">25587358</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krych</surname> <given-names>L.</given-names></name> <name><surname>Hansen</surname> <given-names>C. H. F.</given-names></name> <name><surname>Hansen</surname> <given-names>A. K.</given-names></name> <name><surname>van den Berg</surname> <given-names>F. W. J.</given-names></name> <name><surname>Nielsen</surname> <given-names>D. S.</given-names></name></person-group> (<year>2013</year>). <article-title>Quantitatively different, yet qualitatively alike: a meta-analysis of the mouse core gut microbiome with a view towards the human gut microbiome</article-title>. <source>PLoS ONE</source> <volume>8</volume>:<fpage>e62578</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0062578</pub-id><pub-id pub-id-type="pmid">23658749</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kuczynski</surname> <given-names>J.</given-names></name> <name><surname>Lauber</surname> <given-names>C. L.</given-names></name> <name><surname>Walters</surname> <given-names>W. A.</given-names></name> <name><surname>Parfrey</surname> <given-names>L. W.</given-names></name> <name><surname>Clemente</surname> <given-names>J. C.</given-names></name> <name><surname>Gevers</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2011</year>). <article-title>Experimental and analytical tools for studying the human microbiome</article-title>. <source>Nat. Rev. Genet</source>. <volume>13</volume>, <fpage>47</fpage>&#x02013;<lpage>58</lpage>. <pub-id pub-id-type="doi">10.1038/nrg3129</pub-id><pub-id pub-id-type="pmid">22179717</pub-id></citation></ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kurtz</surname> <given-names>Z. D.</given-names></name> <name><surname>M&#x000FC;ller</surname> <given-names>C. L.</given-names></name> <name><surname>Miraldi</surname> <given-names>E. R.</given-names></name> <name><surname>Littman</surname> <given-names>D. R.</given-names></name> <name><surname>Blaser</surname> <given-names>M. J.</given-names></name> <name><surname>Bonneau</surname> <given-names>R. A.</given-names></name></person-group> (<year>2015</year>). <article-title>Sparse and compositionally robust inference of microbial ecological networks</article-title>. <source>PLoS Comput. Biol</source>. <volume>11</volume>:<fpage>e1004226</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1004226</pub-id><pub-id pub-id-type="pmid">25950956</pub-id></citation></ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lambert</surname> <given-names>D.</given-names></name></person-group> (<year>1992</year>). <article-title>Zero-inflated poisson regression, with an application to defects in manufacturing</article-title>. <source>Technometrics</source> <volume>34</volume>, <fpage>1</fpage>&#x02013;<lpage>14</lpage>.</citation>
</ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>S.</given-names></name> <name><surname>Chugh</surname> <given-names>P. E.</given-names></name> <name><surname>Shen</surname> <given-names>H.</given-names></name> <name><surname>Eberle</surname> <given-names>R.</given-names></name> <name><surname>Dittmer</surname> <given-names>D. P.</given-names></name></person-group> (<year>2013</year>). <article-title>Poisson factor models with applications to non-normalized microRNA profiling</article-title>. <source>Bioinformatics</source> <volume>29</volume>, <fpage>1105</fpage>&#x02013;<lpage>1111</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btt091</pub-id><pub-id pub-id-type="pmid">23428639</pub-id></citation></ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ley</surname> <given-names>R. E.</given-names></name> <name><surname>Hamady</surname> <given-names>M.</given-names></name> <name><surname>Lozupone</surname> <given-names>C.</given-names></name> <name><surname>Turnbaugh</surname> <given-names>P. J.</given-names></name> <name><surname>Ramey</surname> <given-names>R. R.</given-names></name> <name><surname>Bircher</surname> <given-names>J. S.</given-names></name> <etal/></person-group>. (<year>2008</year>). <article-title>Evolution of mammals and their gut microbes</article-title>. <source>Science</source> <volume>320</volume>, <fpage>1647</fpage>&#x02013;<lpage>1651</lpage>. <pub-id pub-id-type="doi">10.1126/science.1155725</pub-id><pub-id pub-id-type="pmid">18497261</pub-id></citation></ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>G.</given-names></name> <name><surname>Huang</surname> <given-names>J. Z.</given-names></name> <name><surname>Shen</surname> <given-names>H.</given-names></name></person-group> (<year>2018</year>). <article-title>Exponential family functional data analysis via a low-rank model</article-title>. <source>Biometrics</source> <volume>74</volume>, <fpage>1301</fpage>&#x02013;<lpage>1310</lpage>. <pub-id pub-id-type="doi">10.1111/biom.12885</pub-id><pub-id pub-id-type="pmid">29738627</pub-id></citation></ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>W.</given-names></name> <name><surname>Shi</surname> <given-names>P.</given-names></name> <name><surname>Feng</surname> <given-names>R.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name></person-group> (<year>2014</year>). <article-title>Variable selection in regression with compositional covariates</article-title>. <source>Biometrika</source> <volume>101</volume>, <fpage>785</fpage>&#x02013;<lpage>797</lpage>. <pub-id pub-id-type="doi">10.1093/biomet/asu031</pub-id></citation>
</ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Love</surname> <given-names>M. I.</given-names></name> <name><surname>Huber</surname> <given-names>W.</given-names></name> <name><surname>Anders</surname> <given-names>S.</given-names></name></person-group> (<year>2014</year>). <article-title>Moderated estimation of fold change and dispersion for RNA-seq data with DESeq2</article-title>. <source>Genome Biol</source>. <volume>15</volume>:<fpage>8</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-014-0550-8</pub-id><pub-id pub-id-type="pmid">25516281</pub-id></citation></ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Machiels</surname> <given-names>K.</given-names></name> <name><surname>Joossens</surname> <given-names>M.</given-names></name> <name><surname>Sabino</surname> <given-names>J.</given-names></name> <name><surname>Preter</surname> <given-names>V. D.</given-names></name> <name><surname>Arijs</surname> <given-names>I.</given-names></name> <name><surname>Eeckhaut</surname> <given-names>V.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>A decrease of the butyrate-producing species <italic>Roseburia hominis</italic> and <italic>Faecalibacterium prausnitzii</italic> defines dysbiosis in patients with ulcerative colitis</article-title>. <source>Gut</source> <volume>63</volume>, <fpage>1275</fpage>&#x02013;<lpage>1283</lpage>. <pub-id pub-id-type="doi">10.1136/gutjnl-2013-304833</pub-id><pub-id pub-id-type="pmid">24021287</pub-id></citation></ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mandal</surname> <given-names>S.</given-names></name> <name><surname>Van Treuren</surname> <given-names>W.</given-names></name> <name><surname>White</surname> <given-names>R. A.</given-names></name> <name><surname>Eggesb&#x000F8;</surname> <given-names>M.</given-names></name> <name><surname>Knight</surname> <given-names>R.</given-names></name> <name><surname>Peddada</surname> <given-names>S. D.</given-names></name></person-group> (<year>2015</year>). <article-title>Analysis of composition of microbiomes: a novel method for studying microbial composition</article-title>. <source>Microb. Ecol. Health Dis</source>. <volume>26</volume>:<fpage>27663</fpage>. <pub-id pub-id-type="doi">10.3402/mehd.v26.27663</pub-id><pub-id pub-id-type="pmid">26028277</pub-id></citation></ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mart&#x000ED;n-Fern&#x000E1;ndez</surname> <given-names>J. A.</given-names></name> <name><surname>Barcel&#x000F3;-Vidal</surname> <given-names>C.</given-names></name> <name><surname>Pawlowsky-Glahn</surname> <given-names>V.</given-names></name></person-group> (<year>2003</year>). <article-title>Dealing with zeros and missing values in compositional data sets using nonparametric imputation</article-title>. <source>Math. Geol</source>. <volume>35</volume>, <fpage>253</fpage>&#x02013;<lpage>278</lpage>. <pub-id pub-id-type="doi">10.1023/a:1023866030544</pub-id></citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mart&#x000ED;n-Fern&#x000E1;ndez</surname> <given-names>J. A.</given-names></name> <name><surname>Hron</surname> <given-names>K.</given-names></name> <name><surname>Templ</surname> <given-names>M.</given-names></name> <name><surname>Filzmoser</surname> <given-names>P.</given-names></name> <name><surname>Palarea-Albaladejo</surname> <given-names>J.</given-names></name></person-group> (<year>2012</year>). <article-title>Model-based replacement of rounded zeros in compositional data: classical and robust approaches</article-title>. <source>Comput. Stat. Data Anal</source>. <volume>56</volume>, <fpage>2688</fpage>&#x02013;<lpage>2704</lpage>. <pub-id pub-id-type="doi">10.1016/j.csda.2012.02.012</pub-id></citation>
</ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mart&#x000ED;n-Fern&#x000E1;ndez</surname> <given-names>J. A.</given-names></name> <name><surname>Hron</surname> <given-names>K.</given-names></name> <name><surname>Templ</surname> <given-names>M.</given-names></name> <name><surname>Filzmoser</surname> <given-names>P.</given-names></name> <name><surname>Palarea-Albaladejo</surname> <given-names>J.</given-names></name></person-group> (<year>2014</year>). <article-title>Bayesian-multiplicative treatment of count zeros in compositional data sets</article-title>. <source>Stat. Model</source>. <volume>15</volume>, <fpage>134</fpage>&#x02013;<lpage>158</lpage>. <pub-id pub-id-type="doi">10.1177/1471082x14535524</pub-id></citation>
</ref>
<ref id="B38">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>McCullagh</surname> <given-names>P.</given-names></name> <name><surname>Nelder</surname> <given-names>J.</given-names></name></person-group> (<year>1989</year>). <source>Generalized Linear Models, 2nd Edn</source>. <publisher-loc>London</publisher-loc>: <publisher-name>Chapman and Hall</publisher-name>.</citation>
</ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>McKnite</surname> <given-names>A. M.</given-names></name> <name><surname>Perez-Munoz</surname> <given-names>M. E.</given-names></name> <name><surname>Lu</surname> <given-names>L.</given-names></name> <name><surname>Williams</surname> <given-names>E. G.</given-names></name> <name><surname>Brewer</surname> <given-names>S.</given-names></name> <name><surname>Andreux</surname> <given-names>P. A.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>Murine gut microbiota is defined by host genetics and modulates variation of metabolic traits</article-title>. <source>PLoS ONE</source> <volume>7</volume>:<fpage>e39191</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0039191</pub-id><pub-id pub-id-type="pmid">22723961</pub-id></citation></ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mishra</surname> <given-names>A. K.</given-names></name> <name><surname>M&#x000FC;ller</surname> <given-names>C. L.</given-names></name></person-group> (<year>2022</year>). <article-title>Negative binomial factor regression with application to microbiome data analysis</article-title>. <source>Stat. Med</source>. <volume>41</volume>, <fpage>2786</fpage>&#x02013;<lpage>2803</lpage>. <pub-id pub-id-type="doi">10.1002/sim.9384</pub-id><pub-id pub-id-type="pmid">35466418</pub-id></citation></ref>
<ref id="B41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Moghimbeigi</surname> <given-names>A.</given-names></name></person-group> (<year>2015</year>). <article-title>Two-part zero-inflated negative binomial regression model for quantitative trait loci mapping with count trait</article-title>. <source>J. Theoret. Biol</source>. <volume>372</volume>, <fpage>74</fpage>&#x02013;<lpage>80</lpage>. <pub-id pub-id-type="doi">10.1016/j.jtbi.2015.02.016</pub-id><pub-id pub-id-type="pmid">25728790</pub-id></citation></ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Parks</surname> <given-names>B. W.</given-names></name> <name><surname>Nam</surname> <given-names>E.</given-names></name> <name><surname>Org</surname> <given-names>E.</given-names></name> <name><surname>Kostem</surname> <given-names>E.</given-names></name> <name><surname>Norheim</surname> <given-names>F.</given-names></name> <name><surname>Hui</surname> <given-names>S. T.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>Genetic control of obesity and gut microbiota composition in response to high-fat, high-sucrose diet in mice</article-title>. <source>Cell Metab</source>. <volume>17</volume>, <fpage>141</fpage>&#x02013;<lpage>152</lpage>. <pub-id pub-id-type="doi">10.1016/j.cmet.2012.12.007</pub-id><pub-id pub-id-type="pmid">23312289</pub-id></citation></ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Paulson</surname> <given-names>J. N.</given-names></name> <name><surname>Bravo</surname> <given-names>H. C.</given-names></name> <name><surname>Pop</surname> <given-names>M.</given-names></name></person-group> (<year>2014</year>). <article-title>Reply to: &#x0201C;a fair comparison.&#x0201D;</article-title> <source>Nat. Methods</source> <volume>11</volume>, <fpage>359</fpage>&#x02013;<lpage>360</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.2898</pub-id><pub-id pub-id-type="pmid">24681718</pub-id></citation></ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Paulson</surname> <given-names>J. N.</given-names></name> <name><surname>Stine</surname> <given-names>O. C.</given-names></name> <name><surname>Bravo</surname> <given-names>H. C.</given-names></name> <name><surname>Pop</surname> <given-names>M.</given-names></name></person-group> (<year>2013</year>). <article-title>Differential abundance analysis for microbial marker-gene surveys</article-title>. <source>Nat. Methods</source> <volume>10</volume>, <fpage>1200</fpage>&#x02013;<lpage>1202</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.2658</pub-id><pub-id pub-id-type="pmid">24076764</pub-id></citation></ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Peng</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>G.</given-names></name> <name><surname>Liu</surname> <given-names>Z.</given-names></name></person-group> (<year>2016</year>). <article-title>Zero-inflated beta regression for differential abundance analysis with metagenomics data</article-title>. <source>J. Comput. Biol</source>. <volume>23</volume>, <fpage>102</fpage>&#x02013;<lpage>110</lpage>. <pub-id pub-id-type="doi">10.1089/cmb.2015.0157</pub-id><pub-id pub-id-type="pmid">26675626</pub-id></citation></ref>
<ref id="B46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pierson</surname> <given-names>E.</given-names></name> <name><surname>Yau</surname> <given-names>C.</given-names></name></person-group> (<year>2015</year>). <article-title>ZIFA: dimensionality reduction for zero-inflated single-cell gene expression analysis</article-title>. <source>Genome Biol</source>. <volume>16</volume>, <fpage>1</fpage>&#x02013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1186/s13059-015-0805-z</pub-id><pub-id pub-id-type="pmid">26527291</pub-id></citation></ref>
<ref id="B47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pinart</surname> <given-names>M.</given-names></name> <name><surname>D&#x000F6;tsch</surname> <given-names>A.</given-names></name> <name><surname>Schlicht</surname> <given-names>K.</given-names></name> <name><surname>Laudes</surname> <given-names>M.</given-names></name> <name><surname>Bouwman</surname> <given-names>J.</given-names></name> <name><surname>Forslund</surname> <given-names>S. K.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Gut microbiome composition in obese and non-obese persons: a systematic review and meta-analysis</article-title>. <source>Nutrients</source> <volume>14</volume>, <fpage>1</fpage>&#x02013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.3390/nu14010012</pub-id><pub-id pub-id-type="pmid">35010887</pub-id></citation></ref>
<ref id="B48">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Robinson</surname> <given-names>M. D.</given-names></name> <name><surname>McCarthy</surname> <given-names>D. J.</given-names></name> <name><surname>Smyth</surname> <given-names>G. K.</given-names></name></person-group> (<year>2009</year>). <article-title>&#x0003C;tt&#x0003E;Edger&#x0003C;/tt&#x0003E;: a bioconductor package for differential expression analysis of digital gene expression data</article-title>. <source>Bioinformatics</source> <volume>26</volume>, <fpage>139</fpage>&#x02013;<lpage>140</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp616</pub-id><pub-id pub-id-type="pmid">19910308</pub-id></citation></ref>
<ref id="B49">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Silverman</surname> <given-names>J. D.</given-names></name> <name><surname>Roche</surname> <given-names>K.</given-names></name> <name><surname>Mukherjee</surname> <given-names>S.</given-names></name> <name><surname>David</surname> <given-names>L. A.</given-names></name></person-group> (<year>2020</year>). <article-title>Naught all zeros in sequence count data are the same</article-title>. <source>Comput. Struct. Biotechnol. J</source>. <volume>18</volume>, <fpage>2789</fpage>&#x02013;<lpage>2798</lpage>. <pub-id pub-id-type="doi">10.1016/j.csbj.2020.09.014</pub-id><pub-id pub-id-type="pmid">33101615</pub-id></citation></ref>
<ref id="B50">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sims</surname> <given-names>D.</given-names></name> <name><surname>Sudbery</surname> <given-names>I.</given-names></name> <name><surname>Ilott</surname> <given-names>N. E.</given-names></name> <name><surname>Heger</surname> <given-names>A.</given-names></name> <name><surname>Ponting</surname> <given-names>C. P.</given-names></name></person-group> (<year>2014</year>). <article-title>Sequencing depth and coverage: key considerations in genomic analyses</article-title>. <source>Nat. Rev. Genet</source>. <volume>15</volume>, <fpage>121</fpage>&#x02013;<lpage>132</lpage>. <pub-id pub-id-type="doi">10.1038/nrg3642</pub-id><pub-id pub-id-type="pmid">24434847</pub-id></citation></ref>
<ref id="B51">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Singh</surname> <given-names>R. P.</given-names></name> <name><surname>Reddy</surname> <given-names>C. R. K.</given-names></name></person-group> (<year>2016</year>). <article-title>Unraveling the functions of the macroalgal microbiome</article-title>. <source>Front. Microbiol</source>. <volume>6</volume>, <fpage>1</fpage>&#x02013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2015.01488</pub-id><pub-id pub-id-type="pmid">26779144</pub-id></citation></ref>
<ref id="B52">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sohn</surname> <given-names>M. B.</given-names></name> <name><surname>Du</surname> <given-names>R.</given-names></name> <name><surname>An</surname> <given-names>L.</given-names></name></person-group> (<year>2015</year>). <article-title>A robust approach for identifying differentially abundant features in metagenomic samples</article-title>. <source>Bioinformatics</source> <volume>31</volume>, <fpage>2269</fpage>&#x02013;<lpage>2275</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btv165</pub-id><pub-id pub-id-type="pmid">25792553</pub-id></citation></ref>
<ref id="B53">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sohn</surname> <given-names>M. B.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name></person-group> (<year>2017</year>). <article-title>A GLM-based latent variable ordination method for microbiome samples</article-title>. <source>Biometrics</source> <volume>74</volume>, <fpage>448</fpage>&#x02013;<lpage>457</lpage>. <pub-id pub-id-type="doi">10.1111/biom.12775</pub-id><pub-id pub-id-type="pmid">28991375</pub-id></citation></ref>
<ref id="B54">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>Q.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Song</surname> <given-names>Y.</given-names></name> <name><surname>Ma</surname> <given-names>X.</given-names></name> <name><surname>Shi</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title><italic>L. plantarum, L. fermentum</italic>, and <italic>B. breve</italic> beads modified the intestinal microbiota and alleviated the inflammatory response in high-fat diet&#x02013;fed mice</article-title>. <source>Probiot. Antimicrob. Prot</source>. <volume>12</volume>, <fpage>535</fpage>&#x02013;<lpage>544</lpage>. <pub-id pub-id-type="doi">10.1007/s12602-019-09564-3</pub-id><pub-id pub-id-type="pmid">31267477</pub-id></citation></ref>
<ref id="B55">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tang</surname> <given-names>Z. Z.</given-names></name> <name><surname>Chen</surname> <given-names>G.</given-names></name></person-group> (<year>2018</year>). <article-title>Zero-inflated generalized dirichlet multinomial regression model for microbiome compositional data analysis</article-title>. <source>Biostatistics</source> <volume>20</volume>, <fpage>698</fpage>&#x02013;<lpage>713</lpage>. <pub-id pub-id-type="doi">10.1093/biostatistics/kxy025</pub-id><pub-id pub-id-type="pmid">29939212</pub-id></citation></ref>
<ref id="B56">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tirozzi</surname> <given-names>P.</given-names></name> <name><surname>Orioli</surname> <given-names>V.</given-names></name> <name><surname>Dondina</surname> <given-names>O.</given-names></name> <name><surname>Kataoka</surname> <given-names>L.</given-names></name> <name><surname>Bani</surname> <given-names>L.</given-names></name></person-group> (<year>2022</year>). <article-title>Population trends from count data: handling environmental bias, overdispersion and excess of zeroes</article-title>. <source>Ecol. Informat</source>. <volume>69</volume>:<fpage>101629</fpage>. <pub-id pub-id-type="doi">10.1016/j.ecoinf.2022.101629</pub-id></citation>
</ref>
<ref id="B57">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Turnbaugh</surname> <given-names>P. J.</given-names></name> <name><surname>Ley</surname> <given-names>R. E.</given-names></name> <name><surname>Mahowald</surname> <given-names>M. A.</given-names></name> <name><surname>Magrini</surname> <given-names>V.</given-names></name> <name><surname>Mardis</surname> <given-names>E. R.</given-names></name> <name><surname>Gordon</surname> <given-names>J. I.</given-names></name></person-group> (<year>2006</year>). <article-title>An obesity-associated gut microbiome with increased capacity for energy harvest</article-title>. <source>Nature</source> <volume>444</volume>, <fpage>1027</fpage>&#x02013;<lpage>1031</lpage>. <pub-id pub-id-type="doi">10.1038/nature05414</pub-id><pub-id pub-id-type="pmid">17183312</pub-id></citation></ref>
<ref id="B58">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tyler</surname> <given-names>A. D.</given-names></name> <name><surname>Smith</surname> <given-names>M. I.</given-names></name> <name><surname>Silverberg</surname> <given-names>M. S.</given-names></name></person-group> (<year>2014</year>). <article-title>Analyzing the human microbiome: a &#x0201C;how to&#x0201D; guide for physicians</article-title>. <source>Am. J. Gastroenterol</source>. <volume>109</volume>, <fpage>983</fpage>&#x02013;<lpage>993</lpage>. <pub-id pub-id-type="doi">10.1038/ajg.2014.73</pub-id><pub-id pub-id-type="pmid">24751579</pub-id></citation></ref>
<ref id="B59">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vijay-Kumar</surname> <given-names>M.</given-names></name> <name><surname>Aitken</surname> <given-names>J. D.</given-names></name> <name><surname>Carvalho</surname> <given-names>F. A.</given-names></name> <name><surname>Cullender</surname> <given-names>T. C.</given-names></name> <name><surname>Mwangi</surname> <given-names>S.</given-names></name> <name><surname>Srinivasan</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2010</year>). <article-title>Metabolic syndrome and altered gut microbiota in mice lacking Toll-like receptor 5</article-title>. <source>Science</source> <volume>328</volume>, <fpage>228</fpage>&#x02013;<lpage>231</lpage>. <pub-id pub-id-type="doi">10.1126/science.1179721</pub-id><pub-id pub-id-type="pmid">20203013</pub-id></citation></ref>
<ref id="B60">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wagh</surname> <given-names>Y. S.</given-names></name> <name><surname>Kamalja</surname> <given-names>K. K.</given-names></name></person-group> (<year>2017</year>). <article-title>Zero-inflated models and estimation in zero-inflated poisson distribution</article-title>. <source>Commun. Stat</source>. <volume>47</volume>, <fpage>2248</fpage>&#x02013;<lpage>2265</lpage>. <pub-id pub-id-type="doi">10.1080/03610918.2017.1341526</pub-id></citation>
</ref>
<ref id="B61">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wen</surname> <given-names>L.</given-names></name> <name><surname>Ley</surname> <given-names>R. E.</given-names></name> <name><surname>Volchkov</surname> <given-names>P. Y.</given-names></name> <name><surname>Stranges</surname> <given-names>P. B.</given-names></name> <name><surname>Avanesyan</surname> <given-names>L.</given-names></name> <name><surname>Stonebraker</surname> <given-names>A. C.</given-names></name> <etal/></person-group>. (<year>2008</year>). <article-title>Innate immunity and intestinal microbiota in the development of Type 1 diabetes</article-title>. <source>Nature</source> <volume>455</volume>, <fpage>1109</fpage>&#x02013;<lpage>1113</lpage>. <pub-id pub-id-type="doi">10.1038/nature07336</pub-id><pub-id pub-id-type="pmid">18806780</pub-id></citation></ref>
<ref id="B62">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wilkinson</surname> <given-names>L.</given-names></name> <name><surname>Friendly</surname> <given-names>M.</given-names></name></person-group> (<year>2009</year>). <article-title>The history of the cluster heat map</article-title>. <source>Am. Stat</source>. <volume>63</volume>, <fpage>179</fpage>&#x02013;<lpage>184</lpage>. <pub-id pub-id-type="doi">10.1198/tas.2009.0033</pub-id><pub-id pub-id-type="pmid">12611515</pub-id></citation></ref>
<ref id="B63">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Willing</surname> <given-names>B. P.</given-names></name> <name><surname>Dicksved</surname> <given-names>J.</given-names></name> <name><surname>Halfvarson</surname> <given-names>J.</given-names></name> <name><surname>Andersson</surname> <given-names>A. F.</given-names></name> <name><surname>Lucio</surname> <given-names>M.</given-names></name> <name><surname>Zheng</surname> <given-names>Z.</given-names></name> <etal/></person-group>. (<year>2010</year>). <article-title>A pyrosequencing study in twins shows that gastrointestinal microbial profiles vary with inflammatory bowel disease phenotypes</article-title>. <source>Gastroenterology</source> <volume>139</volume>, <fpage>1844</fpage>&#x02013;<lpage>1854</lpage>. <pub-id pub-id-type="doi">10.1053/j.gastro.2010.08.049</pub-id><pub-id pub-id-type="pmid">20816835</pub-id></citation></ref>
<ref id="B64">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>C. F. J.</given-names></name></person-group> (<year>1983</year>). <article-title>On the convergence properties of the EM algorithm</article-title>. <source>Ann. Stat</source>. <volume>11</volume>, <fpage>95</fpage>&#x02013;<lpage>103</lpage>.</citation>
</ref>
<ref id="B65">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>L.</given-names></name> <name><surname>Paterson</surname> <given-names>A. D.</given-names></name> <name><surname>Turpin</surname> <given-names>W.</given-names></name> <name><surname>Xu</surname> <given-names>W.</given-names></name></person-group> (<year>2015</year>). <article-title>Assessment and selection of competing models for zero-inflated microbiome data</article-title>. <source>PLoS ONE</source> <volume>10</volume>:<fpage>e0129606</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0129606</pub-id><pub-id pub-id-type="pmid">26148172</pub-id></citation></ref>
<ref id="B66">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>T.</given-names></name> <name><surname>Demmer</surname> <given-names>R. T.</given-names></name> <name><surname>Li</surname> <given-names>G.</given-names></name></person-group> (<year>2020</year>). <article-title>Zero-inflated poisson factor model with application to microbiome read counts</article-title>. <source>Biometrics</source> <volume>77</volume>, <fpage>91</fpage>&#x02013;<lpage>101</lpage>. <pub-id pub-id-type="doi">10.1111/biom.13272</pub-id><pub-id pub-id-type="pmid">32277466</pub-id></citation></ref>
<ref id="B67">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yau</surname> <given-names>K. K. W.</given-names></name> <name><surname>Wang</surname> <given-names>K.</given-names></name> <name><surname>Lee</surname> <given-names>A. H.</given-names></name></person-group> (<year>2003</year>). <article-title>Zero&#x02013;inflated negative binomial mixed regression modeling of over&#x02013;dispersed count data with extra zeros</article-title>. <source>Biometr. J</source>. <volume>45</volume>, <fpage>437</fpage>&#x02013;<lpage>452</lpage>. <pub-id pub-id-type="doi">10.1002/bimj.200390024</pub-id><pub-id pub-id-type="pmid">16794991</pub-id></citation></ref>
<ref id="B68">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zeng</surname> <given-names>Y.</given-names></name> <name><surname>Pang</surname> <given-names>D.</given-names></name> <name><surname>Zhao</surname> <given-names>H.</given-names></name> <name><surname>Wang</surname> <given-names>T.</given-names></name></person-group> (<year>2022</year>). <article-title>A zero-inflated logistic normal multinomial model for extracting microbial compositions</article-title>. <source>J. Am. Stat. Assoc</source>. <volume>2022</volume>:<fpage>2044827</fpage>. <pub-id pub-id-type="doi">10.1080/01621459.2022.2044827</pub-id></citation>
</ref>
<ref id="B69">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Pei</surname> <given-names>Y. F.</given-names></name> <name><surname>Zhang</surname> <given-names>L.</given-names></name> <name><surname>Guo</surname> <given-names>B.</given-names></name> <name><surname>Pendegraft</surname> <given-names>A. H.</given-names></name> <name><surname>Zhuang</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Negative binomial mixed models for analyzing longitudinal microbiome data</article-title>. <source>Front. Microbiol</source>. <volume>9</volume>:<fpage>1683</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2018.01683</pub-id><pub-id pub-id-type="pmid">30093893</pub-id></citation></ref>
<ref id="B70">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Yi</surname> <given-names>N.</given-names></name></person-group> (<year>2020</year>). <article-title>Fast zero-inflated negative binomial mixed modeling approach for analyzing longitudinal metagenomics data</article-title>. <source>Bioinformatics</source> <volume>36</volume>, <fpage>2345</fpage>&#x02013;<lpage>2351</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz973</pub-id><pub-id pub-id-type="pmid">31904815</pub-id></citation></ref>
</ref-list>
</back>
</article>