<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Bioinform.</journal-id>
<journal-title>Frontiers in Bioinformatics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Bioinform.</abbrev-journal-title>
<issn pub-type="epub">2673-7647</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1484113</article-id>
<article-id pub-id-type="doi">10.3389/fbinf.2025.1484113</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Bioinformatics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Using short-read 16S rRNA sequencing of multiple variable regions to generate high-quality results to a species level</article-title>
<alt-title alt-title-type="left-running-head">Graham et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fbinf.2025.1484113">10.3389/fbinf.2025.1484113</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Graham</surname>
<given-names>Amy S.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1498664/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Patel</surname>
<given-names>Fadheela</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2986942/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Little</surname>
<given-names>Francesca</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/555697/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>van der Kouwe</surname>
<given-names>Andre</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/209480/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Kaba</surname>
<given-names>Mamadou</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/212724/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Holmes</surname>
<given-names>Martha J.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff7">
<sup>7</sup>
</xref>
<xref ref-type="aff" rid="aff8">
<sup>8</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2351491/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Imaging Sciences, Neuroscience Institute</institution>, <institution>University of Cape Town</institution>, <addr-line>Cape Town</addr-line>, <country>South Africa</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Human Biology</institution>, <institution>Division of Biomedical Engineering</institution>, <institution>University of Cape Town</institution>, <addr-line>Cape Town</addr-line>, <country>South Africa</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Pathology</institution>, <institution>Division of Medical Microbiology</institution>, <institution>University of Cape Town</institution>, <addr-line>Cape Town</addr-line>, <country>South Africa</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Department of Statistical Sciences</institution>, <institution>University of Cape Town</institution>, <addr-line>Cape Town</addr-line>, <country>South Africa</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Athinoula A. Martinos Centre for Biomedical Imaging</institution>, <institution>Department of Radiology</institution>, <institution>Massachusetts General Hospital</institution>, <addr-line>Boston</addr-line>, <addr-line>MA</addr-line>, <country>United States</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Department of Radiology</institution>, <institution>Harvard Medical School</institution>, <addr-line>Boston</addr-line>, <addr-line>MA</addr-line>, <country>United States</country>
</aff>
<aff id="aff7">
<sup>7</sup>
<institution>Department of Biomedical Physiology and Kinesiology</institution>, <institution>Simon Fraser University</institution>, <addr-line>Burnaby</addr-line>, <addr-line>BC</addr-line>, <country>Canada</country>
</aff>
<aff id="aff8">
<sup>8</sup>
<institution>ImageTech</institution>, <institution>Simon Fraser University</institution>, <addr-line>Surrey</addr-line>, <addr-line>BC</addr-line>, <country>Canada</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/106537/overview">David W. Ussery</ext-link>, University of Arkansas for Medical Sciences, United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/641628/overview">Maryam Omrani</ext-link>, San Raffaele Hospital, Italy</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2890046/overview">Chen Li</ext-link>, St Jude Children Hospital, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Amy S. Graham, <email>grhamy001@myuct.ac.za</email>
</corresp>
<fn fn-type="equal" id="fn001">
<label>
<sup>&#x2020;</sup>
</label>
<p>These authors have contributed equally to this work and share last authorship</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>17</day>
<month>03</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>5</volume>
<elocation-id>1484113</elocation-id>
<history>
<date date-type="received">
<day>21</day>
<month>08</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>11</day>
<month>02</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Graham, Patel, Little, van der Kouwe, Kaba and Holmes.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Graham, Patel, Little, van der Kouwe, Kaba and Holmes</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Short-read amplicon sequencing studies have typically focused on 1-2 variable regions of the 16S rRNA gene. Species-level resolution is limited in these studies, as each variable region enables the characterisation of a different subsection of the microbiome. Although long-read sequencing techniques can take advantage of all 9 variable regions by sequencing the entire 16S rRNA gene, short-read sequencing has remained a commonly used approach in 16S rRNA research. This work assessed the feasibility of accurate species-level resolution and reproducibility using a relatively new sequencing kit and bioinformatics pipeline developed for short-read sequencing of multiple variable regions of the 16S rRNA gene. In addition, we evaluated the potential impact of different sample collection methods on our outcomes.</p>
</sec>
<sec>
<title>Methods</title>
<p>Using xGen&#x2122; 16S Amplicon Panel v2 kits, sequencing of all 9 variable regions of the 16S rRNA gene was carried out on an Illumina MiSeq platform. Mock cells and mock DNA for 8 bacterial species were included as extraction and sequencing controls respectively. Within-run and between-run replicate samples, and pairs of stool and rectal swabs collected at 0&#x2013;5 weeks from the same infants, were incorporated. Observed relative abundances of each species were compared to theoretical abundances provided by ZymoBIOMICS. Paired Wilcoxon rank sum tests and distance-based intraclass correlation coefficients were used to statistically compare alpha and beta diversity measures, respectively, for pairs of replicates and stool/rectal swab sample pairs.</p>
</sec>
<sec>
<title>Results</title>
<p>Using multiple variable regions of the 16S ribosomal Ribonucleic Acid (rRNA) gene, we found that we could accurately identify taxa to a species level and obtain highly reproducible results at a species level. Yet, the microbial profiles of stool and rectal swab sample pairs differed substantially despite being collected concurrently from the same infants.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>This protocol provides an effective means for studying infant gut microbial samples at a species level. However, sample collection approaches need to be accounted for in any downstream analysis.</p>
</sec>
</abstract>
<kwd-group>
<kwd>microbiome</kwd>
<kwd>16S rRNA sequencing</kwd>
<kwd>short-read</kwd>
<kwd>multiple variable regions</kwd>
<kwd>species-level</kwd>
</kwd-group>
<contract-num rid="cn001">R01HD093578 R01HD085813</contract-num>
<contract-sponsor id="cn001">National Institutes of Health<named-content content-type="fundref-id">10.13039/100000002</named-content>
</contract-sponsor>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Genomic Analysis</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Our ability to describe and appreciate the complexities of the human microbiome has been radically improved by next-generation sequencing tools (<xref ref-type="bibr" rid="B8">Bharti and Grimm, 2021</xref>; <xref ref-type="bibr" rid="B40">Ji and Nielsen, 2015</xref>; <xref ref-type="bibr" rid="B67">Rogers and Bruce, 2010</xref>). Although the use of second-generation, short-read sequencing platforms allows high read depths to be rapidly sequenced (<xref ref-type="bibr" rid="B38">Hu et al., 2021</xref>; <xref ref-type="bibr" rid="B77">Tucker et al., 2009</xref>), it is limited in terms of the assembly of contiguous sequences (<xref ref-type="bibr" rid="B50">Li et al., 2010</xref>; <xref ref-type="bibr" rid="B85">Zerbino and Birney, 2008</xref>). Various third-generation sequencing techniques exist and allow long reads to be sequenced, yet a greater number of errors have previously been found to result with these techniques (<xref ref-type="bibr" rid="B2">Amarasinghe et al., 2020</xref>; <xref ref-type="bibr" rid="B57">Midha et al., 2019</xref>; <xref ref-type="bibr" rid="B70">Sedlazeck et al., 2018</xref>; <xref ref-type="bibr" rid="B79">Van Dijk et al., 2018</xref>; <xref ref-type="bibr" rid="B63">Quail et al., 2012</xref>). As short-read sequencing approaches are still frequently used, however, there is a need to consider alternative ways to improve these approaches.</p>
<p>The 16S rRNA gene has been identified as a particularly useful target of research as it is common to all bacteria (<xref ref-type="bibr" rid="B1">Acinas et al., 2004</xref>; <xref ref-type="bibr" rid="B61">Patel, 2001</xref>). The gene consists of regions of DNA in which the sequence is conserved across all bacteria, while in other regions there is variation according to the individual bacterial species (<xref ref-type="bibr" rid="B82">Wang and Qian, 2009</xref>; <xref ref-type="bibr" rid="B46">Lane et al., 1985</xref>). As such, targeted amplicon sequencing can be done, comparing variable region sequences to a database of known taxa, to identify which bacterial species are present in a sample (<xref ref-type="bibr" rid="B82">Wang and Qian, 2009</xref>). In the past, amplicon sequencing studies have typically focused on one or two variable regions at a time (<xref ref-type="bibr" rid="B22">Claassen-Weitz et al., 2018</xref>; <xref ref-type="bibr" rid="B30">Gao et al., 2018</xref>; <xref ref-type="bibr" rid="B84">Yu et al., 2017</xref>; <xref ref-type="bibr" rid="B36">Hosgood III et al., 2014</xref>; <xref ref-type="bibr" rid="B15">Caporaso et al., 2011</xref>; <xref ref-type="bibr" rid="B86">Zhou et al., 2011</xref>). Yet, certain variable regions are better for enabling classification to lower taxonomic levels and each variable region favours classification of specific taxa (<xref ref-type="bibr" rid="B13">Bukin et al., 2019</xref>; <xref ref-type="bibr" rid="B35">Guo et al., 2013</xref>; <xref ref-type="bibr" rid="B18">Chakravorty et al., 2007</xref>). Consequently, this approach limits the ability to obtain accurate species-level resolution when focusing only on a single short fragment of the 16S rRNA gene. Using the entire 16S rRNA sequence is expected to provide better classification potential to a species level (<xref ref-type="bibr" rid="B41">Johnson et al., 2019</xref>).</p>
<p>The use of short-read sequencing techniques to study multiple variable regions of the 16S rRNA gene has captured the interest of researchers. There has been a rapid development in sequencing kits and bioinformatics pipelines to process multiple variable region 16S rRNA sequencing data (<xref ref-type="bibr" rid="B14">Callahan et al., 2021</xref>; <xref ref-type="bibr" rid="B29">Fuks et al., 2018</xref>; <xref ref-type="bibr" rid="B68">Schriefer et al., 2018</xref>; <xref ref-type="bibr" rid="B81">Wang et al., 2016</xref>; <xref ref-type="bibr" rid="B3">Amir et al., 2013</xref>). The xGen&#x2122; 16S Amplicon Panel v2 kits (Integrated DNA Technologies, Coralville, IA, United States) are an example, having been developed to amplify all nine variable regions of the 16S rRNA gene. Furthermore, a complementary bioinformatics pipeline known as the Swift Normalase Amplicon Panels APP for Python 3 (SNAPP-py3), was developed specifically for the analysis of sequencing data obtained using these kits (<xref ref-type="bibr" rid="B17">Chai, 2021</xref>).</p>
<p>Being relatively new, there are only a few publications in which the SNAPP-py3 pipeline has been used to analyse data sequenced with the xGen kits (<xref ref-type="bibr" rid="B59">Nuccio et al., 2023</xref>; <xref ref-type="bibr" rid="B7">Bennato et al., 2022</xref>). However, neither of these studies took advantage of the species-level classification that can be achieved with the xGen kits and SNAPP-py3 pipeline. Although <xref ref-type="bibr" rid="B7">Bennato et al. (2022)</xref> included a control containing DNA for 20 known bacterial species, they only reported the ability to pick up these bacteria at a genus level. To our knowledge, the combined ability of these kits and pipeline to obtain accurate species-level classification has not been assessed. Therefore, we sought to establish a protocol in which the SNAPP-py3 pipeline and additional processing steps were utilised to analyse short-read multiple variable region 16S rRNA data following sequencing with xGen amplicon panel kits.</p>
<p>The accuracy of sequencing protocols can be evaluated in a few ways using mock controls. Firstly, researchers can calculate the proportion of expected species that have been detected down to a species level when using a given protocol and for select regions of the 16S rRNA gene (<xref ref-type="bibr" rid="B41">Johnson et al., 2019</xref>; <xref ref-type="bibr" rid="B27">Fouhy et al., 2016</xref>). F-scores can be calculated based on the precision and sensitivity with which these species are identified (<xref ref-type="bibr" rid="B60">&#xd6;zkurt et al., 2022</xref>). The classification process can also be assessed according to the percentage of overall reads that are classified as belonging to one of the expected control species (<xref ref-type="bibr" rid="B75">Szoboszlay et al., 2023</xref>; <xref ref-type="bibr" rid="B78">Urban et al., 2021</xref>). Furthermore, accuracy can be assessed by comparing observed relative abundances to expected abundances (provided by suppliers) for each taxon in a control (<xref ref-type="bibr" rid="B52">Maki et al., 2023</xref>; <xref ref-type="bibr" rid="B75">Szoboszlay et al., 2023</xref>; <xref ref-type="bibr" rid="B24">Drengenes et al., 2021</xref>; <xref ref-type="bibr" rid="B47">Laursen et al., 2017</xref>; <xref ref-type="bibr" rid="B15">Caporaso et al., 2011</xref>). This can be done at different taxonomic levels and gives an indication of whether the amplification or sequencing processes have introduced bias by favouring certain species over others.</p>
<p>As stool collection is not always possible due to various factors, rectal swab collection has become a common sampling method for studying the gut microbiome (<xref ref-type="bibr" rid="B6">Bassis et al., 2017</xref>). Storage of rectal swabs differs to that of stool, as swabs generally need to be placed in a medium (<xref ref-type="bibr" rid="B16">CDC, 2015</xref>), for example, PrimeStore (<xref ref-type="bibr" rid="B26">Flygel et al., 2020</xref>). The results obtained from sequencing rectal swab samples can be inconsistent in terms of numbers of bacteria detected (<xref ref-type="bibr" rid="B19">Chanderraj et al., 2022</xref>).</p>
<p>Previous studies have explored whether rectal swab samples can provide a reliable alternative to stool samples (<xref ref-type="bibr" rid="B64">Radhakrishnan et al., 2023</xref>; <xref ref-type="bibr" rid="B9">Bokulich et al., 2019</xref>; <xref ref-type="bibr" rid="B66">Reyman et al., 2019</xref>; <xref ref-type="bibr" rid="B6">Bassis et al., 2017</xref>; <xref ref-type="bibr" rid="B28">Freedman et al., 2017</xref>). Although pairs of stool and rectal swab samples collected concurrently from the same individual generally display similar diversity and functional profiles (<xref ref-type="bibr" rid="B64">Radhakrishnan et al., 2023</xref>; <xref ref-type="bibr" rid="B66">Reyman et al., 2019</xref>; <xref ref-type="bibr" rid="B6">Bassis et al., 2017</xref>), there have been other studies that suggest that these samples are not equivalent for detecting specific taxa (<xref ref-type="bibr" rid="B42">Jones et al., 2018</xref>; <xref ref-type="bibr" rid="B28">Freedman et al., 2017</xref>; <xref ref-type="bibr" rid="B31">Goldfarb et al., 2014</xref>). In particular, rectal swab samples have been found to be more effective for detecting a greater number of harmful species in children with gastrointestinal infections (<xref ref-type="bibr" rid="B28">Freedman et al., 2017</xref>; <xref ref-type="bibr" rid="B31">Goldfarb et al., 2014</xref>). When sequencing the meconium (first stool) sample passed by newborn infants, rectal swab samples have been found to provide a less accurate representation of the microbiome compared to stool samples (<xref ref-type="bibr" rid="B33">Graspeuntner et al., 2023</xref>). Moreover, rectal swab samples provide a poorer representation of the microbiome if they are sequenced after greater than 48 h at room temperature (<xref ref-type="bibr" rid="B9">Bokulich et al., 2019</xref>). As a result, the interchangeability of stool and rectal swab samples needs to be assessed for new protocols. Moreover, sample collection approach is an important variable to consider with regards to the research objectives of a study.</p>
<p>The first aim of this research involves assessing the accuracy of extraction and sequencing protocols to achieve classification at the species level. This will be done by sequencing and analysing mock controls containing either whole cells or already-extracted DNA from eight known bacterial species, assessing the precision and sensitivity with which species were identified and how well their relative abundances matched theoretical abundances. Secondly, our goal is to evaluate the within-run and between-run reproducibility of species-level analysis. Technical replicate samples sequenced on either the same plate (within-run) or across different sequencing plates and runs (between-run), will be compared to determine this. Finally, we aim to identify whether there are differences at a species level between different sample collection approaches. To achieve this, we will compare pairs of stool and rectal swab samples collected from the same participants at the same time point (0&#x2013;5 week-old newborns). We hypothesise that by sequencing multiple variable regions of the 16S rRNA gene and using the SNAPP-py3 pipeline, we could obtain accurate species-level resolution and achieve reproducible results.</p>
</sec>
<sec sec-type="methods" id="s2">
<title>2 Methods</title>
<sec id="s2-1">
<title>2.1 Extraction controls, sequencing controls and technical replicates</title>
<p>To assess the reproducibility and accuracy of DNA extraction and sequencing steps, mock controls and technical replicates were included on each plate (<xref ref-type="fig" rid="F1">Figure 1</xref>). A ZymoBIOMICS&#x2122; Microbial Community Standard (catalog number ZR D6300), consisting of eight known bacterial species including both gram-positive and gram-negative bacteria, was included on each plate as a DNA extraction control. Each plate included a ZymoBIOMICS&#x2122; Microbial Community DNA Standard (catalog number ZR D6305), which contains already extracted DNA for eight known bacterial species. This served as a sequencing control. Seventeen within-run and 8 between-run technical replicate pairs were also included across the sequencing plates. These included samples that were collected across several different time points during infancy, specifically 0&#x2013;5 weeks, 3 months, 6 months, 9 months and 12 months.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The plate layout used for sequencing microbial samples, including the negative controls, positive (mock) controls and replicates.</p>
</caption>
<graphic xlink:href="fbinf-05-1484113-g001.tif"/>
</fig>
</sec>
<sec id="s2-2">
<title>2.2 Stool and swab sample collection</title>
<p>Twenty six pairs of stool and rectal swab samples were collected at the age of 0&#x2013;5 weeks (baseline samples), for infants born between 37 and 42 weeks gestational age. Swabs were stored in Primestore solution (PrimeStore&#xae; Molecular Transport medium). All stool and swab samples were transferred to a &#x2212;80&#xb0;C freezer for long-term storage.</p>
<p>Stool samples were thawed, and half of a pea-sized scoop was collected from the side/centre of the sample. This was placed in a tube with 750 &#xb5;L of lysis buffer. For rectal swab samples, 400 &#xb5;L of sample in Primestore was placed in a tube with 400 &#xb5;L of lysis buffer. These then underwent off-board lysis, using the QT Qiagen bead beater, prior to DNA extraction.</p>
</sec>
<sec id="s2-3">
<title>2.3 DNA extraction, preparation of sequencing library and illumina sequencing</title>
<p>Manual DNA extraction of the stool and rectal swab samples and mock extraction controls was carried out using the Quick-DNA&#x2122; Fecal/Soil Microbe Microprep Kit (ZymoBIOMICS catalog number D6012). Prior to carrying out polymerase chain reaction (PCR), Qubit&#x2122; (Thermo Fisher Scientific, Waltham, MA, United States) was done to check the starting DNA concentrations. xGen&#x2122; 16S Amplicon Panel v2 kits (Integrated DNA Technologies, Coralville, IA, United States) were used in library preparation for sequencing. Kits included primer pairs for amplification of all nine hypervariable regions of the 16S rRNA gene. Additionally, the primers for these kits have dual indices to allow greater numbers of samples to be run together in a single flow cell. Moreover, the xGen kits include Normalase&#x2122; which could be used to enzymatically normalise library sizes prior to sequencing. Finally, qPCR was performed following the Normalase step to quantify the final library size prior to sequencing.</p>
<p>Negative controls, including Primestore, Milli-Q water, elution buffer and Tris EDTA/Nuclease free water, were added to each plate (<xref ref-type="fig" rid="F1">Figure 1</xref>) together with prepared libraries from the stool and rectal swab samples. Mock controls and technical replicates, as described above, were also included on each plate. Sequencing was conducted across two sequencing runs on seven plates. The combined 16S library per run was subjected to paired-end sequencing on the Illumina&#xae; MiSeq&#x2122; platform, employing the MiSeq Reagent v3 kit with 600 cycles (Illumina, San Diego, CA, United States).</p>
</sec>
<sec id="s2-4">
<title>2.4 Bioinformatics processing of sequencing data</title>
<p>Ethics approval for this research was provided by the Human Research Ethics Committees at the University of Cape Town (801/2016 and 557/2020) and at Stellenbosch University (M16/10/041). Following sequencing, preprocessing steps were carried out for quality control and to prepare the data for statistical analysis (<xref ref-type="fig" rid="F2">Figure 2</xref>). Raw sequencing data was run through FastQC (<xref ref-type="bibr" rid="B4">Andrews, 2010</xref>) to assess the quality of the reads. Following this quality control step, forward and reverse reads for each sample were processed using the SNAPP-py3 pipeline (<xref ref-type="bibr" rid="B17">Chai, 2021</xref>). Sequencing data from run 1 and run 2 were processed separately. Of the four main output files from the pipeline, the lineage table and an adapted taxonomy table were used for further analysis.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>A flowchart summarising the processing and analysis steps carried out using raw forward and reverse read sequencing files. Boxes in green indicate the steps included in addition to the main SNAPP-py3 pipeline.</p>
</caption>
<graphic xlink:href="fbinf-05-1484113-g002.tif"/>
</fig>
<p>The remaining processing and analysis were carried out in R version 4.2.1 (<xref ref-type="bibr" rid="B65">R Core Team, 2022</xref>). This stage of processing began with creating a phyloseq object (<xref ref-type="bibr" rid="B56">McMurdie and Holmes, 2015</xref>; <xref ref-type="bibr" rid="B55">McMurdie and Holmes, 2013</xref>). Using the decontam package (<xref ref-type="bibr" rid="B23">Davis et al., 2018</xref>), decontamination was carried out separately for each plate, using plate-specific negative controls. A combined frequency and prevalence approach was used, selecting a threshold of 0.1 for the prevalence component. Phyloseq objects from runs 1 and 2 were then combined into a single phyloseq object for further downstream processing and analysis.</p>
<p>A normalisation step to account for different library sizes was implemented by determining the median library size and normalising each sample accordingly (<xref ref-type="bibr" rid="B5">Balle et al., 2020</xref>; <xref ref-type="bibr" rid="B76">The Jackson Laboratory, 2019</xref>). Finally, subsetting into various phyloseq objects was done to prepare for downstream statistical analysis. We ultimately had separate phyloseq objects containing the mock extraction controls, mock sequencing controls, within-run repeats, between-run repeats and baseline pairs of stool and rectal swab samples. Excel spreadsheets containing this data, as well as the corresponding code for importing the files into R as phyloseq objects, are provided in the <xref ref-type="sec" rid="s11">Supplementary Datasheets S1, S3</xref>, respectively.</p>
<p>Batch effect correction was done using MMUPHin (<xref ref-type="bibr" rid="B53">Ma, 2022</xref>; <xref ref-type="bibr" rid="B54">Ma et al., 2022</xref>). This data was compared to data in which no batch effect correction was carried out, to determine the necessity of accounting for batch effects.</p>
</sec>
<sec id="s2-5">
<title>2.5 Statistical analysis</title>
<p>Relative abundances for controls, replicates and stool/rectal swab sample pairs were visualised using QIIME2 software (<xref ref-type="bibr" rid="B10">Bolyen et al., 2019</xref>). All microbiome analysis was done at a species level in R. Genus-level and phylum-level analyses were additionally included in select steps to provide additional insights.</p>
<p>Performance measures were calculated as outlined by <xref ref-type="bibr" rid="B60">&#xd6;zkurt et al. (2022)</xref> using data for sequences classified to a species-level. Precision and sensitivity were calculated based on the number of correctly identified species expected to be in the mock control [true positives (TP)], the number of expected species that were not detected [false negatives (FN)] and the number of non-expected species classified as being in the control [false positive (FP)]. F-scores could then be calculated based on these values. The calculations used were as follows:<disp-formula id="equ1">
<mml:math id="m1">
<mml:mrow>
<mml:mtext>Precision</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>TP</mml:mtext>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtext>TP</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>FP</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ2">
<mml:math id="m2">
<mml:mrow>
<mml:mtext>Sensitivity</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>TP</mml:mtext>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtext>TP</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>FN</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>F-score &#x3d; 2&#x2a;precision&#x2a;sensitivity/(precision &#x2b; sensitivity) (<xref ref-type="bibr" rid="B60">&#xd6;zkurt et al., 2022</xref>).</p>
<p>The percentage relative abundances of each of the eight expected species were determined for mock cell (extraction) and mock DNA (sequencing) controls on each sequencing plate. For each control, the total percentage of sequences that were correctly classified as an expected species, was calculated. Furthermore, observed relative abundances were compared to the theoretical abundances provided by ZymoBIOMICS for each species in the mock controls by calculating Observed/Expected (O/E) ratios (<xref ref-type="bibr" rid="B52">Maki et al., 2023</xref>).</p>
<p>Functions from the phyloseq package in R (<xref ref-type="bibr" rid="B56">McMurdie and Holmes, 2015</xref>; <xref ref-type="bibr" rid="B55">McMurdie and Holmes, 2013</xref>), specifically the estimate_distance and distance function, were used to calculate alpha and beta diversity measures for replicates and stool/rectal swab samples. The alpha diversity measures included are observed richness (<xref ref-type="bibr" rid="B25">Fisher et al., 1943</xref>), Shannon&#x2019;s index (<xref ref-type="bibr" rid="B71">Shannon, 1948</xref>) and Simpson&#x2019;s index (<xref ref-type="bibr" rid="B73">Simpson, 1949</xref>). Bray Curtis (<xref ref-type="bibr" rid="B11">Bray and Curtis, 1957</xref>) and Jaccard&#x2019;s (<xref ref-type="bibr" rid="B51">Ludwig and Reynolds, 1988</xref>) distances were the beta diversity measures included in our analysis. Plot_richness and plot_ordination functions were used to plot alpha and beta diversity measures, respectively. Paired Wilcoxon tests were used to compare alpha diversity measures between pairs of within-run replicates and to identify differences between pairs of between-run replicates. In order to determine whether beta diversity measures were reproducible between technical replicate pairs, distance-based intraclass correlation coefficients (dICCs) were calculated separately for within-run and between-run replicates (<xref ref-type="bibr" rid="B20">Chen and Zhang, 2022</xref>). Paired Wilcoxon tests and dICCs were similarly used to compare pairs of stool and rectal swab samples collected from the same infants.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<p>Analysis of mock controls and technical replicates was carried out to assess the use of the xGen Amplicon kits and the SNAPP-py3 pipeline as a multivariate 16S rRNA sequencing approach for achieving accurate and reproducible species-level resolution. Furthermore, the similarity of samples collected using different sample collection techniques was investigated by comparing pairs of baseline stool and rectal swab samples from the same participants.</p>
<sec id="s3-1">
<title>3.1 DNA extraction reliability</title>
<p>All eight expected bacterial species were detected in four of our seven mock extraction controls (<xref ref-type="table" rid="T1">Table 1</xref>). <italic>Bacillus subtilis</italic> was not detected to the species level in two controls (<xref ref-type="table" rid="T2">Table 2</xref>; <xref ref-type="fig" rid="F3">Figure 3A</xref>), however classification to a genus level (<italic>Bacillus</italic>) was achieved (<xref ref-type="sec" rid="s11">Supplementary Table S1</xref>). <italic>Listeria monocytogenes</italic> was not detected even at a genus level in the control from run 2, plate 1 (<xref ref-type="sec" rid="s11">Supplementary Table S1</xref>). For the three controls in which we were unable to detect all eight species, the total percentage of sequencing data correctly classified as expected mock species was consequently lower (<xref ref-type="table" rid="T2">Table 2</xref>). Sensitivity scores were over 0.88 for all controls. Precision scores were lower &#x2013; particularly for the control on run 1, plate 1, which had a score of 0.32. This was driven by a high number of false positive results in this control. A median F-score of 0.84 (range of 0.47&#x2013;1.00) was obtained for the mock extraction controls (<xref ref-type="table" rid="T1">Table 1</xref>).</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>A summary of performance and accuracy measures at a species level for mock cell controls containing eight known bacterial species.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="left">R1P1<break/>Zymoex</th>
<th align="left">R1P2<break/>Zymoex</th>
<th align="left">R1P3<break/>Zymoex</th>
<th align="left">R1P4<break/>Zymoex</th>
<th align="left">R2P1<break/>Zymoex</th>
<th align="left">R2P2<break/>Zymoex</th>
<th align="left">R2P3<break/>Zymoex</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">True positives</td>
<td align="left">7</td>
<td align="left">8</td>
<td align="left">8</td>
<td align="left">7</td>
<td align="left">7</td>
<td align="left">8</td>
<td align="left">8</td>
</tr>
<tr>
<td align="left">False positives</td>
<td align="left">15</td>
<td align="left">3</td>
<td align="left">3</td>
<td align="left">0</td>
<td align="left">3</td>
<td align="left">0</td>
<td align="left">2</td>
</tr>
<tr>
<td align="left">False negatives</td>
<td align="left">1</td>
<td align="left">0</td>
<td align="left">0</td>
<td align="left">1</td>
<td align="left">1</td>
<td align="left">0</td>
<td align="left">0</td>
</tr>
<tr>
<td align="left">Precision</td>
<td align="left">0.32</td>
<td align="left">0.73</td>
<td align="left">0.73</td>
<td align="left">1.00</td>
<td align="left">0.70</td>
<td align="left">1.00</td>
<td align="left">0.80</td>
</tr>
<tr>
<td align="left">Sensitivity</td>
<td align="left">0.88</td>
<td align="left">1.00</td>
<td align="left">1.00</td>
<td align="left">0.88</td>
<td align="left">0.88</td>
<td align="left">1.00</td>
<td align="left">1.00</td>
</tr>
<tr>
<td align="left">F-score</td>
<td align="left">0.47</td>
<td align="left">0.84</td>
<td align="left">0.84</td>
<td align="left">0.93</td>
<td align="left">0.78</td>
<td align="left">1.00</td>
<td align="left">0.89</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>R&#x23;, run number; P&#x23;, plate number.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>The theoretical abundances and relative abundances of the eight expected bacterial species in mock cell controls given as percentages. The median observed/expected (O/E) ratios are also provided. The total percentage of sequences correctly classified as one of the expected species is given for each control.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Species</th>
<th align="left">Theoretical abundance (%)</th>
<th align="left">R1P1<break/>Zymoex</th>
<th align="left">R1P2<break/>Zymoex</th>
<th align="left">R1P3<break/>Zymoex</th>
<th align="left">R1P4<break/>Zymoex</th>
<th align="left">R2P1<break/>Zymoex</th>
<th align="left">R2P2<break/>Zymoex</th>
<th align="left">R2P3<break/>Zymoex</th>
<th align="left">O/E ratio [median (range)]</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<italic>Limosilactobacillus fermentum</italic>
</td>
<td align="left">18.4</td>
<td align="left">18.21</td>
<td align="left">18.27</td>
<td align="left">17.99</td>
<td align="left">17.64</td>
<td align="left">18.60</td>
<td align="left">16.60</td>
<td align="left">18.02</td>
<td align="left">0.98 (0.90&#x2013;1.01)</td>
</tr>
<tr>
<td align="left">
<italic>Bacillus subtilis</italic>
</td>
<td align="left">17.4</td>
<td align="left">0.00</td>
<td align="left">19.55</td>
<td align="left">19.71</td>
<td align="left">0.00</td>
<td align="left">18.76</td>
<td align="left">17.63</td>
<td align="left">18.55</td>
<td align="left">1.07 (0.00&#x2013;1.13)</td>
</tr>
<tr>
<td align="left">
<italic>Staphylococcus aureus</italic>
</td>
<td align="left">15.5</td>
<td align="left">10.82</td>
<td align="left">9.13</td>
<td align="left">9.13</td>
<td align="left">11.47</td>
<td align="left">10.96</td>
<td align="left">9.31</td>
<td align="left">10.75</td>
<td align="left">0.69 (0.59&#x2013;0.74)</td>
</tr>
<tr>
<td align="left">
<italic>Listeria monocytogenes</italic>
</td>
<td align="left">14.1</td>
<td align="left">4.41</td>
<td align="left">4.47</td>
<td align="left">4.59</td>
<td align="left">4.97</td>
<td align="left">0.00</td>
<td align="left">4.58</td>
<td align="left">4.67</td>
<td align="left">0.32 (0.00&#x2013;0.35)</td>
</tr>
<tr>
<td align="left">
<italic>Salmonella enterica</italic>
</td>
<td align="left">10.4</td>
<td align="left">17.58</td>
<td align="left">18.90</td>
<td align="left">17.61</td>
<td align="left">16.37</td>
<td align="left">10.95</td>
<td align="left">20.39</td>
<td align="left">19.22</td>
<td align="left">1.69 (1.05&#x2013;1.96)</td>
</tr>
<tr>
<td align="left">
<italic>Escherichia/Shigella coli</italic>
</td>
<td align="left">10.1</td>
<td align="left">17.06</td>
<td align="left">18.29</td>
<td align="left">18.26</td>
<td align="left">18.89</td>
<td align="left">20.57</td>
<td align="left">19.26</td>
<td align="left">20.50</td>
<td align="left">1.87 (1.69&#x2013;2.04)</td>
</tr>
<tr>
<td align="left">
<italic>Enterococcus faecalis</italic>
</td>
<td align="left">9.9</td>
<td align="left">5.39</td>
<td align="left">4.53</td>
<td align="left">4.53</td>
<td align="left">5.28</td>
<td align="left">5.36</td>
<td align="left">4.96</td>
<td align="left">4.81</td>
<td align="left">0.50 (0.46&#x2013;0.54)</td>
</tr>
<tr>
<td align="left">
<italic>Pseudomonas aeruginosa</italic>
</td>
<td align="left">4.2</td>
<td align="left">7.37</td>
<td align="left">6.52</td>
<td align="left">7.77</td>
<td align="left">4.50</td>
<td align="left">4.63</td>
<td align="left">7.25</td>
<td align="left">3.40</td>
<td align="left">1.55 (0.81&#x2013;1.85)</td>
</tr>
<tr>
<td align="left">Other</td>
<td align="left">0</td>
<td align="left">19.16</td>
<td align="left">0.34</td>
<td align="left">0.42</td>
<td align="left">20.89</td>
<td align="left">10.18</td>
<td align="left">0.02</td>
<td align="left">0.09</td>
<td align="left">-</td>
</tr>
<tr>
<td align="left">% Correctly classified</td>
<td align="left">100</td>
<td align="left">80.84</td>
<td align="left">99.66</td>
<td align="left">99.58</td>
<td align="left">79.11</td>
<td align="left">89.82</td>
<td align="left">99.98</td>
<td align="left">99.91</td>
<td align="left">-</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>R&#x23;, run number; P&#x23;, plate number.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Relative abundances for <bold>(A)</bold> mock cell and <bold>(B)</bold> mock DNA controls; Legend: s &#x3d; species, g &#x3d; genus level classification. R&#x23; &#x3d; run number; P&#x23; &#x3d; plate number. The legend only includes taxa of interest to the species or genus level.</p>
</caption>
<graphic xlink:href="fbinf-05-1484113-g003.tif"/>
</fig>
<p>The percentage abundances of species in these controls did not accurately follow the order of theoretical abundances for some species (<xref ref-type="table" rid="T2">Table 2</xref>; <xref ref-type="fig" rid="F4">Figure 4A</xref>). In particular, the relative abundances of <italic>L. monocytogenes</italic> were well below the theoretical abundances suggested by ZymoBIOMICS at both a species and genus level. This is emphasised by the low median Observed/Expected (O/E) ratio of 0.32 at a species level. Similarly, <italic>Enterococcus faecalis</italic> and <italic>Staphylococcus aureus</italic> had O/E ratios well below the value of 1. The relative abundances of <italic>Escherichia coli</italic> and <italic>Salmonella enterica</italic> were greater than expected, with a range of O/E ratios all lying well above the value of 1.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Box and whisker plots showing the relative abundance percentages of the eight expected bacterial species across <bold>(A)</bold> mock extraction controls and <bold>(B)</bold> mock sequencing controls; theoretical abundances shown as dots; outliers shown as crosses.</p>
</caption>
<graphic xlink:href="fbinf-05-1484113-g004.tif"/>
</fig>
</sec>
<sec id="s3-2">
<title>3.2 Sequencing reliability</title>
<p>Among the mock sequencing controls, the eight anticipated bacterial species were detected in five of the seven controls (<xref ref-type="table" rid="T3">Table 3</xref>; <xref ref-type="fig" rid="F3">Figure 3B</xref>). The overall percentages of sequences correctly classified as expected species were slightly lower for these controls compared to the mock extraction controls (<xref ref-type="table" rid="T4">Table 4</xref>). For the run 2 plate 1 control the prevalence of <italic>S. enterica</italic> was particularly low, and this was not resolved at the genus level (<xref ref-type="sec" rid="s11">Supplementary Table S2</xref>). <italic>B. subtilis</italic> again was not detected at a species level for two of these controls (<xref ref-type="table" rid="T3">Table 3</xref>). Precision scores for the mock sequencing controls ranged from 0.50 and up, while sensitivity was greater than 0.88 for all controls. F-scores had a median of 0.80 (range of 0.67&#x2013;0.94).</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>A summary of performance and accuracy measures at a species level for mock DNA controls containing eight known bacterial species.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="left">R1P1<break/>Zymoseq</th>
<th align="left">R1P2<break/>Zymoseq</th>
<th align="left">R1P3<break/>Zymoseq</th>
<th align="left">R1P4<break/>Zymoseq</th>
<th align="left">R2P1<break/>Zymoseq</th>
<th align="left">R2P2<break/>Zymoseq</th>
<th align="left">R2P3<break/>Zymoseq</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">True positives</td>
<td align="left">8</td>
<td align="left">8</td>
<td align="left">7</td>
<td align="left">8</td>
<td align="left">8</td>
<td align="left">7</td>
<td align="left">8</td>
</tr>
<tr>
<td align="left">False positives</td>
<td align="left">4</td>
<td align="left">8</td>
<td align="left">4</td>
<td align="left">6</td>
<td align="left">1</td>
<td align="left">0</td>
<td align="left">3</td>
</tr>
<tr>
<td align="left">False negatives</td>
<td align="left">0</td>
<td align="left">0</td>
<td align="left">1</td>
<td align="left">0</td>
<td align="left">0</td>
<td align="left">1</td>
<td align="left">0</td>
</tr>
<tr>
<td align="left">Precision</td>
<td align="left">0.67</td>
<td align="left">0.50</td>
<td align="left">0.64</td>
<td align="left">0.57</td>
<td align="left">0.89</td>
<td align="left">1.00</td>
<td align="left">0.73</td>
</tr>
<tr>
<td align="left">Sensitivity</td>
<td align="left">1.00</td>
<td align="left">1.00</td>
<td align="left">0.88</td>
<td align="left">1.00</td>
<td align="left">1.00</td>
<td align="left">0.88</td>
<td align="left">1.00</td>
</tr>
<tr>
<td align="left">F-score</td>
<td align="left">0.80</td>
<td align="left">0.67</td>
<td align="left">0.74</td>
<td align="left">0.73</td>
<td align="left">0.94</td>
<td align="left">0.93</td>
<td align="left">0.84</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>R&#x23;, run number; P&#x23;, plate number.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>The theoretical abundances and relative abundances of the eight expected bacterial species in mock DNA controls are given as percentages. The median observed/expected (O/E) ratios are also provided. The total percentage of sequences correctly classified as one of the expected species is given for each control.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Species</th>
<th align="left">Theoretical abundance (%)</th>
<th align="left">R1P1<break/>Zymoseq</th>
<th align="left">R1P2<break/>Zymoseq</th>
<th align="left">R1P3<break/>Zymoseq</th>
<th align="left">R1P4<break/>Zymoseq</th>
<th align="left">R2P1<break/>Zymoseq</th>
<th align="left">R2P2<break/>Zymoseq</th>
<th align="left">R2P3<break/>Zymoseq</th>
<th align="left">O/E ratio [median (range)]</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<italic>Limosilactobacillus fermentum</italic>
</td>
<td align="left">18.4</td>
<td align="left">13.08</td>
<td align="left">12.60</td>
<td align="left">13.32</td>
<td align="left">11.94</td>
<td align="left">13.88</td>
<td align="left">13.16</td>
<td align="left">12.86</td>
<td align="left">0.71 (0.65&#x2013;0.75)</td>
</tr>
<tr>
<td align="left">
<italic>Bacillus subtilis</italic>
</td>
<td align="left">17.4</td>
<td align="left">17.89</td>
<td align="left">17.84</td>
<td align="left">0.00</td>
<td align="left">17.05</td>
<td align="left">17.81</td>
<td align="left">0.00</td>
<td align="left">17.99</td>
<td align="left">1.02 (0.00&#x2013;1.03)</td>
</tr>
<tr>
<td align="left">
<italic>Staphylococcus aureus</italic>
</td>
<td align="left">15.5</td>
<td align="left">13.84</td>
<td align="left">13.34</td>
<td align="left">13.83</td>
<td align="left">14.04</td>
<td align="left">13.95</td>
<td align="left">14.22</td>
<td align="left">14.03</td>
<td align="left">0.90 (0.86&#x2013;0.92)</td>
</tr>
<tr>
<td align="left">
<italic>Listeria monocytogenes</italic>
</td>
<td align="left">14.1</td>
<td align="left">15.27</td>
<td align="left">14.76</td>
<td align="left">14.54</td>
<td align="left">15.41</td>
<td align="left">15.85</td>
<td align="left">15.95</td>
<td align="left">16.05</td>
<td align="left">1.09 (1.03&#x2013;1.14)</td>
</tr>
<tr>
<td align="left">
<italic>Salmonella enterica</italic>
</td>
<td align="left">10.4</td>
<td align="left">11.49</td>
<td align="left">12.04</td>
<td align="left">13.12</td>
<td align="left">6.32</td>
<td align="left">0.01</td>
<td align="left">11.23</td>
<td align="left">9.06</td>
<td align="left">1.08 (0.00&#x2013;1.26)</td>
</tr>
<tr>
<td align="left">
<italic>Escherichia/Shigella coli</italic>
</td>
<td align="left">10.1</td>
<td align="left">11.59</td>
<td align="left">12.40</td>
<td align="left">10.84</td>
<td align="left">11.62</td>
<td align="left">12.18</td>
<td align="left">11.20</td>
<td align="left">11.64</td>
<td align="left">1.15 (1.07&#x2013;1.23)</td>
</tr>
<tr>
<td align="left">
<italic>Enterococcus faecalis</italic>
</td>
<td align="left">9.9</td>
<td align="left">11.98</td>
<td align="left">11.18</td>
<td align="left">11.50</td>
<td align="left">11.12</td>
<td align="left">11.89</td>
<td align="left">12.40</td>
<td align="left">11.76</td>
<td align="left">1.19 (1.12&#x2013;1.25)</td>
</tr>
<tr>
<td align="left">
<italic>Pseudomonas aeruginosa</italic>
</td>
<td align="left">4.2</td>
<td align="left">4.44</td>
<td align="left">5.10</td>
<td align="left">4.64</td>
<td align="left">4.94</td>
<td align="left">3.72</td>
<td align="left">4.09</td>
<td align="left">3.39</td>
<td align="left">1.06 (0.81&#x2013;1.21)</td>
</tr>
<tr>
<td align="left">Other</td>
<td align="left">0</td>
<td align="left">0.42</td>
<td align="left">0.74</td>
<td align="left">18.20</td>
<td align="left">7.54</td>
<td align="left">10.72</td>
<td align="left">17.74</td>
<td align="left">3.21</td>
<td align="left">-</td>
</tr>
<tr>
<td align="left">% Correctly classified</td>
<td align="left"/>
<td align="left">99.58</td>
<td align="left">99.26</td>
<td align="left">81.80</td>
<td align="left">92.46</td>
<td align="left">89.28</td>
<td align="left">82.26</td>
<td align="left">96.79</td>
<td align="left">-</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>R&#x23;, run number; P&#x23;, plate number.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The relative abundances of species in the mock sequencing controls more closely matched the expected abundances than was observed for the mock extraction controls (<xref ref-type="table" rid="T4">Table 4</xref>; <xref ref-type="fig" rid="F4">Figure 4B</xref>), as seen by the O/E ratios being closer to 1. In these controls the abundance of <italic>Limosilactobacillus fermentum</italic> was well below the theoretical threshold expected. The O/E ratios for this species and <italic>S. aureus</italic> were consistently less than 1. Whereas <italic>E. faecalis</italic>, <italic>E. coli</italic> and <italic>L. monocytogenes</italic> had O/E ratio ranges above 1.</p>
<p>Although the inclusion of a batch effect correction step was trialled, substantial differences in the relative abundances of the various species across the controls were observed compared to when no batch effect correction was done (<xref ref-type="sec" rid="s11">Supplementary Figure S1</xref>). Similarly, stricter decontamination thresholds led to poorer reproducibility in mock controls.</p>
</sec>
<sec id="s3-3">
<title>3.3 Within-run and between-run reproducibility</title>
<p>Similar patterns in the relative abundance of species could be seen when comparing pairs of within-run and between-run repeats (<xref ref-type="sec" rid="s11">Supplementary Figures S2, S3</xref>). When comparing alpha diversity of within-run repeats using paired Wilcoxon tests, we found no evidence to indicate differences at a species level in the Observed richness [95% confidence interval (CI) (&#x2212;3.50, 5.50); p &#x3d; 0.587], Shannon&#x2019;s index [95% CI (&#x2212;0.09, 0.12); p &#x3d; 0.782] or Simpson&#x2019;s index [95% CI (&#x2212;0.02, 0.01); p &#x3d; 0.487] between these pairs of technical replicates (<xref ref-type="fig" rid="F5">Figure 5</xref>). Furthermore, when comparing the beta diversity distance matrices of these within-run technical replicate pairs (<xref ref-type="fig" rid="F6">Figure 6</xref>), we observed a good level of reproducibility for Bray Curtis as seen by a distance-based intraclass correlation coefficient (dICC) value of 0.940 and Jaccard&#x2019;s distance showed good, albeit lower, reproducibility with a dICC of 0.762. Similarly, we found no differences between these technical replicates when looking at their alpha and beta diversity measures at a genus level.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>
<bold>(A)</bold> Observed richness, <bold>(B)</bold> Shannon&#x2019;s index and <bold>(C)</bold> Simpson&#x2019;s index for within-run technical replicates at species-level classification. s1, sample 1; s2, sample 2.</p>
</caption>
<graphic xlink:href="fbinf-05-1484113-g005.tif"/>
</fig>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>A comparison of <bold>(A)</bold> Jaccard and <bold>(B)</bold> Bray Curtis beta diversity measures between pairs of within-run technical replicates at species-level classification.</p>
</caption>
<graphic xlink:href="fbinf-05-1484113-g006.tif"/>
</fig>
<p>A comparison of alpha and beta diversity measures for between-run technical replicates similarly found no clear differences between these pairs at a species level (<xref ref-type="fig" rid="F7">Figure 7</xref>). Paired Wilcoxon tests comparing Observed species [95% CI (&#x2212;2.00, 13.00); p &#x3d; 0.362], Shannon&#x2019;s index [95% CI (&#x2212;0.05, 0.49); p &#x3d; 0.195] and Simpson&#x2019;s index [95% CI (&#x2212;0.004, 0.11); p &#x3d; 0.148] did not motivate for the existence of differences in alpha diversity measures. Moreover, dICC results showed good reproducibility for Bray Curtis (dICC &#x3d; 0.899) and moderate reproducibility for Jaccard&#x2019;s distance (dICC &#x3d; 0.597) (<xref ref-type="fig" rid="F8">Figure 8</xref>).</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>
<bold>(A)</bold> Observed richness, <bold>(B)</bold> Shannon&#x2019;s index and <bold>(C)</bold> Simpson&#x2019;s index for between-run technical replicates at species-level classification. p&#x23; &#x3d; run and plate number.</p>
</caption>
<graphic xlink:href="fbinf-05-1484113-g007.tif"/>
</fig>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>A comparison of <bold>(A)</bold> Jaccard and <bold>(B)</bold> Bray Curtis beta diversity measures between pairs of between-run technical replicates at species-level classification.</p>
</caption>
<graphic xlink:href="fbinf-05-1484113-g008.tif"/>
</fig>
</sec>
<sec id="s3-4">
<title>3.4 Interchangeability of stool and rectal swab samples</title>
<p>To assess whether stool and rectal swab samples may be combined for analysis, we compared 26 sample pairs collected at 0&#x2013;5 weeks after birth. The relative abundances of species did not display similar patterns across pairs of samples (<xref ref-type="sec" rid="s11">Supplementary Figure S4</xref>) and alpha diversity was found to differ when running paired Wilcoxon tests. Stool and swab samples from the same participants differed at a species level in terms of Observed species (p &#x3c; 0.001) and Shannon&#x2019;s index (p &#x3d; 0.027), with no differences in Simpson&#x2019;s index (p &#x3d; 0.394) found (<xref ref-type="fig" rid="F9">Figure 9</xref>). A comparison of beta diversity measures at a species level found that while Bray Curtis measures were moderately reliable between the pairs of stool and swab samples (dICC &#x3d; 0.684), there was poor reliability when comparing their Jaccard&#x2019;s distances (dICC &#x3d; 0.310) (<xref ref-type="fig" rid="F10">Figure 10</xref>).</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>
<bold>(A)</bold> Observed richness, <bold>(B)</bold> Shannon&#x2019;s index and <bold>(C)</bold> Simpson&#x2019;s index for pairs of stool and rectal swab samples collected from the same infants using data classified to a species level. p&#x23; &#x3d; run and plate number.</p>
</caption>
<graphic xlink:href="fbinf-05-1484113-g009.tif"/>
</fig>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>Principal coordinates analysis plots showing <bold>(A)</bold> Jaccard and <bold>(B)</bold> Bray Curtis beta diversity measures for pairs of stool and rectal swab samples collected from the same infants at species-level classification.</p>
</caption>
<graphic xlink:href="fbinf-05-1484113-g010.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<p>We have outlined a short-read sequencing protocol which can be used to carry out species-level analysis. Our findings, following analysis of mock controls and technical replicates, indicate that the kits and analytical pipelines used in this study can effectively enable species-level classification and provide reproducible results within and across sequencing plates.</p>
<p>When assessing data from different sample collection approaches using this protocol, our results indicate that care needs to be taken as stool and swab samples collected from the same participant are not comparable at a species level.</p>
<sec id="s4-1">
<title>4.1 Multiple variable regions of 16S rRNA enable adequate species-level analysis</title>
<p>We were able to obtain relatively good species-level resolution in terms of sensitivity using a nearly complete 16S rRNA sequence, which the SNAPP-py3 pipeline developers refer to as a &#x2018;consensus&#x2019; sequence. The eight species expected to be found in the ZymoBIOMICS mock controls were not consistently detected to a species level in all mock controls across the seven sequencing plates, suggesting that there is still room for improvement in terms of reproducibility. The main species which were not detected in all controls are <italic>B. subtilis</italic> and <italic>L. monocytogenes</italic>. These species are gram-positive, containing a strong wall of peptidoglycan which can be a challenge to lyse, as has been presented in previous literature (<xref ref-type="bibr" rid="B21">Claassen-Weitz et al., 2020</xref>). ZymoBIOMICS have intentionally developed these mock controls to include both gram-negative and gram-positive species to enable researchers to identify inconsistencies and to optimise their lysis protocols (<xref ref-type="bibr" rid="B89">ZymoBIOMICS, 2024</xref>). Moreover, research indicates that the primers used in library preparation may have a greater affinity for some species compared to others (<xref ref-type="bibr" rid="B44">Klindworth et al., 2013</xref>), which may also contribute to false negatives in some controls. Our results indicate that the methods used are capable of detecting all species, as seen in several of our mock controls, however future optimisation of the protocol will be required.</p>
<p>A previous study comparing the use of different variable regions in Illumina Miseq sequencing, found that using 1-2 variable regions could at best identify 16 of 20 (80%) mock control species correctly (<xref ref-type="bibr" rid="B27">Fouhy et al., 2016</xref>). In five mock controls, we detected 7 of 8 mock species (88%), yet for the remaining nine controls we detected all the expected species (100%). Thus, our results would indicate that using all nine variable regions of the 16S rRNA gene improves accuracy compared to methods that look at only a couple of variable regions. Other short-read multiple variable region methods have been able to identify all taxa within mock controls containing a greater number of species (<xref ref-type="bibr" rid="B29">Fuks et al., 2018</xref>; <xref ref-type="bibr" rid="B68">Schriefer et al., 2018</xref>), however in these studies there were also mock controls in which not all species were identified. <xref ref-type="bibr" rid="B68">Schriefer et al. (2018)</xref> suggested that the depth to which sequencing was done may play a role, however an increase of sequencing depth by 10-fold ultimately had no impact on their results. Therefore, the approach we present requires further optimisation to ensure that 100% sensitivity is consistently achieved and to assess whether this approach can perform at a similar level to other multiple variable region tools when using more complex mock controls.</p>
<p>When focusing only on sequences that were classified to a species level, sensitivity was high in both the mock extraction and mock sequencing controls. However, our results indicate that high sensitivity comes at the expense of obtaining poorer precision in some controls. The overall performance for mock extraction and mock sequencing controls were median F-scores of 0.84 and 0.80 respectively. These results are comparable with other research that obtained F-scores greater than 0.80 using a bioinformatics tool for analysing sequencing data, for which the authors described their results as being indicative of good performance (<xref ref-type="bibr" rid="B60">&#xd6;zkurt et al., 2022</xref>). Poor precision scores in our case were often driven by false positive taxa that were present in very low relative abundances. Thus, removing rare taxa could yield even better results. <xref ref-type="bibr" rid="B68">Schriefer et al. (2018)</xref> and <xref ref-type="bibr" rid="B29">Fuks et al. (2018)</xref>, also reported false positives at low abundances for their multiple variable region approaches.</p>
<p>As our measure of precision is based on the presence or absence of expected taxa, it is quite strict in how it penalises false positives. It is not weighted according to the abundances at which these occur. Therefore, it is important for us to look at the precision scores in conjunction with our abundance-based measures, to get a complete picture of the performance of the methodology used in this paper. The R1P1 mock extraction control displayed poor precision due to having a particularly high number of false positives. However, as these occurred at such low abundances, false positives had little influence on our abundance-related analysis. Rather, we found that the inability to detect all expected species (i.e., false negative results) in our controls was the main factor responsible for differences in observed abundances compared to expected abundances.</p>
<p>The overall percentages of sequences that were classified as expected taxa and to a species level, ranged between 79.11% and 99.98% for mock extraction controls, and between 81.80% and 99.58% for mock sequencing controls. In a study by <xref ref-type="bibr" rid="B75">Szoboszlay et al. (2023)</xref> the authors found that between 58.9%&#x2013;68.9% of reads obtained from Illumina sequencing of the V4 region could be correctly classified to a species level. This improved if rare taxa were excluded. Nanopore sequencing of the entire 16S rRNA gene could yield classification of over 81% of reads to a species level (<xref ref-type="bibr" rid="B75">Szoboszlay et al., 2023</xref>). Therefore, although we did not remove rare taxa, our protocol could compete with a full-length sequencing approach and perform better than using a single variable region for short-read sequencing. Ultimately, our results indicate that the RDP classifier is able to accurately identify the expected taxa in our controls.</p>
<p>The relative abundances of species in our mock extraction controls sometimes diverged from the theoretical abundances outlined by the suppliers (<xref ref-type="bibr" rid="B88">ZymoBIOMICS, 2022</xref>), indicating that there may be some bias introduced during the extraction process. O/E ratios provided a quantitative means of assessing the bias introduced in each control. In particular, they assisted with identifying the species for which observed abundances differed most from theoretical abundances and showed that there was greater bias in the mock extraction controls.</p>
<p>There are various stages in 16S rRNA sequencing where prejudice can arise, favouring certain taxa or altering the relative composition of a sample (<xref ref-type="bibr" rid="B58">Nearing et al., 2021</xref>). Studies have found that DNA extraction kits and methods can differ in terms of their ability to extract DNA from gram negative and gram positive bacteria, indicating that the results of a study can be influenced depending on which kit and extraction method are used (<xref ref-type="bibr" rid="B80">Videnska et al., 2019</xref>; <xref ref-type="bibr" rid="B83">Yuan et al., 2012</xref>). The mock sequencing controls (for which already-extracted DNA was obtained from ZymoBIOMICS) more closely resembled the expected abundances. <italic>L. fermentum</italic> differed substantially from the theoretical abundance in the sequencing controls &#x2013; unlike the extraction controls for which this particular species had followed the expected abundance more closely. This implies that, for the most part, there is minimal bias introduced at the sequencing step. However, there may be some bias specifically in sequencing <italic>L. fermentum</italic>. Our findings complement other studies which have shown that the extraction and amplification of DNA is particularly prone to prejudicing results, while there is less bias introduced at the sequencing stage of microbial research (<xref ref-type="bibr" rid="B12">Brooks et al., 2015</xref>; <xref ref-type="bibr" rid="B48">Lee et al., 2012</xref>). Similar to our findings, previous research using a short-read multiple variable region approach, also found that the observed abundance of taxa did not always accurately match the expected abundances (<xref ref-type="bibr" rid="B29">Fuks et al., 2018</xref>). The authors also suggested that bias may be introduced during the amplification stage of library preparation.</p>
<p>Our findings suggest that using this multivariate analysis approach might provide a means to improve the ability of short-read 16S rRNA sequencing studies to study microbial samples at a species level. This approach would benefit from additional ways to minimise bias &#x2013; particularly in the DNA extraction process.</p>
<p>We considered including a batch effect correction step in the processing of our data. The goal was to reduce bias introduced due to samples being sequenced on different plates and in different runs. However, when visualising the relative abundances of mock sequencing controls, we found that batch effect correction led to far less consistent results across the controls. <xref ref-type="bibr" rid="B23">Davis et al. (2018)</xref> reported that the decontam package corrects for batch effects. As a result, it may be redundant to carry out both decontamination and batch effect correction. Moreover, as our protocol included a plate-wise decontamination step, batch effects would have already been taken into account here. Thus, based on our findings, we considered an additional batch effect correction step excessive and problematic &#x2013; leading us to exclude this step. Similarly, we found that using a very stringent threshold for decontamination also had a negative impact on our analysis. As such, we would suggest that care needs to be taken when doing additional processing of data, to ensure that bias is not introduced to the data by overcorrecting for contaminants and batch effects. Researchers should take care in deciding whether batch effect correction is necessary if their pipeline includes a decontamination step, which removes contaminants according to batches.</p>
</sec>
<sec id="s4-2">
<title>4.2 Short-read multiple variable region analysis can achieve good reproducibility</title>
<p>Observed richness provides an indication of the number of different taxa within samples, while Shannon&#x2019;s and Simpson&#x2019;s indices additionally account for how equally represented these taxa are within samples (evenness) (<xref ref-type="bibr" rid="B43">Kers and Saccenti, 2021</xref>). Based on our results we have no reason to reject the null hypothesis that there is no difference in alpha diversity measures between within-run technical replicate pairs.</p>
<p>Jaccard&#x2019;s distance quantifies how diversity varies between samples based on whether or not taxa are present, while Bray Curtis factors in the abundance of taxa (<xref ref-type="bibr" rid="B43">Kers and Saccenti, 2021</xref>; <xref ref-type="bibr" rid="B69">Schroeder and Jenkins, 2018</xref>). More confidence is usually placed in Bray Curtis compared to Jaccard&#x2019;s distance (<xref ref-type="bibr" rid="B69">Schroeder and Jenkins, 2018</xref>). The similarity between these beta diversity measures for replicate sample pairs was assessed in terms of dICC. dICC is a distance measure which has been established specifically for microbiome data, building on the concept of intraclass correlation coefficients, to look at the similarity between replicate samples (<xref ref-type="bibr" rid="B20">Chen and Zhang, 2022</xref>). Higher ICC values are indicative of strong similarity between measures (<xref ref-type="bibr" rid="B45">Koo and Li, 2016</xref>), in our case diversity measures for microbial sample pairs. An ICC of 0.5, however, shows moderate reproducibility and a value of &#x3c;0.5 suggests poor similarity (<xref ref-type="bibr" rid="B45">Koo and Li, 2016</xref>). Thus, our results in which we get dICC values of 0.940 and 0.762 for Bray Curtis and Jaccard respectively, indicate that we get good beta diversity reproducibility when running samples in duplicate on the same plate. Likewise, the diversity of between-run technical replicates was similar with dICC values of 0.899 and 0.597. This suggests that we do not have significant batch effects across sequencing plates of the same run, or across different runs, when following the protocols set out in this study for sequencing and processing short-read multivariate 16S rRNA data. It should be noted that we only had one pair of replicates across runs and this was grouped with the between-plate replicates in what we referred to as our &#x201c;between-run replicates.&#x201d; Collectively our findings indicate that there is no batch effect introduced. The sequencing and processing of infant stool and rectal swab samples is consistent, providing confidence in the analysis of the microbiome datasets generated using this protocol.</p>
</sec>
<sec id="s4-3">
<title>4.3 Stool and rectal swab collection approaches are not interchangeable</title>
<p>The implementation of a new kit and pipeline for analysing both stool and rectal swab samples in this study warranted investigation. Stool and rectal swab samples collected from the same infants and at the same time point were found to differ for several diversity measures. Contrary to our findings, studies have suggested that rectal swab samples can be used in place of stool samples and provide a reliable representative measure (<xref ref-type="bibr" rid="B64">Radhakrishnan et al., 2023</xref>; <xref ref-type="bibr" rid="B66">Reyman et al., 2019</xref>; <xref ref-type="bibr" rid="B6">Bassis et al., 2017</xref>). However, a couple of studies exploring diversity and functional roles of the microbiome have previously reported differences between stool and swab samples (<xref ref-type="bibr" rid="B72">Short et al., 2021</xref>; <xref ref-type="bibr" rid="B74">Sun et al., 2021</xref>). <xref ref-type="bibr" rid="B72">Short et al. (2021)</xref> specifically found that beta diversity differed between these sample types, while <xref ref-type="bibr" rid="B74">Sun et al. (2021)</xref> found that rectal swab samples varied from stool in terms of both alpha and beta diversity measures. Moreover, there have been indications that differences in the relative abundances of specific taxa may occur between different sample types (<xref ref-type="bibr" rid="B42">Jones et al., 2018</xref>). This would be particularly important to consider when comparing the prevalence of individual taxa between groups, such as when using differential abundance analysis.</p>
<p>Furthermore, a study by <xref ref-type="bibr" rid="B9">Bokulich et al. (2019)</xref> found that the length of time between the collection and processing of samples is an important factor to take into consideration when using rectal swab samples in sequencing studies. Although samples in their study were ultimately frozen, swabs were posted to the laboratory, while stool samples were immediately placed on ice until they could be collected and taken to the laboratory. Rectal swab samples received and processed after 48 h following sample collection did not accurately match stool samples &#x2013; with a higher proportion of <italic>Enterobacteriaceae</italic> being favoured (<xref ref-type="bibr" rid="B9">Bokulich et al., 2019</xref>). The stool and rectal swab samples in our study were collected together and stored at &#x2212;80&#xb0;C until DNA extraction and sequencing were done, and therefore we do not face the same challenges as <xref ref-type="bibr" rid="B9">Bokulich et al. (2019)</xref> in terms of the length of time for which samples remained at room temperature. However, it may be important to consider factors such as the time for which samples are stored in freezers, and the time between DNA extraction and sequencing in future work involving both stool and swab samples.</p>
<p>The fact that swabs were stored in Primestore, while stool was not, would suggest a substantial difference which requires consideration when using a combination of stool and swab samples in carrying out analysis. Several studies have found that bacterial compositions differ across different regions of entire stool samples (<xref ref-type="bibr" rid="B87">Zreloff et al., 2023</xref>; <xref ref-type="bibr" rid="B37">Huson et al., 2017</xref>; <xref ref-type="bibr" rid="B32">Gorzelak et al., 2015</xref>). Spectroscopic analysis has shown that patterns of metabolites can also vary substantially across different regions of stool (<xref ref-type="bibr" rid="B49">Liang et al., 2020</xref>; <xref ref-type="bibr" rid="B34">Gratton et al., 2016</xref>). Therefore, using only part of the sample does not provide a good characterisation of the entire sample. These studies highlight the importance of homogenising the entire stool sample in order to carry out sequencing (<xref ref-type="bibr" rid="B87">Zreloff et al., 2023</xref>; <xref ref-type="bibr" rid="B49">Liang et al., 2020</xref>; <xref ref-type="bibr" rid="B34">Gratton et al., 2016</xref>). As sequencing of rectal swab samples is not limited to only a component of the collected sample, this sample would be homogenous. This may explain the differences observed between stool and rectal swab sample pairs.</p>
</sec>
<sec id="s4-4">
<title>4.4 Limitations and future work</title>
<p>The work presented included only one mock extraction and one mock sequencing control on each sequencing plate. Future work could include duplicates of controls on each plate to better assess reproducibility of the controls. It would also expand the options available for statistically comparing these controls.</p>
<p>A limitation of this study is the inconsistent detection of all eight species in each mock control. We were able to detect all eight species in several controls, suggesting that the methods used for library preparation, sequencing and processing of the data are all capable of achieving consistent detection. Given that gram positive species particularly are not always detected (<xref ref-type="bibr" rid="B21">Claassen-Weitz et al., 2020</xref>), and these are more challenging to lyse, future work to optimise the lysis protocol should be done &#x2013; for example, assessing whether carrying out mechanical lysis for longer enables the consistent detection of all species. A study involving sequencing of the entire genome, has previously shown that sequencing depth may play a crucial role in whether or not all expected taxa are detected (<xref ref-type="bibr" rid="B62">Pereira-Marques et al., 2019</xref>). Sequencing depth is a variable that differed for each control and may explain the inconsistencies in our results. Sequencing was done based on the Illumina guidelines which recommend a library depth of at least 100,000 reads (<xref ref-type="bibr" rid="B39">ILLUMINA. 16S Metagenomic Sequencing Library Preparation, 2024</xref>). However, as we are using a pipeline which combines multiple variable region reads into consensus sequences, and as other short-read multiple variable region 16S rRNA studies have utilised greater sequencing depths (<xref ref-type="bibr" rid="B29">Fuks et al., 2018</xref>; <xref ref-type="bibr" rid="B68">Schriefer et al., 2018</xref>), future work could seek to optimise the sequencing depth required for this protocol in order to better identify all species in mock controls.</p>
<p>In this study we evaluated a single pipeline and protocol for analysing data from multiple variable region 16S rRNA sequencing, however in future this protocol would benefit from comparisons to other tools and pipelines. Moreover, different DNA extraction kits and amplicon sequencing kits could be compared to identify the best kits for carrying out multiple variable region analysis. In this analysis we have worked with compositional data, looking at the relative abundances of taxa. However, in future, it would be useful to explore absolute abundances for samples sequenced using this protocol. Moreover, we could explore whether other databases &#x2013; such as databases specific to the human gut microbiome &#x2013; might enable more effective and accurate classification of consensus sequences from the SNAPP-py3 pipeline.</p>
<p>Researchers intending to utilise this sequencing kit and pipeline should ideally strive (where possible) to use data from samples obtained through the same collection approach. In cases where a combination of sample collection approaches is used, analysis should be done to assess whether collection approaches influence microbial composition. If they do, sample collection approach should be controlled for in any analysis in which this data is used.</p>
<p>The goal of this paper was to assess the use of xGen kits and the SNAPP-py3 pipeline, to set the stage for an observational clinical study. Our assessment of this methodology is limited by the fact that we did not have access to a wider variety of datasets sequenced using the xGen kits. Future research should explore the use of this methodology in a greater selection of clinical settings and in the study of environmental microbiome samples. This might result in the approach being of greater relevance and interest to a broader audience.</p>
</sec>
<sec id="s4-5">
<title>4.5 Conclusion</title>
<p>The results of our 16S rRNA multiple variable region analysis, using short-read Illumina sequencing data from all 9 variable regions, show that there is promise for using this protocol for species-level analysis. We have identified areas for improvement and future work should assess whether this approach is comparable to other bioinformatics pipelines and tools that have been established for analysing multiple variable region short-read data. As we found differences in diversity between stool and swab sample pairs, future analysis of data which has been generated using this protocol should take this into account. Furthermore, our findings indicate that when using new sequencing kits and protocols to study both stool and swab samples, it would be advisable to do analysis to compare pairs of stool and rectal swabs from the same individual to confirm whether these yield comparable results.</p>
</sec>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The data presented in the study are deposited in the Sequence Read Archive (SRA) repository, BioProject accession number PRJNA1195741.</p>
</sec>
<sec sec-type="ethics-statement" id="s6">
<title>Ethics statement</title>
<p>This study involving humans was approved by the Human Research Ethics Committees at the University of Cape Town (801/2016 and 557/2020) and Stellenbosch University (M16/10/041). The study was conducted in accordance with the local legislation and institutional requirements. Written informed consent for participation in this study was provided by the participants&#x2019; legal guardians/next of kin. Written informed consent was obtained from the minor(s)&#x2019; legal guardian/next of kin for the publication of any potentially identifiable images or data included in this article.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>AG: Writing&#x2013;original draft, Formal Analysis, Visualization. FP: Writing&#x2013;review and editing, Methodology. FL: Writing&#x2013;review and editing, Supervision. AK: Funding acquisition, Writing&#x2013;review and editing. MK: Funding acquisition, Supervision, Writing&#x2013;review and editing, Conceptualization. MH: Funding acquisition, Supervision, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This work has been supported in part by the National Research Foundation of South Africa (Grant Number: MND200610529926, NRF Postgraduate Scholarships). The National Institutes of Health (NIH) (Fogarty International Center (FIC) and National Institute of Child Health and Human Development (NICHD) R01HD093578 and R01HD085813) provided funding for this research.</p>
</sec>
<ack>
<p>We would like to acknowledge and thank Dr Veronica Allen for assisting Dr Fadheela Patel with DNA extraction and the 16S rRNA sequencing. We thank Dr Samantha Fry and Thandiwe Hamana for organizing visits and the sometimes difficult task of sample collection. We thank Dr Farai Mberi and Caylin Mc Farlane for their assistance with sample transportation and storage. Our sincere thanks to Slindile Mbhele for all she did to arrange the storage and transportation of samples as project manager. Thank you also to Dr Benli Chai, the developer of the SNAPP-py3 pipeline, for providing assistance and advice. The original preprint for this manuscript can be accessed at <ext-link ext-link-type="uri" xlink:href="https://www.biorxiv.org/content/10.1101/2024.05.13.591068v1">https://www.biorxiv.org/content/10.1101/2024.05.13.591068v1</ext-link>. (<xref ref-type="bibr" rid="B90">Graham et al., 2024</xref>).</p>
</ack>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s11">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fbinf.2025.1484113/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fbinf.2025.1484113/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet3.docx" id="SM1" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="DataSheet2.zip" id="SM2" mimetype="application/zip" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="DataSheet1.docx" id="SM3" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Acinas</surname>
<given-names>S. G.</given-names>
</name>
<name>
<surname>Marcelino</surname>
<given-names>L. A.</given-names>
</name>
<name>
<surname>Klepac-Ceraj</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Polz</surname>
<given-names>M. F.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Divergence and redundancy of 16S rRNA sequences in genomes with multiple rrn operons</article-title>. <source>J. Bacteriol.</source> <volume>186</volume>, <fpage>2629</fpage>&#x2013;<lpage>2635</lpage>. <pub-id pub-id-type="doi">10.1128/jb.186.9.2629-2635.2004</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Amarasinghe</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zappia</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ritchie</surname>
<given-names>M. E.</given-names>
</name>
<name>
<surname>Gouil</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Opportunities and challenges in long-read sequencing data analysis</article-title>. <source>Genome Biol.</source> <volume>21</volume> (<issue>30</issue>), <fpage>1</fpage>&#x2013;<lpage>16</lpage>. <pub-id pub-id-type="doi">10.1186/s13059-020-1935-5</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Amir</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zeisel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zuk</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Elgart</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Stern</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shamir</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>High-resolution microbial community reconstruction by integrating short reads from multiple 16S rRNA regions</article-title>. <source>Nucleic acids Res.</source> <volume>41</volume>, <fpage>e205</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkt1070</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Andrews</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Fastqc A quality control tool for high throughput sequence data</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="http://www.bioinformatics.babraham.ac.uk/projects/fastqc">http://www.bioinformatics.babraham.ac.uk/projects/fastqc</ext-link>.</comment>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Balle</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Konstantinus</surname>
<given-names>I. N.</given-names>
</name>
<name>
<surname>Jaumdally</surname>
<given-names>S. Z.</given-names>
</name>
<name>
<surname>Havyarimana</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Lennard</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Esra</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Hormonal contraception alters vaginal microbiota and cytokines in South African adolescents in a randomized trial</article-title>. <source>Nat. Commun.</source> <volume>11</volume>, <fpage>5578</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-020-19382-9</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bassis</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Moore</surname>
<given-names>N. M.</given-names>
</name>
<name>
<surname>Lolans</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Seekatz</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Weinstein</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Young</surname>
<given-names>V. B.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Comparison of stool versus rectal swab samples and storage conditions on bacterial community profiles</article-title>. <source>BMC Microbiol.</source> <volume>17</volume> (<issue>78</issue>), <fpage>1</fpage>&#x2013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1186/s12866-017-0983-9</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bennato</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Martino</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Di Domenico</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ianni</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chai</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Di Marcantonio</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Metagenomic characterization and volatile compounds determination in rumen from saanen goat kids fed olive leaves</article-title>. <source>Veterinary Sci.</source> <volume>9</volume>, <fpage>452</fpage>. <pub-id pub-id-type="doi">10.3390/vetsci9090452</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bharti</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Grimm</surname>
<given-names>D. G.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Current challenges and best-practice protocols for microbiome analysis</article-title>. <source>Briefings Bioinforma.</source> <volume>22</volume>, <fpage>178</fpage>&#x2013;<lpage>193</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbz155</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bokulich</surname>
<given-names>N. A.</given-names>
</name>
<name>
<surname>Maldonado</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>D.-W.</given-names>
</name>
<name>
<surname>Krajmalnik-Brown</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Caporaso</surname>
<given-names>J. G.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Rapidly processed stool swabs approximate stool microbiota profiles</article-title>. <source>Msphere</source> <volume>4</volume>. <pub-id pub-id-type="doi">10.1128/msphere.00208-19</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bolyen</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Rideout</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>Dillon</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Bokulich</surname>
<given-names>N. A.</given-names>
</name>
<name>
<surname>Abnet</surname>
<given-names>C. C.</given-names>
</name>
<name>
<surname>Al-Ghalith</surname>
<given-names>G. A.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Reproducible, interactive, scalable and extensible microbiome data science using QIIME 2</article-title>. <source>Nat. Biotechnol.</source> <volume>37</volume>, <fpage>852</fpage>&#x2013;<lpage>857</lpage>. <pub-id pub-id-type="doi">10.1038/s41587-019-0209-9</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bray</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>Curtis</surname>
<given-names>J. T.</given-names>
</name>
</person-group> (<year>1957</year>). <article-title>An ordination of the upland forest communities of southern Wisconsin</article-title>. <source>Ecol. Monogr.</source> <volume>27</volume>, <fpage>325</fpage>&#x2013;<lpage>349</lpage>. <pub-id pub-id-type="doi">10.2307/1942268</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brooks</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Edwards</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Harwich</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>Rivera</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Fettweis</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Serrano</surname>
<given-names>M. G.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>The truth about metagenomics: quantifying and counteracting bias in 16S rRNA studies</article-title>. <source>BMC Microbiol.</source> <volume>15</volume> (<issue>66</issue>), <fpage>1</fpage>&#x2013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.1186/s12866-015-0351-6</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bukin</surname>
<given-names>Y. S.</given-names>
</name>
<name>
<surname>Galachyants</surname>
<given-names>Y. P.</given-names>
</name>
<name>
<surname>Morozov</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Bukin</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zakharenko</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zemskaya</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>The effect of 16S rRNA region choice on bacterial community metabarcoding results</article-title>. <source>Sci. Data</source> <volume>6</volume>, <fpage>190007</fpage>&#x2013;<lpage>190014</lpage>. <pub-id pub-id-type="doi">10.1038/sdata.2019.7</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Callahan</surname>
<given-names>B. J.</given-names>
</name>
<name>
<surname>Grinevich</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Thakur</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Balamotis</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Yehezkel</surname>
<given-names>T. B.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Ultra-accurate microbial amplicon sequencing with synthetic long reads</article-title>. <source>Microbiome</source> <volume>9</volume> (<issue>130</issue>), <fpage>1</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1186/s40168-021-01072-3</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Caporaso</surname>
<given-names>J. G.</given-names>
</name>
<name>
<surname>Lauber</surname>
<given-names>C. L.</given-names>
</name>
<name>
<surname>Walters</surname>
<given-names>W. A.</given-names>
</name>
<name>
<surname>Berg-Lyons</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Lozupone</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Turnbaugh</surname>
<given-names>P. J.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Global patterns of 16S rRNA diversity at a depth of millions of sequences per sample</article-title>. <source>Proc. Natl. Acad. Sci.</source> <volume>108</volume>, <fpage>4516</fpage>&#x2013;<lpage>4522</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1000080107</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="book">
<collab>CDC</collab> (<year>2015</year>). <source>Guidelines for specimen collection: Instructions for collecting stool specimens</source>. <publisher-loc>Atlanta, GA</publisher-loc>: <publisher-name>Centers for Disease Control and Prevention</publisher-name>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://www.cdc.gov/foodsafety/outbreaks/investigating-outbreaks/specimen-collection.html">https://www.cdc.gov/foodsafety/outbreaks/investigating-outbreaks/specimen-collection.html</ext-link> (Accessed March, 2023)</comment>.</citation>
</ref>
<ref id="B17">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Chai</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>16S-SNAPP-py3</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://github.com/swiftbiosciences/16S-SNAPP-py3">https://github.com/swiftbiosciences/16S-SNAPP-py3</ext-link> (Accessed February, 2022)</comment>.</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chakravorty</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Helb</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Burday</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Connell</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Alland</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>A detailed analysis of 16S ribosomal RNA gene segments for the diagnosis of pathogenic bacteria</article-title>. <source>J. Microbiol. methods</source> <volume>69</volume>, <fpage>330</fpage>&#x2013;<lpage>339</lpage>. <pub-id pub-id-type="doi">10.1016/j.mimet.2007.02.005</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chanderraj</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Brown</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Hinkle</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Falkowski</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Woods</surname>
<given-names>R. J.</given-names>
</name>
<name>
<surname>Dickson</surname>
<given-names>R. P.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>The bacterial density of clinical rectal swabs is highly variable, correlates with sequencing contamination, and predicts patient risk of extraintestinal infection</article-title>. <source>Microbiome</source> <volume>10</volume>, <fpage>2</fpage>. <pub-id pub-id-type="doi">10.1186/s40168-021-01190-y</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>dICC: distance-based intraclass correlation coefficient for metagenomic reproducibility studies</article-title>. <source>Bioinformatics</source> <volume>38</volume>, <fpage>4969</fpage>&#x2013;<lpage>4971</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btac618</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Claassen-Weitz</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gardner-Lubbe</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Mwaikono</surname>
<given-names>K. S.</given-names>
</name>
<name>
<surname>Du Toit</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Zar</surname>
<given-names>H. J.</given-names>
</name>
<name>
<surname>Nicol</surname>
<given-names>M. P.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Optimizing 16S rRNA gene profile analysis from low biomass nasopharyngeal and induced sputum specimens</article-title>. <source>BMC Microbiol.</source> <volume>20</volume> (<issue>113</issue>), <fpage>1</fpage>&#x2013;<lpage>26</lpage>. <pub-id pub-id-type="doi">10.1186/s12866-020-01795-7</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Claassen-Weitz</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gardner-Lubbe</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Nicol</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Botha</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Mounaud</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shankar</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>HIV-exposure, early life feeding practices and delivery mode impacts on faecal bacterial profiles in a South African birth cohort</article-title>. <source>Sci. Rep.</source> <volume>8</volume> (<issue>5078</issue>), <fpage>1</fpage>&#x2013;<lpage>15</lpage>. <pub-id pub-id-type="doi">10.1038/s41598-018-22244-6</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Davis</surname>
<given-names>N. M.</given-names>
</name>
<name>
<surname>Proctor</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Holmes</surname>
<given-names>S. P.</given-names>
</name>
<name>
<surname>Relman</surname>
<given-names>D. A.</given-names>
</name>
<name>
<surname>Callahan</surname>
<given-names>B. J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Simple statistical identification and removal of contaminant sequences in marker-gene and metagenomics data</article-title>. <source>Microbiome</source> <volume>6</volume> (<issue>226</issue>), <fpage>1</fpage>&#x2013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.1186/s40168-018-0605-2</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Drengenes</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Eagan</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Haaland</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Wiker</surname>
<given-names>H. G.</given-names>
</name>
<name>
<surname>Nielsen</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Exploring protocol bias in airway microbiome studies: one versus two PCR steps and 16S rRNA gene region V3 V4 versus V4</article-title>. <source>BMC genomics</source> <volume>22</volume> (<issue>3</issue>), <fpage>1</fpage>&#x2013;<lpage>15</lpage>. <pub-id pub-id-type="doi">10.1186/s12864-020-07252-z</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fisher</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Corbet</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>Williams</surname>
<given-names>C. B.</given-names>
</name>
</person-group> (<year>1943</year>). <article-title>The relation between the number of species and the number of individuals in a random sample of an animal population</article-title>. <source>J. Animal Ecol.</source> <volume>12</volume>, <fpage>42</fpage>&#x2013;<lpage>58</lpage>. <pub-id pub-id-type="doi">10.2307/1411</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Flygel</surname>
<given-names>T. T.</given-names>
</name>
<name>
<surname>Sovershaeva</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Claassen-Weitz</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hjerde</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Mwaikono</surname>
<given-names>K. S.</given-names>
</name>
<name>
<surname>Odland</surname>
<given-names>J. &#xd8;.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Composition of gut microbiota of children and adolescents with perinatal human immunodeficiency virus infection taking antiretroviral therapy in Zimbabwe</article-title>. <source>J. Infect. Dis.</source> <volume>221</volume>, <fpage>483</fpage>&#x2013;<lpage>492</lpage>. <pub-id pub-id-type="doi">10.1093/infdis/jiz473</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fouhy</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Clooney</surname>
<given-names>A. G.</given-names>
</name>
<name>
<surname>Stanton</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Claesson</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Cotter</surname>
<given-names>P. D.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>16S rRNA gene sequencing of mock microbial populations-impact of DNA extraction method, primer choice and sequencing platform</article-title>. <source>BMC Microbiol.</source> <volume>16</volume> (<issue>123</issue>), <fpage>1</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1186/s12866-016-0738-z</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Freedman</surname>
<given-names>S. B.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Nettel-Aguirre</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Chui</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Pang</surname>
<given-names>X.-L.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Enteropathogen detection in children with diarrhoea, or vomiting, or both, comparing rectal flocked swabs with stool specimens: an outpatient cohort study</article-title>. <source>lancet Gastroenterology and hepatology</source> <volume>2</volume>, <fpage>662</fpage>&#x2013;<lpage>669</lpage>. <pub-id pub-id-type="doi">10.1016/s2468-1253(17)30160-7</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fuks</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Elgart</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Amir</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zeisel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Turnbaugh</surname>
<given-names>P. J.</given-names>
</name>
<name>
<surname>Soen</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Combining 16S rRNA gene variable regions enables high-resolution microbial community profiling</article-title>. <source>Microbiome</source> <volume>6</volume> (<issue>17</issue>), <fpage>1</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1186/s40168-017-0396-x</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Kuang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wan</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>A study of the correlation between obesity and intestinal flora in school-age children</article-title>. <source>Sci. Rep.</source> <volume>8</volume>, <fpage>14511</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-018-32730-6</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goldfarb</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Steenhoff</surname>
<given-names>A. P.</given-names>
</name>
<name>
<surname>Pernica</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Chong</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Luinstra</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Mokomane</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Evaluation of anatomically designed flocked rectal swabs for molecular detection of enteric pathogens in children admitted to hospital with severe gastroenteritis in Botswana</article-title>. <source>J. Clin. Microbiol.</source> <volume>52</volume>, <fpage>3922</fpage>&#x2013;<lpage>3927</lpage>. <pub-id pub-id-type="doi">10.1128/jcm.01894-14</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gorzelak</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Gill</surname>
<given-names>S. K.</given-names>
</name>
<name>
<surname>Tasnim</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Ahmadi-Vand</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Jay</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Gibson</surname>
<given-names>D. L.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Methods for improving human gut microbiome data by reducing variability through sample processing and storage of stool</article-title>. <source>PloS one</source> <volume>10</volume>, <fpage>e0134802</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0134802</pub-id>
</citation>
</ref>
<ref id="B90">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Graham</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>Patel</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Little</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>van der Kouwe</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kaba</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Holmes</surname>
<given-names>M. J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Using short-read 16S rRNA sequencing of multiple variable regions to generate high-quality results to a species level. bioRxiv</article-title>. </citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Graspeuntner</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lupatsii</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Dashdorj</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Rody</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Rupp</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Bossung</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>First-Day-of-Life rectal swabs fail to represent meconial microbiota composition and underestimate the presence of antibiotic resistance genes</article-title>. <source>Microbiol. Spectr.</source> <volume>11</volume>. <pub-id pub-id-type="doi">10.1128/spectrum.05254-22</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gratton</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Phetcharaburanin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Mullish</surname>
<given-names>B. H.</given-names>
</name>
<name>
<surname>Williams</surname>
<given-names>H. R.</given-names>
</name>
<name>
<surname>Thursz</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Nicholson</surname>
<given-names>J. K.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Optimized sample handling strategy for metabolic profiling of human feces</article-title>. <source>Anal. Chem.</source> <volume>88</volume>, <fpage>4661</fpage>&#x2013;<lpage>4668</lpage>. <pub-id pub-id-type="doi">10.1021/acs.analchem.5b04159</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Ju</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Taxonomic precision of different hypervariable regions of 16S rRNA gene and annotation methods for functional bacterial groups in biological wastewater treatment</article-title>. <source>PloS one</source> <volume>8</volume>, <fpage>e76185</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0076185</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hosgood III</surname>
<given-names>H. D.</given-names>
</name>
<name>
<surname>Sapkota</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Rothman</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Rohan</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>The potential role of lung microbiota in lung cancer attributed to household coal burning exposures</article-title>. <source>Environ. Mol. Mutagen.</source> <volume>55</volume>, <fpage>643</fpage>&#x2013;<lpage>651</lpage>. <pub-id pub-id-type="doi">10.1002/em.21878</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huson</surname>
<given-names>D. H.</given-names>
</name>
<name>
<surname>Steel</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>El-Hadidi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mitra</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Peter</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Willmann</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>A simple statistical test of taxonomic or functional homogeneity using replicated microbiome sequencing samples</article-title>. <source>J. Biotechnol.</source> <volume>250</volume>, <fpage>45</fpage>&#x2013;<lpage>50</lpage>. <pub-id pub-id-type="doi">10.1016/j.jbiotec.2016.10.020</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chitnis</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Monos</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Dinh</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Next-generation sequencing technologies: an overview</article-title>. <source>Hum. Immunol.</source> <volume>82</volume>, <fpage>801</fpage>&#x2013;<lpage>811</lpage>. <pub-id pub-id-type="doi">10.1016/j.humimm.2021.02.012</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<collab>ILLUMINA. 16S Metagenomic Sequencing Library Preparation</collab> (<year>2024</year>). <source>Prep. 16S ribosomal RNA gene amplicons Illumina MiSeq Syst</source>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://support.illumina.com/documents/documentation/chemistry_documentation/16s/16s-metagenomic-library-prep-guide-15044223-b.pdf">https://support.illumina.com/documents/documentation/chemistry_documentation/16s/16s-metagenomic-library-prep-guide-15044223-b.pdf</ext-link> (Accessed December, 2024)</comment>.</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ji</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Nielsen</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>From next-generation sequencing to systematic modeling of the gut microbiome</article-title>. <source>Front. Genet.</source> <volume>6</volume>, <fpage>219</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2015.00219</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Johnson</surname>
<given-names>J. S.</given-names>
</name>
<name>
<surname>Spakowicz</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Hong</surname>
<given-names>B.-Y.</given-names>
</name>
<name>
<surname>Petersen</surname>
<given-names>L. M.</given-names>
</name>
<name>
<surname>Demkowicz</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Evaluation of 16S rRNA gene sequencing for species and strain-level microbiome analysis</article-title>. <source>Nat. Commun.</source> <volume>10</volume> (<issue>5029</issue>), <fpage>1</fpage>&#x2013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1038/s41467-019-13036-1</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jones</surname>
<given-names>R. B.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Moan</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Murff</surname>
<given-names>H. J.</given-names>
</name>
<name>
<surname>Ness</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Seidner</surname>
<given-names>D. L.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Inter-niche and inter-individual variation in gut microbial community assessment using stool, rectal swab, and mucosal samples</article-title>. <source>Sci. Rep.</source> <volume>8</volume>, <fpage>4139</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-018-22408-4</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kers</surname>
<given-names>J. G.</given-names>
</name>
<name>
<surname>Saccenti</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>The power of microbiome studies: some considerations on which alpha and beta metrics to use and how to report analysis the results</article-title>. <source>Front. Microbiol.</source> <volume>12</volume>:<fpage>796025</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2021.796025</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Klindworth</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Pruesse</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Schweer</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Peplies</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Quast</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Horn</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Evaluation of general 16S ribosomal RNA gene PCR primers for classical and next-generation sequencing-based diversity studies</article-title>. <source>Nucleic acids Res.</source> <volume>41</volume>, <fpage>e1</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gks808</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Koo</surname>
<given-names>T. K.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>M. Y.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>A guideline of selecting and reporting intraclass correlation coefficients for reliability research</article-title>. <source>J. Chiropr. Med.</source> <volume>15</volume>, <fpage>155</fpage>&#x2013;<lpage>163</lpage>. <pub-id pub-id-type="doi">10.1016/j.jcm.2016.02.012</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lane</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Pace</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Olsen</surname>
<given-names>G. J.</given-names>
</name>
<name>
<surname>Stahl</surname>
<given-names>D. A.</given-names>
</name>
<name>
<surname>Sogin</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>Pace</surname>
<given-names>N. R.</given-names>
</name>
</person-group> (<year>1985</year>). <article-title>Rapid determination of 16S ribosomal RNA sequences for phylogenetic analyses</article-title>. <source>Proc. Natl. Acad. Sci.</source> <volume>82</volume>, <fpage>6955</fpage>&#x2013;<lpage>6959</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.82.20.6955</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Laursen</surname>
<given-names>M. F.</given-names>
</name>
<name>
<surname>Dalgaard</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>Bahl</surname>
<given-names>M. I.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Genomic GC-content affects the accuracy of 16S rRNA gene sequencing based microbial profiling due to PCR bias</article-title>. <source>Front. Microbiol.</source> <volume>8</volume>, <fpage>1934</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2017.01934</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname>
<given-names>C. K.</given-names>
</name>
<name>
<surname>Herbold</surname>
<given-names>C. W.</given-names>
</name>
<name>
<surname>Polson</surname>
<given-names>S. W.</given-names>
</name>
<name>
<surname>Wommack</surname>
<given-names>K. E.</given-names>
</name>
<name>
<surname>Williamson</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Mcdonald</surname>
<given-names>I. R.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Groundtruthing next-gen sequencing for microbial ecology&#x2013;biases and errors in community structure estimates from PCR amplicon pyrosequencing</article-title>. <source>PLoS One</source> <volume>7</volume>, <fpage>e44224</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0044224</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Systematic analysis of impact of sampling regions and storage methods on fecal gut microbiome and metabolome profiles</article-title>. <source>Msphere</source> <volume>5</volume>. <pub-id pub-id-type="doi">10.1128/msphere.00763-19</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ruan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Qian</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>
<italic>De novo</italic> assembly of human genomes with massively parallel short read sequencing</article-title>. <source>Genome Res.</source> <volume>20</volume>, <fpage>265</fpage>&#x2013;<lpage>272</lpage>. <pub-id pub-id-type="doi">10.1101/gr.097261.109</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ludwig</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Reynolds</surname>
<given-names>J. F.</given-names>
</name>
</person-group> (<year>1988</year>). <source>Statistical ecology: a primer in methods and computing</source>. <publisher-name>John Wiley and Sons</publisher-name>.</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Maki</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Wolff</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Varuzza</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Green</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Barb</surname>
<given-names>J. J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Multi-amplicon microbiome data analysis pipelines for mixed orientation sequences using QIIME2: assessing reference database, variable region and pre-processing bias in classification of mock bacterial community samples</article-title>. <source>Plos one</source> <volume>18</volume>, <fpage>e0280293</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0280293</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>_MMUPHin: meta-analysis methods with uniform pipeline for heterogeneity in microbiome studies_</article-title>. <source>Genome Biol.</source> <volume>23</volume> (<issue>208</issue>), <fpage>1</fpage>&#x2013;<lpage>31</lpage>. <pub-id pub-id-type="doi">10.18129/B9.bioc.MMUPHin</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shungin</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Mallick</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Schirmer</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>L. H.</given-names>
</name>
<name>
<surname>Kolde</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Population structure discovery in meta-analyzed microbial communities and inflammatory bowel disease using MMUPHin</article-title>. <source>Genome Biol.</source> <volume>23</volume>, <fpage>208</fpage>&#x2013;<lpage>231</lpage>. <pub-id pub-id-type="doi">10.1186/s13059-022-02753-4</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mcmurdie</surname>
<given-names>P. J.</given-names>
</name>
<name>
<surname>Holmes</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>phyloseq: an R package for reproducible interactive analysis and graphics of microbiome census data</article-title>. <source>PloS one</source> <volume>8</volume>, <fpage>e61217</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0061217</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mcmurdie</surname>
<given-names>P. J.</given-names>
</name>
<name>
<surname>Holmes</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Shiny-phyloseq: web application for interactive microbiome analysis with provenance tracking</article-title>. <source>Bioinformatics</source> <volume>31</volume>, <fpage>282</fpage>&#x2013;<lpage>283</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btu616</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Midha</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chiu</surname>
<given-names>K.-P.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Long-read sequencing in deciphering human genetics to a greater depth</article-title>. <source>Hum. Genet.</source> <volume>138</volume>, <fpage>1201</fpage>&#x2013;<lpage>1215</lpage>. <pub-id pub-id-type="doi">10.1007/s00439-019-02064-y</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nearing</surname>
<given-names>J. T.</given-names>
</name>
<name>
<surname>Comeau</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Langille</surname>
<given-names>M. G.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Identifying biases and their potential solutions in human microbiome studies</article-title>. <source>Microbiome</source> <volume>9</volume> (<issue>113</issue>), <fpage>1</fpage>&#x2013;<lpage>22</lpage>. <pub-id pub-id-type="doi">10.1186/s40168-021-01059-0</pub-id>
</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nuccio</surname>
<given-names>D. A.</given-names>
</name>
<name>
<surname>Normann</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Grippo</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Microbiome and metabolome variation as indicator of social stress in female prairie voles</article-title>. <source>Int. J. Mol. Sci.</source> <volume>24</volume>, <fpage>1677</fpage>. <pub-id pub-id-type="doi">10.3390/ijms24021677</pub-id>
</citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>&#xd6;zkurt</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Fritscher</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Soranzo</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Ng</surname>
<given-names>D. Y.</given-names>
</name>
<name>
<surname>Davey</surname>
<given-names>R. P.</given-names>
</name>
<name>
<surname>Bahram</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>LotuS2: an ultrafast and highly accurate tool for amplicon sequencing analysis</article-title>. <source>Microbiome</source> <volume>10</volume> (<issue>176</issue>), <fpage>1</fpage>&#x2013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.1186/s40168-022-01365-1</pub-id>
</citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Patel</surname>
<given-names>J. B.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>16S rRNA gene sequencing for bacterial pathogen identification in the clinical laboratory</article-title>. <source>Mol. Diagn.</source> <volume>6</volume>, <fpage>313</fpage>&#x2013;<lpage>321</lpage>. <pub-id pub-id-type="doi">10.2165/00066982-200106040-00012</pub-id>
</citation>
</ref>
<ref id="B62">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pereira-Marques</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hout</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ferreira</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Weber</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pinto-Ribeiro</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Van Doorn</surname>
<given-names>L.-J.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Impact of host DNA and sequencing depth on the taxonomic resolution of whole metagenome sequencing for microbiome analysis</article-title>. <source>Front. Microbiol.</source> <volume>10</volume>, <fpage>1277</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2019.01277</pub-id>
</citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Quail</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Coupland</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Otto</surname>
<given-names>T. D.</given-names>
</name>
<name>
<surname>Harris</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Connor</surname>
<given-names>T. R.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>A tale of three next generation sequencing platforms: comparison of Ion Torrent, Pacific Biosciences and Illumina MiSeq sequencers</article-title>. <source>BMC genomics</source> <volume>13</volume> (<issue>341</issue>), <fpage>1</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1186/1471-2164-13-341</pub-id>
</citation>
</ref>
<ref id="B64">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Radhakrishnan</surname>
<given-names>S. T.</given-names>
</name>
<name>
<surname>Gallagher</surname>
<given-names>K. I.</given-names>
</name>
<name>
<surname>Mullish</surname>
<given-names>B. H.</given-names>
</name>
<name>
<surname>Serrano-Contreras</surname>
<given-names>J. I.</given-names>
</name>
<name>
<surname>Alexander</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Miguens Blanco</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Rectal swabs as a viable alternative to faecal sampling for the analysis of gut microbiota functionality and composition</article-title>. <source>Sci. Rep.</source> <volume>13</volume>, <fpage>493</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-022-27131-9</pub-id>
</citation>
</ref>
<ref id="B65">
<citation citation-type="book">
<collab>R CORE TEAM</collab> (<year>2022</year>). <source>R: a language and environment for statistical computing</source>. <publisher-loc>Vienna, Austria</publisher-loc>: <publisher-name>R Foundation for Statistical Computing</publisher-name>.</citation>
</ref>
<ref id="B66">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reyman</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Van Houten</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Arp</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Sanders</surname>
<given-names>E. A.</given-names>
</name>
<name>
<surname>Bogaert</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Rectal swabs are a reliable proxy for faecal samples in infant gut microbiota research based on 16S-rRNA sequencing</article-title>. <source>Sci. Rep.</source> <volume>9</volume>, <fpage>16072</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-019-52549-z</pub-id>
</citation>
</ref>
<ref id="B67">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rogers</surname>
<given-names>G. B.</given-names>
</name>
<name>
<surname>Bruce</surname>
<given-names>K. D.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Next-generation sequencing in the analysis of human microbiota: essential considerations for clinical application</article-title>. <source>Mol. diagnosis and Ther.</source> <volume>14</volume>, <fpage>343</fpage>&#x2013;<lpage>350</lpage>. <pub-id pub-id-type="doi">10.1007/bf03256391</pub-id>
</citation>
</ref>
<ref id="B68">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schriefer</surname>
<given-names>A. E.</given-names>
</name>
<name>
<surname>Cliften</surname>
<given-names>P. F.</given-names>
</name>
<name>
<surname>Hibberd</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Sawyer</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Brown-Kennerly</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Burcea</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>A multi-amplicon 16S rRNA sequencing and analysis method for improved taxonomic profiling of bacterial communities</article-title>. <source>J. Microbiol. methods</source> <volume>154</volume>, <fpage>6</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1016/j.mimet.2018.09.019</pub-id>
</citation>
</ref>
<ref id="B69">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schroeder</surname>
<given-names>P. J.</given-names>
</name>
<name>
<surname>Jenkins</surname>
<given-names>D. G.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>How robust are popular beta diversity indices to sampling error?</article-title> <source>Ecosphere</source> <volume>9</volume>, <fpage>e02100</fpage>. <pub-id pub-id-type="doi">10.1002/ecs2.2100</pub-id>
</citation>
</ref>
<ref id="B70">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sedlazeck</surname>
<given-names>F. J.</given-names>
</name>
<name>
<surname>Rescheneder</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Smolka</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Nattestad</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>VON Haeseler</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Accurate detection of complex structural variations using single-molecule sequencing</article-title>. <source>Nat. methods</source> <volume>15</volume>, <fpage>461</fpage>&#x2013;<lpage>468</lpage>. <pub-id pub-id-type="doi">10.1038/s41592-018-0001-7</pub-id>
</citation>
</ref>
<ref id="B71">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shannon</surname>
<given-names>C. E.</given-names>
</name>
</person-group> (<year>1948</year>). <article-title>A mathematical theory of communication</article-title>. <source>Bell Syst. Tech. J.</source> <volume>27</volume>, <fpage>379</fpage>&#x2013;<lpage>423</lpage>. <pub-id pub-id-type="doi">10.1002/j.1538-7305.1948.tb01338.x</pub-id>
</citation>
</ref>
<ref id="B72">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Short</surname>
<given-names>M. I.</given-names>
</name>
<name>
<surname>Hudson</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Besasie</surname>
<given-names>B. D.</given-names>
</name>
<name>
<surname>Reveles</surname>
<given-names>K. R.</given-names>
</name>
<name>
<surname>Shah</surname>
<given-names>D. P.</given-names>
</name>
<name>
<surname>Nicholson</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Comparison of rectal swab, glove tip, and participant-collected stool techniques for gut microbiome sampling</article-title>. <source>BMC Microbiol.</source> <volume>21</volume> (<issue>26</issue>), <fpage>1</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1186/s12866-020-02080-3</pub-id>
</citation>
</ref>
<ref id="B73">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Simpson</surname>
<given-names>E. H.</given-names>
</name>
</person-group> (<year>1949</year>). <article-title>Measurement of diversity</article-title>. <source>nature</source> <volume>163</volume>, <fpage>688</fpage>. <pub-id pub-id-type="doi">10.1038/163688a0</pub-id>
</citation>
</ref>
<ref id="B74">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Murff</surname>
<given-names>H. J.</given-names>
</name>
<name>
<surname>Ness</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Seidner</surname>
<given-names>D. L.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>On the robustness of inference of association with the gut microbiota in stool, rectal swab and mucosal tissue samples</article-title>. <source>Sci. Rep.</source> <volume>11</volume>, <fpage>14828</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-021-94205-5</pub-id>
</citation>
</ref>
<ref id="B75">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Szoboszlay</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Schramm</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Pinzauti</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Scerri</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sandionigi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Biazzo</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Nanopore is preferable over Illumina for 16S amplicon sequencing of the gut microbiota when species-level taxonomic classification, accurate estimation of richness, or focus on rare taxa is required</article-title>. <source>Microorganisms</source> <volume>11</volume>, <fpage>804</fpage>. <pub-id pub-id-type="doi">10.3390/microorganisms11030804</pub-id>
</citation>
</ref>
<ref id="B76">
<citation citation-type="book">
<collab>THE JACKSON LABORATORY</collab>. (<year>2019</year>). <source>Analysing 16S data: Part 2</source> <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://thejacksonlaboratory.github.io/microbiome-workshop-2019/jekyll/update/2019/10/30/Analysing-16S-data-part-2.html">https://thejacksonlaboratory.github.io/microbiome-workshop-2019/jekyll/update/2019/10/30/Analysing-16S-data-part-2.html</ext-link>.</comment>
</citation>
</ref>
<ref id="B77">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tucker</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Marra</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Friedman</surname>
<given-names>J. M.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Massively parallel sequencing: the next big thing in genetic medicine</article-title>. <source>Am. J. Hum. Genet.</source> <volume>85</volume>, <fpage>142</fpage>&#x2013;<lpage>154</lpage>. <pub-id pub-id-type="doi">10.1016/j.ajhg.2009.06.022</pub-id>
</citation>
</ref>
<ref id="B78">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Urban</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Holzer</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Baronas</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Hall</surname>
<given-names>M. B.</given-names>
</name>
<name>
<surname>Braeuninger-Weimer</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Scherm</surname>
<given-names>M. J.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Freshwater monitoring by nanopore sequencing</article-title>. <source>Elife</source> <volume>10</volume>, <fpage>e61504</fpage>. <pub-id pub-id-type="doi">10.7554/elife.61504</pub-id>
</citation>
</ref>
<ref id="B79">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Van Dijk</surname>
<given-names>E. L.</given-names>
</name>
<name>
<surname>Jaszczyszyn</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Naquin</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Thermes</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>The third revolution in sequencing technology</article-title>. <source>Trends Genet.</source> <volume>34</volume>, <fpage>666</fpage>&#x2013;<lpage>681</lpage>. <pub-id pub-id-type="doi">10.1016/j.tig.2018.05.008</pub-id>
</citation>
</ref>
<ref id="B80">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Videnska</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Smerkova</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zwinsova</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Popovici</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Micenkova</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Sedlar</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Stool sampling and DNA isolation kits affect DNA quality and bacterial composition following 16S rRNA gene sequencing using MiSeq Illumina platform</article-title>. <source>Sci. Rep.</source> <volume>9</volume>, <fpage>13837</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-019-49520-3</pub-id>
</citation>
</ref>
<ref id="B81">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Tu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Improving the microbial community reconstruction at the genus level by multiple 16S rRNA regions</article-title>. <source>J. Theor. Biol.</source> <volume>398</volume>, <fpage>1</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1016/j.jtbi.2016.03.016</pub-id>
</citation>
</ref>
<ref id="B82">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Qian</surname>
<given-names>P.-Y.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Conservative fragments in bacterial 16S rRNA genes and primer design for 16S ribosomal DNA amplicons in metagenomic studies</article-title>. <source>PloS one</source> <volume>4</volume>, <fpage>e7401</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0007401</pub-id>
</citation>
</ref>
<ref id="B83">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yuan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Cohen</surname>
<given-names>D. B.</given-names>
</name>
<name>
<surname>Ravel</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Abdo</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Forney</surname>
<given-names>L. J.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Evaluation of methods for the extraction and purification of DNA from the human microbiome</article-title>. <source>PloS one</source> <volume>7</volume>, <fpage>e33865</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0033865</pub-id>
</citation>
</ref>
<ref id="B84">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Phillips</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gail</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Goedert</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Humphrys</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Ravel</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>The effect of cigarette smoking on the oral and nasal microbiota</article-title>. <source>Microbiome</source> <volume>5</volume> (<issue>3</issue>), <fpage>1</fpage>&#x2013;<lpage>16</lpage>. <pub-id pub-id-type="doi">10.1186/s40168-016-0226-6</pub-id>
</citation>
</ref>
<ref id="B85">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zerbino</surname>
<given-names>D. R.</given-names>
</name>
<name>
<surname>Birney</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Velvet: algorithms for <italic>de novo</italic> short read assembly using de Bruijn graphs</article-title>. <source>Genome Res.</source> <volume>18</volume>, <fpage>821</fpage>&#x2013;<lpage>829</lpage>. <pub-id pub-id-type="doi">10.1101/gr.074492.107</pub-id>
</citation>
</ref>
<ref id="B86">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>H.-W.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>D.-F.</given-names>
</name>
<name>
<surname>Tam</surname>
<given-names>N. F.-Y.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>X.-T.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Sheng</surname>
<given-names>H.-F.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>BIPES, a cost-effective high-throughput method for assessing microbial diversity</article-title>. <source>ISME J.</source> <volume>5</volume>, <fpage>741</fpage>&#x2013;<lpage>749</lpage>. <pub-id pub-id-type="doi">10.1038/ismej.2010.160</pub-id>
</citation>
</ref>
<ref id="B87">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zreloff</surname>
<given-names>Z. J.</given-names>
</name>
<name>
<surname>Lange</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Vernon</surname>
<given-names>S. D.</given-names>
</name>
<name>
<surname>Carlin</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Cano</surname>
<given-names>R. J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Accelerating gut microbiome research with robust sample collection</article-title>. <source>Res. and Rev. J. Microbiol. Biotechnol.</source> <volume>12</volume>, <fpage>33</fpage>&#x2013;<lpage>47</lpage>.</citation>
</ref>
<ref id="B88">
<citation citation-type="web">
<collab>ZYMOBIOMICS</collab> (<year>2022</year>). <article-title>ZymoBIOMICS&#x2122; microbial community DNA standard; catalog nos. D6305 (200ng) and D6306 (2000ng)</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://files.zymoresearch.com/protocols/_d6305_d6306_zymobiomics_microbial_community_dna_standard.pdf">https://files.zymoresearch.com/protocols/_d6305_d6306_zymobiomics_microbial_community_dna_standard.pdf</ext-link> (Accessed October, 2022)</comment>.</citation>
</ref>
<ref id="B89">
<citation citation-type="web">
<collab>ZYMOBIOMICS</collab> (<year>2024</year>). <article-title>ZymoBIOMICS&#x2122; DNA miniprep kit</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://files.zymoresearch.com/protocols/_d4300t_d4300_d4304_zymobiomics_dna_miniprep_kit.pdf">https://files.zymoresearch.com/protocols/_d4300t_d4300_d4304_zymobiomics_dna_miniprep_kit.pdf</ext-link> (Accessed December, 2024)</comment>.</citation>
</ref>
</ref-list>
</back>
</article>