<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Microbiol.</journal-id>
<journal-title>Frontiers in Microbiology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Microbiol.</abbrev-journal-title>
<issn pub-type="epub">1664-302X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmicb.2018.00063</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Microbiology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Comparative Genomics of Completely Sequenced <italic>Lactobacillus helveticus</italic> Genomes Provides Insights into Strain-Specific Genes and Resolves Metagenomics Data Down to the Strain Level</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Schmid</surname> <given-names>Michael</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/478487/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Muri</surname> <given-names>Jonathan</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Melidis</surname> <given-names>Damianos</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Varadarajan</surname> <given-names>Adithi R.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/278361/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Somerville</surname> <given-names>Vincent</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/486371/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Wicki</surname> <given-names>Adrian</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Moser</surname> <given-names>Aline</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/426876/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Bourqui</surname> <given-names>Marc</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Wenzel</surname> <given-names>Claudia</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Eugster-Meier</surname> <given-names>Elisabeth</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/453444/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Frey</surname> <given-names>Juerg E.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Irmler</surname> <given-names>Stefan</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/426844/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Ahrens</surname> <given-names>Christian H.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/269195/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Agroscope, Research Group Molecular Diagnostics, Genomics and Bioinformatics</institution>, <addr-line>W&#x000E4;denswil</addr-line>, <country>Switzerland</country></aff>
<aff id="aff2"><sup>2</sup><institution>Swiss Institute of Bioinformatics</institution>, <addr-line>W&#x000E4;denswil</addr-line>, <country>Switzerland</country></aff>
<aff id="aff3"><sup>3</sup><institution>Agroscope, Research Group Biochemistry of Milk and Microorganisms</institution>, <addr-line>Bern</addr-line>, <country>Switzerland</country></aff>
<aff id="aff4"><sup>4</sup><institution>School of Agricultural, Forest and Food Sciences HAFL, Bern University of Applied Sciences</institution>, <addr-line>Zollikofen</addr-line>, <country>Switzerland</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: David Rodriguez-Lazaro, University of Burgos, Spain</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Baltasar Mayo, Consejo Superior de Investigaciones Cient&#x000ED;ficas (CSIC), Spain; Giorgio Giraffa, Centro di Zootecnia e Acquacoltura (CREA-ZA), Italy</p></fn>
<fn fn-type="corresp" id="fn001"><p>&#x0002A;Correspondence: Christian H. Ahrens <email>christian.ahrens&#x00040;agroscope.admin.ch</email></p></fn>
<fn fn-type="other" id="fn002"><p>This article was submitted to Food Microbiology, a section of the journal Frontiers in Microbiology</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>30</day>
<month>01</month>
<year>2018</year>
</pub-date>
<pub-date pub-type="collection">
<year>2018</year>
</pub-date>
<volume>9</volume>
<elocation-id>63</elocation-id>
<history>
<date date-type="received">
<day>31</day>
<month>10</month>
<year>2017</year>
</date>
<date date-type="accepted">
<day>10</day>
<month>01</month>
<year>2018</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2018 Schmid, Muri, Melidis, Varadarajan, Somerville, Wicki, Moser, Bourqui, Wenzel, Eugster-Meier, Frey, Irmler and Ahrens.</copyright-statement>
<copyright-year>2018</copyright-year>
<copyright-holder>Schmid, Muri, Melidis, Varadarajan, Somerville, Wicki, Moser, Bourqui, Wenzel, Eugster-Meier, Frey, Irmler and Ahrens</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Although complete genome sequences hold particular value for an accurate description of core genomes, the identification of strain-specific genes, and as the optimal basis for functional genomics studies, they are still largely underrepresented in public repositories. Based on an assessment of the genome assembly complexity for all lactobacilli, we used Pacific Biosciences&#x00027; long read technology to sequence and <italic>de novo</italic> assemble the genomes of three <italic>Lactobacillus helveticus</italic> starter strains, raising the number of completely sequenced strains to 12. The first comparative genomics study for <italic>L. helveticus</italic>&#x02014;to our knowledge&#x02014;identified a core genome of 988 genes and sets of unique, strain-specific genes ranging from about 30 to more than 200 genes. Importantly, the comparison of MiSeq- and PacBio-based assemblies uncovered that not only accessory but also core genes can be missed in incomplete genome assemblies based on short reads. Analysis of the three genomes revealed that a large number of pseudogenes were enriched for functional Gene Ontology categories such as amino acid transmembrane transport and carbohydrate metabolism, which is in line with a reductive genome evolution in the rich natural habitat of <italic>L. helveticus</italic>. Notably, the functional Clusters of Orthologous Groups of proteins categories &#x0201C;cell wall/membrane biogenesis&#x0201D; and &#x0201C;defense mechanisms&#x0201D; were found to be enriched among the strain-specific genes. A genome mining effort uncovered examples where an experimentally observed phenotype could be linked to the underlying genotype, such as for cell envelope proteinase PrtH3 of strain FAM8627. Another possible link identified for peptidoglycan hydrolases will require further experiments. Of note, strain FAM22155 did not harbor a CRISPR/Cas system; its loss was also observed in other <italic>L. helveticus</italic> strains and lactobacillus species, thus questioning the value of the CRISPR/Cas system for diagnostic purposes. Importantly, the complete genome sequences proved to be very useful for the analysis of natural whey starter cultures with metagenomics, as a larger percentage of the sequenced reads of these complex mixtures could be unambiguously assigned down to the strain level.</p>
</abstract>
<kwd-group>
<kwd>whole genome sequencing</kwd>
<kwd>PacBio</kwd>
<kwd>comparative genomics</kwd>
<kwd>strain-specific genes</kwd>
<kwd>CRISPR/Cas</kwd>
<kwd>metagenomics</kwd>
<kwd>natural whey starter cultures</kwd>
<kwd>dairy industry</kwd>
</kwd-group>
<contract-num rid="cn001">31003A-156320</contract-num>
<contract-sponsor id="cn001">Schweizerischer Nationalfonds zur F&#x000F6;rderung der Wissenschaftlichen Forschung<named-content content-type="fundref-id">10.13039/501100001711</named-content></contract-sponsor>
<counts>
<fig-count count="7"/>
<table-count count="5"/>
<equation-count count="0"/>
<ref-count count="82"/>
<page-count count="20"/>
<word-count count="13726"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>Introduction</title>
<p>Lactic acid bacteria (LAB) degrade sugar to lactic acid and are often used in food fermentation (Leroy and De Vuyst, <xref ref-type="bibr" rid="B49">2004</xref>; Giraffa et al., <xref ref-type="bibr" rid="B34">2010</xref>). <italic>Lactobacillus</italic> is one of several genera that belong to LAB (Sun et al., <xref ref-type="bibr" rid="B78">2015</xref>) and, due to their role in fermented food production or their use as probiotics, they are among the most important bacteria in food microbiology (Salvetti et al., <xref ref-type="bibr" rid="B68">2012</xref>). In general, lactobacilli are microaerophilic, Gram-positive bacteria that form rods or cocci (Makarova et al., <xref ref-type="bibr" rid="B52">2006</xref>). In milk, lactobacilli degrade lactose, citric acid, milk proteins, and lipids (McSweeney, <xref ref-type="bibr" rid="B55">2011</xref>). The breakdown of milk proteins is considered to contribute the most to the development of flavor. In addition, various metabolites formed by these biochemical activities are precursors for aroma-active compounds.</p>
<p><italic>Lactobacillus helveticus</italic> strains are abundant in the natural whey starter cultures (NWCs) that are used for the production of Gruy&#x000E8;re, a protected designation of origin (PDO) cheese (Moser et al., <xref ref-type="bibr" rid="B57">2017</xref>) (<ext-link ext-link-type="uri" xlink:href="http://gruyere.com/en/specifications/?vt=ch">http://gruyere.com/en/specifications/?vt=ch</ext-link>). <italic>L. helveticus</italic> exhibits diverse proteolytic and peptidolytic activities. It is widely used with other thermophilic LAB, including <italic>Streptococcus thermophilus</italic> and <italic>Lactobacillus delbrueckii</italic> subsp<italic>. lactis</italic>, in the manufacture of Swiss cheese and Italian hard cheeses (Slattery et al., <xref ref-type="bibr" rid="B72">2010</xref>; Eugster-Meier et al., <xref ref-type="bibr" rid="B29">2017</xref>) such as Grana Padano, Parmigiano Reggiano, and Provolone, and as flavor-enhancing adjunct culture in Cheddar cheesemaking (Hannon et al., <xref ref-type="bibr" rid="B36">2007</xref>; Slattery et al., <xref ref-type="bibr" rid="B72">2010</xref>). Several <italic>L. helveticus</italic> strains may be exploited as probiotics that provide health-promoting properties (Taverniti and Guglielmetti, <xref ref-type="bibr" rid="B80">2012</xref>).</p>
<p>The computational mining of genome sequences, for example for genes encoding specific metabolic activities, or involved in toxin formation, may facilitate the selection of strains for specific biotechnological applications. Next-generation sequencing (NGS) technologies such as the cost-efficient Illumina short read sequencing technology have been widely used to sequence bacterial genomes (Mavromatis et al., <xref ref-type="bibr" rid="B53">2012</xref>). However, the presence of repeated sequences such as insertion sequence (IS) elements and rDNA operons severely compromises the ability to completely assemble complex genomes (Koren et al., <xref ref-type="bibr" rid="B44">2013</xref>). Accordingly, although the number of publicly available bacterial genome assemblies has been increasing exponentially (Reddy et al., <xref ref-type="bibr" rid="B63">2014</xref>), the large majority has been reported as draft genomes with a large number of contigs, which represents a serious limitation for follow-up analyses (Ricker et al., <xref ref-type="bibr" rid="B66">2012</xref>). This is particularly relevant for LAB which often harbor many repeats and IS elements (Cahill et al., <xref ref-type="bibr" rid="B10">2010</xref>; Sun et al., <xref ref-type="bibr" rid="B78">2015</xref>). The phylogeny of LAB has recently been resolved in fine detail (Sun et al., <xref ref-type="bibr" rid="B78">2015</xref>). However, only two <italic>L. helveticus</italic> strains were included in that study and almost all genome assemblies were fragmented, prompting the authors to emphasize the need to add more complete genome sequences in the future. The value of complete genome sequences, both to accurately describe the pan-core genome and to identify functions uniquely encoded in individual strains, is obvious, as is the value of the development of specific diagnostic tests (Ercolini, <xref ref-type="bibr" rid="B27">2013</xref>; Hornischer and H&#x000E4;u&#x000DF;ler, <xref ref-type="bibr" rid="B38">2016</xref>) or to study genome re-arrangements, adaptation, and evolution (Ricker et al., <xref ref-type="bibr" rid="B66">2012</xref>).</p>
<p>The first complete <italic>L. helveticus</italic> genome sequence was that of strain DPC 4571, a cheese isolate. The shotgun-sequenced genome harbored a remarkably high number of IS elements (213) and 141 non-transposase encoding pseudogenes (Callanan et al., <xref ref-type="bibr" rid="B12">2008</xref>). In 2013, the complete genome of strain CNRZ 32, a strain used as a commercial cheese flavor adjunct and for the production of bioactive peptides in milk, was published (Broadbent et al., <xref ref-type="bibr" rid="B7">2013</xref>). Similar to strain DPC 4571, CNRZ 32 harbored a large number of repeats (356 IS elements and 163 non-transposase encoding pseudogenes). The unusually high number of IS elements and pseudogenes indicates ongoing genome degeneration (Callanan et al., <xref ref-type="bibr" rid="B12">2008</xref>; Broadbent et al., <xref ref-type="bibr" rid="B7">2013</xref>), a feature that has also been observed in other dairy species such as <italic>Lactobacillus casei</italic> (Cai et al., <xref ref-type="bibr" rid="B11">2009</xref>), <italic>S. thermophilus</italic> (Bolotin et al., <xref ref-type="bibr" rid="B6">2004</xref>), and <italic>L. delbrueckii</italic> (Makarova et al., <xref ref-type="bibr" rid="B52">2006</xref>). A comparison of five <italic>L. helveticus</italic> strains sequenced with second-generation short read NGS technologies (Illumina or Roche 454) was published in 2013 (Cremonesi et al., <xref ref-type="bibr" rid="B17">2013</xref>).</p>
<p>To circumvent the problems associated with short reads, we used long reads from Pacific Biosciences&#x00027; (PacBio) third-generation NGS technology (Eid et al., <xref ref-type="bibr" rid="B25">2009</xref>) and state-of-the-art assembly algorithms (Koren et al., <xref ref-type="bibr" rid="B45">2012</xref>; Chin et al., <xref ref-type="bibr" rid="B14">2013</xref>) to sequence and <italic>de novo</italic> assemble the complete genomes of three <italic>L. helveticus</italic> isolates from the dairy environment. The analysis of the repeat structure for all LAB indicated that PacBio long reads should be particularly suitable to <italic>de novo</italic> assemble genomes that contain a large number of repetitive sequences or IS elements. Here, we present the results of our study providing a phylogenetic profile of <italic>Lactobacillales</italic> with a focus on <italic>L. helveticus</italic>, and the first pan-core genome analysis based on 12 completely sequenced <italic>L. helveticus</italic> strains. Notably, the complete genome sequences proved to be very useful for analyzing NWCs with metagenomics, as a larger percentage of the sequenced reads of these complex mixtures could be unambiguously assigned even down to the strain level.</p>
</sec>
<sec sec-type="materials and methods" id="s2">
<title>Materials and methods</title>
<sec>
<title>Repeat analysis</title>
<p>All completely sequenced <italic>Lactobacillus</italic> genomes (132) were obtained from the NCBI RefSeq database (O&#x00027;Leary et al., <xref ref-type="bibr" rid="B58">2016</xref>) on December 31, 2016. Genomic repeats were identified as described before (Koren et al., <xref ref-type="bibr" rid="B44">2013</xref>) using Nucmer 3.1 (Kurtz et al., <xref ref-type="bibr" rid="B47">2004</xref>). The number of repeats (longer than 500 bp, sequence identity of 95% or greater) was plotted vs. the maximal repeat length using Seaborn 0.7.1 (<ext-link ext-link-type="uri" xlink:href="https://github.com/mwaskom/seaborn">https://github.com/mwaskom/seaborn</ext-link>, Figure <xref ref-type="fig" rid="F1">1A</xref>); two strains were excluded as they were chimeric or could not be processed.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Classification of genome assembly difficulty (Koren et al., <xref ref-type="bibr" rid="B44">2013</xref>) for selected completely sequenced genomes. <bold>(A)</bold> Genome assembly difficulty for genomes of the genus <italic>Lactobacillus</italic> including nine completely sequenced <italic>L. helveticus</italic> genomes (blue circles) and the three FAM strains sequenced in this study (red circles). For each genome, the length of the longest repeat (in kbp; y-axis) is plotted vs. the number of repeats (greater 500 bp), with more than 95% sequence identity (x-axis). The three classes of genome assembly difficulty are shown (roman numerals). <bold>(B)</bold> Overview of the total number of repeats for different <italic>Lactobacillus</italic> strains grouped according to species. Species with only one strain were excluded from the analysis. The colors indicate the assembly difficulty classification; the table on top of the graph shows the percentages per class indicated by color. The black bars show the mean value of repeats per species; the FAM strains are marked (red circles).</p></caption>
<graphic xlink:href="fmicb-09-00063-g0001.tif"/>
</fig>
</sec>
<sec>
<title>Bacterial culture and genomic DNA extraction</title>
<p><italic>L. helveticus</italic> strains FAM8105, FAM22155, and FAM8627 were obtained from the Agroscope culture collection (Agroscope, Liebefeld, Switzerland) and grown under aerobic conditions in 10 mL MRS broth (De Man et al., <xref ref-type="bibr" rid="B19">1960</xref>) at 37&#x000B0;C overnight (see Supplementary Figure <xref ref-type="supplementary-material" rid="SM12">1</xref> for light microscopy images). Bacterial cells were treated with lysozyme (50 mg/mL) for 1 h at 37&#x000B0;C. Genomic DNA (gDNA) was isolated as described elsewhere (Moser et al., <xref ref-type="bibr" rid="B57">2017</xref>), and the concentration determined (Qubit dsDNA BR Assay kit). The gDNA purity was assessed by controlling for RNA contamination using the Qubit RNA HS Assay kit (Thermo Scientific, Massachusetts, USA).</p>
</sec>
<sec>
<title>Genome sequencing and assembly</title>
<p>The gDNA was sequenced on the PacBio RS II platform (three SMRT cells per strain, P6-C4 chemistry, size-selection step (10 kb inserts) with BluePippin; for details, see Supplementary Table <xref ref-type="supplementary-material" rid="SM1">1</xref>). Subsequent <italic>de novo</italic> genome assembly using HGAP (Chin et al., <xref ref-type="bibr" rid="B14">2013</xref>) and resequencing steps with Quiver were performed as described before (Remus-Emsermann et al., <xref ref-type="bibr" rid="B64">2016</xref>). Terminal repeats were removed and the genome was circularized using Circlator 1.1.2 (Hunt et al., <xref ref-type="bibr" rid="B41">2015</xref>). Additional rounds of sequence polishing resulted in one complete chromosome and one complete plasmid sequence per strain. FAM8105 and FAM22155 were also sequenced with Illumina MiSeq (paired-end, 2 &#x000D7; 300 bp), and the reads mapped to the polished PacBio assemblies using BWA-MEM (version 0.7.10-r789, Li, <xref ref-type="bibr" rid="B50">2013</xref>). MiSeq reads which did not map to the respective chromosome and plasmid assemblies were assembled with SPAdes (Bankevich et al., <xref ref-type="bibr" rid="B5">2012</xref>) to check for the existence of additional small plasmids.</p>
</sec>
<sec>
<title>Genome coverage exploration of MiSeq data</title>
<p>We compared the genomes assembled only from Illumina MiSeq data vs. the complete PacBio assemblies (<bold>Table 2</bold>). The short read-based genome assemblies were generated with SPAdes (v3.11.0) (Bankevich et al., <xref ref-type="bibr" rid="B5">2012</xref>), requiring a minimum read coverage cutoff of 4 and a minimum contig size of 400 bp (i.e., following recommendations of the SPAdes tutorial). The assembled contigs were mapped to the PacBio assemblies for FAM8105 and FAM22155 with BWA (option -x intractg), files were parsed and the mapping quality was assessed using Samtools (v1.3.1), Bedtools (v2.26.0), and Qualimap (v2.2.1). Subsequently, all coding sequences (CDSs) and pseudogenes in areas with zero coverage were counted and classified as core, accessory or unique gene (see section Comparative Genomics). The circular plot was created with Circos (v0.69-6) (Krzywinski et al., <xref ref-type="bibr" rid="B46">2009</xref>).</p>
</sec>
<sec>
<title>Genome annotation and mining for features of interest</title>
<p>The complete genome sequences were deposited at NCBI GenBank (<bold>Table 2</bold>) and annotated by their Prokaryotic Genome Annotation Pipeline version 3.3 (Tatusova et al., <xref ref-type="bibr" rid="B79">2016</xref>). Putative CRISPR repeats (and their total number) were detected with CRISPRs finder (Grissa et al., <xref ref-type="bibr" rid="B35">2007</xref>) and PILER-CR (Edgar, <xref ref-type="bibr" rid="B24">2007</xref>). Putative Cas proteins were first searched in the NCBI annotation and secondly identified using a Hidden Markov Model (HMM) based search with HMMCAS (Chai et al., <xref ref-type="bibr" rid="B13">2017</xref>). Putative prophages were identified using PHASTER (Arndt et al., <xref ref-type="bibr" rid="B4">2016</xref>), potential genomic islands with Islandviewer 3 (Dhillon et al., <xref ref-type="bibr" rid="B20">2015</xref>); see Supplementary Table <xref ref-type="supplementary-material" rid="SM2">2</xref> for their respective predicted genome positions. All CDSs (without pseudogenes) were compared against the EggNOG 4.5 databases &#x0201C;bactNOG&#x0201D; (bacteria), &#x0201C;bacNOG&#x0201D; (bacilli), &#x0201C;firmNOG&#x0201D; (firmicutes) (Huerta-Cepas et al., <xref ref-type="bibr" rid="B40">2016</xref>), selecting only the hit with the smallest e-value (hits with e-values above 0.001 were not considered) and extracting the respective Clusters of Orthologous Groups (COG) category.</p>
<sec>
<title>GO term enrichment analysis of pseudogenes</title>
<p>First, nucleotide sequences of all annotated pseudogenes of our three strains were extracted, and six-frame translated using transeq (EMBOSS suite 6.6.0.0, <ext-link ext-link-type="uri" xlink:href="http://emboss.open-bio.org/">http://emboss.open-bio.org/</ext-link>; Rice et al., <xref ref-type="bibr" rid="B65">2000</xref>) to capture potential protein domains present in different reading frames (due to frameshifts). The translated sequences were searched against Pfam database version 31.0 (Finn et al., <xref ref-type="bibr" rid="B32">2016</xref>); hits with e-value smaller than 1e-10 were kept and the corresponding Gene Ontology (GO) terms were extracted if available. An additional Pfam search was performed with the CDS (i.e., intact protein coding genes) from all three strains followed by extraction of GO terms. Using R package &#x0201C;topGO&#x0201D; 2.26.0 (Alexa et al., <xref ref-type="bibr" rid="B2">2006</xref>) from Bioconductor 3.4 (Huber et al., <xref ref-type="bibr" rid="B39">2015</xref>), an enrichment analysis for GO terms of biologic processes (BP) was performed individually per strain (<bold>Table 4</bold>). This was done by using CDSs and pseudogenes combined and their respective GO terms as &#x0201C;gene universe&#x0201D; and pseudogenes as &#x0201C;genes of interest&#x0201D; (Ontology: &#x0201C;BP,&#x0201D; algorithm: &#x0201C;weight01,&#x0201D; statistic: &#x0201C;fisher&#x0201D;; results ranked according to <italic>p</italic>-values). For more details see &#x0201C;topGO&#x0201D; documentation.</p>
</sec>
<sec>
<title>Selected gene families of interest</title>
<p>We analyzed whether selected gene families involved in amino acid metabolism, or encoding peptide transporters, proteases, and peptidases were encoded in the three FAM, four additional <italic>L. helveticus</italic> strains and in <italic>L. acidophilus</italic> NCFM (Supplementary Tables <xref ref-type="supplementary-material" rid="SM8">8</xref>, <xref ref-type="supplementary-material" rid="SM9">9</xref>). The analysis was mainly done using blast searches for reference genes or by using the Kyoto Encyclopedia of Genes and Genomes (KEGG) database (for details, see Supplementary Methods in Supplementary Data Sheet <xref ref-type="supplementary-material" rid="SM16">1</xref>). The protein sequence of cell envelope protease (CEP, Supplementary Table <xref ref-type="supplementary-material" rid="SM8">8</xref>) PrtH3 from strain CNRZ 32 was analyzed with InterPro (<ext-link ext-link-type="uri" xlink:href="https://www.ebi.ac.uk/interpro/">https://www.ebi.ac.uk/interpro/</ext-link> Finn et al., <xref ref-type="bibr" rid="B31">2017</xref>) to retrieve protein domains, which were compared to the predicted <italic>prtH3</italic> pseuodogenes of DPC 4571 and FAM8627. An analysis for peptidoglycan hydrolases (PGH) was done involving all 12 complete <italic>L. helveticus</italic> genomes (<bold>Table 5</bold>) in a similar way. Protein domains of two PGHs were analyzed, the M23 family peptidase of strain H9 and Lysin of strain CNRZ 32. The five copies of 6-Phospho-beta-glucosidases (Supplementary Table <xref ref-type="supplementary-material" rid="SM6">6</xref>) and four genes involved in lipid metabolic processes (Supplementary Table <xref ref-type="supplementary-material" rid="SM7">7</xref>), both implied by the enrichment analysis of pseudogenes, were also analyzed in more detail for all 12 genomes.</p>
</sec>
</sec>
<sec>
<title>Phylogenetic analysis</title>
<p>The GenBank records of 24 selected, completely sequenced LAB genomes (see Supplementary Table <xref ref-type="supplementary-material" rid="SM3">3</xref>) were downloaded from NCBI RefSeq (March 3rd, 2017). This included 9 <italic>L. helveticus</italic> genomes, 14 other LAB genomes (mostly reference or representative complete genomes; 9 from genus <italic>Lactobacillus</italic> and 5 other LAB) plus <italic>Bacillus subtilis subsp. subtilis</italic> str. 168 as outgroup. The predicted CDSs of all strains, including the three <italic>L. helveticus</italic> strains of this study, were used to calculate a maximum likelihood phylogenetic tree using bcgTree (Ankenbrand and Keller, <xref ref-type="bibr" rid="B3">2016</xref>), which uses HMM models of 107 known housekeeping genes (Dupont et al., <xref ref-type="bibr" rid="B22">2012</xref>). bcgTree was parameterized to perform 100 bootstrap runs while executing RAxML (Stamatakis, <xref ref-type="bibr" rid="B74">2014</xref>), the output (<bold>Figure 4</bold>) was generated using FigTree 1.4.2 (<ext-link ext-link-type="uri" xlink:href="http://tree.bio.ed.ac.uk/software/figtree/">http://tree.bio.ed.ac.uk/software/figtree/</ext-link>).</p>
</sec>
<sec>
<title>Comparative genomics</title>
<sec>
<title>Pan-core genome prediction</title>
<p>GenBank records of the 12 <italic>L. helveticus</italic> genomes (Table <xref ref-type="table" rid="T1">1</xref>) were converted to GFF files and analyzed with Roary 3.8.0 (Page et al., <xref ref-type="bibr" rid="B61">2015</xref>) applying standard parameters (minimum blastp identity of 95%) but without paralog splitting. The number of orthologous gene clusters for pan and core genome profiles (<bold>Figure 5A</bold>) and for accessory and unique gene clusters were extracted or calculated from the respective Roary output files or gene presence/absence table. Gene cluster lists for pan, accessory, core and the 12 unique (i.e., strain-specific) genomes are provided as csv files. For every core genome cluster, a multiple sequence alignment of all protein sequences was performed with MUSCLE v3.8.31 (Edgar, <xref ref-type="bibr" rid="B23">2004</xref>) and a profile HMM was calculated using HMMBUILD (HMMER 3.1b2, <ext-link ext-link-type="uri" xlink:href="http://hmmer.org">hmmer.org</ext-link>) with default parameters. We provide one representative sequence per cluster as amino acid (aa) FASTA file and the profile HMM. For pan, accessory and unique genomes we provide just the representative sequences (aa FASTA file). All data files from the comparative genomics analysis are described and summarized in Supplementary Table <xref ref-type="supplementary-material" rid="SM11">10</xref>, the data is provided in Supplementary Data Sheet <xref ref-type="supplementary-material" rid="SM17">2</xref>.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Completely sequenced <italic>L. helveticus</italic> strains used in this study and their respective origin.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>Strain</bold></th>
<th valign="top" align="left"><bold>Accession</bold></th>
<th valign="top" align="left"><bold>Origin</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">CAUH18</td>
<td valign="top" align="left"><ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="NZ_CP012381">NZ_CP012381</ext-link></td>
<td valign="top" align="left">Isolated from Koumiss (Xinjiang Uighur Autonomous Region, China)</td>
</tr>
<tr>
<td valign="top" align="left">CNRZ 32</td>
<td valign="top" align="left"><ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="NC_021744">NC_021744</ext-link></td>
<td valign="top" align="left">Used as industrial cheese starter and cheese flavor adjunct</td>
</tr>
<tr>
<td valign="top" align="left">D76</td>
<td valign="top" align="left"><ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="NZ_CP016827">NZ_CP016827</ext-link></td>
<td valign="top" align="left">Ingredient in nutritional supplements &#x0201C;Vitaflor&#x0201D;, isolated from intestine from healthy child (Leningrad, Russia)</td>
</tr>
<tr>
<td valign="top" align="left">DPC 4571</td>
<td valign="top" align="left"><ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="NC_010080">NC_010080</ext-link></td>
<td valign="top" align="left">Cheese starter and cheese flavor adjunct isolated from Swiss cheese</td>
</tr>
<tr>
<td valign="top" align="left">H10</td>
<td valign="top" align="left"><ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="NC_017467">NC_017467</ext-link><break/> (Plasmid: <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="NC_017468">NC_017468</ext-link>)</td>
<td valign="top" align="left">Isolated from traditional fermented milk (Shigatse City of Tibet, China)</td>
</tr>
<tr>
<td valign="top" align="left">H9</td>
<td valign="top" align="left"><ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="NZ_CP002427">NZ_CP002427</ext-link></td>
<td valign="top" align="left">Isolated from kurut (Nagqu County of Tibet, China)</td>
</tr>
<tr>
<td valign="top" align="left">KLDS1.8701</td>
<td valign="top" align="left"><ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="NZ_CP009907">NZ_CP009907</ext-link><break/> (Plasmid: <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="NZ_CP009908">NZ_CP009908</ext-link>)</td>
<td valign="top" align="left">Isolated from sour milk (Sinkiang, China)</td>
</tr>
<tr>
<td valign="top" align="left">MB2-1</td>
<td valign="top" align="left"><ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="NZ_CP011386">NZ_CP011386</ext-link></td>
<td valign="top" align="left">Isolated from fermented milk (Baicheng, southern Xinjiang, China)</td>
</tr>
<tr>
<td valign="top" align="left">R0052</td>
<td valign="top" align="left"><ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="NC_018528">NC_018528</ext-link><break/> (Plasmid: <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="NC_014386">NC_014386</ext-link>)</td>
<td valign="top" align="left">Probiotic strain isolated from sweet acidophilus milk (France)</td>
</tr>
<tr>
<td valign="top" align="left"><bold>FAM8105</bold></td>
<td valign="top" align="left"><ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="CP015496">CP015496</ext-link><break/> (Plasmid: <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="CP015497">CP015497</ext-link>)</td>
<td valign="top" align="left">Isolated from raw milk (Thurgau, Switzerland)</td>
</tr>
<tr>
<td valign="top" align="left"><bold>FAM22155</bold></td>
<td valign="top" align="left"><ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="CP015498">CP015498</ext-link><break/> (Plasmid: <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="CP015499">CP015499</ext-link>)</td>
<td valign="top" align="left">Isolated from natural whey culture (Luzern, Switzerland)</td>
</tr>
<tr>
<td valign="top" align="left"><bold>FAM8627</bold></td>
<td valign="top" align="left"><ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="CP015444">CP015444</ext-link><break/> (Plasmid: <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="CP015445">CP015445</ext-link>)</td>
<td valign="top" align="left">Isolated from dairy product (not further specified) (Switzerland)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><italic>The three strains sequenced in this study are listed at the bottom (in bold). Data source: NCBI RefSeq (as of Feb 28, 2017). Strain MTCC 5643 (Prajapati et al., <xref ref-type="bibr" rid="B62">2011</xref>) is not completely sequenced (its 50 contigs can be downloaded from the NCBI); it was thus not included in our comparison</italic>.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec>
<title>COG categories</title>
<p>For clusters of core and accessory genome, as well as the unique genes, a representative sequence was extracted and the COG category was determined as described above.</p>
</sec>
</sec>
<sec>
<title>Metagenome sequencing and analysis</title>
<p>DNA from a NWC was extracted as described elsewhere (Moser et al., <xref ref-type="bibr" rid="B57">2017</xref>), DNA libraries were prepared (TruSeq DNA PCR-Free LT Library Prep Kit; insert size: 350 bp) and sequenced on an Illumina HiSeq 3000 (paired-end, 2 &#x000D7; 151 bp). Reads were assembled using SPAdes 3.9.0 (Bankevich et al., <xref ref-type="bibr" rid="B5">2012</xref>). Performing a blastn search of the resulting contigs against NCBI RefSeq, we first determined the genomes (and plasmids, where available) with most hits, requiring that at least 95% of the raw metagenome reads were assigned. This implied 8 <italic>L. helveticus</italic>, 7 <italic>S. thermophiles</italic>, and one <italic>L. delbrueckii</italic> RefSeq strain. The reads were then mapped to these genome sequences plus our three FAM strains using BWA-MEM (version 0.7.15-r1140 Li, <xref ref-type="bibr" rid="B50">2013</xref>; using option -a); the resulting SAM file was filtered to remove non mapping reads and supplementary alignments. To determine a qualitative species level distribution, we counted how many reads mapped to genomes of respective species; reads mapping to several targets were attributed in a proportional fashion. As few reads mapped to more than one species, we did not correct for different numbers of target genomes. As all genomes had roughly the same size we did not correct for genome size either. Next, we counted reads that mapped exclusively either to the three FAM or to NCBI <italic>L. helveticus</italic> genomes. To do this, the resulting SAM file from the previous steps was first parsed and a correction for the different number of target genomes (three FAM, eight NCBI strains) was performed. In a last analysis step, we determined how many of the uniquely mapping reads (i.e. reads just mapping to one target sequence) mapped to the FAM and NCBI <italic>L. helveticus</italic> genomes, respectively.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<sec>
<title>Determining the genome assembly difficulty of <italic>Lactobacillus</italic> and <italic>L. helveticus</italic></title>
<p>Aiming to increase the number of completely sequenced <italic>L. helveticus</italic> strains, we first explored the repeat structure of all <italic>Lactobacillus</italic> strains for which complete genome sequences have been deposited in NCBI&#x00027;s RefSeq. The total number of repeats and the length of the longest repeat of bacterial genomes are two parameters that have a profound impact on their genome assembly complexity (Koren et al., <xref ref-type="bibr" rid="B44">2013</xref>).</p>
<p>Using an in-house software prototype, we calculated the overall number of repeats above 500 bp and &#x0003E;95% identity identified by Nucmer (Kurtz et al., <xref ref-type="bibr" rid="B47">2004</xref>) vs. the length of the overall longest repeat (see Methods) for 130 completely sequenced <italic>Lactobacillus</italic> strains. This analysis resulted in the classification of LAB genomes into three classes with increasing demands for assembly due to the number and size of the repeats. Most genomes are classified as class I genomes (<italic>N</italic> &#x0003D; 57, 43.8%) that are straight-forward to assemble with PacBio long reads as they harbor few repeats, with the longest repeat representing multi-copy rDNA operons typically around 6&#x02013;7 kb in length (Koren et al., <xref ref-type="bibr" rid="B44">2013</xref>; Figure <xref ref-type="fig" rid="F1">1A</xref>). Furthermore, 32 (24.6%) class II genomes were observed that are characterized by the presence of a large number (more than 100) of repeats (500 bp to several kb), but none larger than the rDNA operon. Using data from the PacBio RSII platform, such genomes should also be straightforward to be <italic>de novo</italic> assembled into complete genome sequences; in contrast, relying on only Illumina short reads would produce tens to hundreds of contigs. Finally, we also noted a sizable fraction of class III genomes (41, 31.5%; Figure <xref ref-type="fig" rid="F1">1A</xref>). Due to their long, almost identical repeats well above 6&#x02013;7 kb, these genomes can be extremely difficult to assemble. Such repeats can be resolved only with very long reads, which can be obtained by including a size selection step in the library preparation.</p>
<p>We next specifically assessed <italic>L. helveticus</italic> strains, for which nine complete genomes were available at NCBI&#x00027;s RefSeq (blue circles, Figure <xref ref-type="fig" rid="F1">1A</xref>). Eight of these strains were isolates from the dairy environment, while strain D76 was described as acting as an ingredient in nutritional supplements (Table <xref ref-type="table" rid="T1">1</xref>). Eight of the genomes are class II genomes (Figure <xref ref-type="fig" rid="F1">1A</xref>), while strain H10 (Zhao et al., <xref ref-type="bibr" rid="B82">2011</xref>) is classified as a class III genome and harbors long, nearly identical repeats &#x0003E;30 kb.</p>
<p>Finally, an analysis of the predominant assembly complexity classification for different <italic>Lactobacillus</italic> species indicated that <italic>L. helveticus</italic> strains had the second highest mean repeat number among all <italic>Lactobacillus</italic> species, outnumbered only by <italic>L. backii</italic> (Geissler et al., <xref ref-type="bibr" rid="B33">2016</xref>) (Figure <xref ref-type="fig" rid="F1">1B</xref>). More importantly, due to the overall higher percentage of class III genomes, <italic>L. delbrueckii</italic> (73%), <italic>L. backii</italic> (60%), <italic>L. casei</italic> (50%), <italic>L. plantarum</italic> (43%), and <italic>L. fermentum</italic> (40%) represent <italic>Lactobacillus</italic> species that may pose considerable challenges for researchers aiming to carry out complete genome assembly projects similar to that we describe here for <italic>L. helveticus</italic>.</p>
</sec>
<sec>
<title>Genome sequencing, assembly and annotation of three <italic>L. helveticus</italic> isolates</title>
<p>We selected three <italic>L. helveticus</italic> strains (FAM8105, FAM22155, and FAM8627) from the Agroscope culture collection, which originated from different dairy products (including raw milk and natural whey cultures, Table <xref ref-type="table" rid="T1">1</xref>), and sequenced them on the PacBio RSII platform. To obtain a high sequence coverage, we used three single-molecule, real-time (SMRT) cells per strain, and to possibly even completely assemble class III genomes we used the BluePippin size selection protocol (see Methods). A high coverage would allow us to rely on algorithms that remove the random errors of PacBio reads (Chin et al., <xref ref-type="bibr" rid="B14">2013</xref>) and to obtain completely sequenced genomes of high quality.</p>
<p>This <italic>de novo</italic> genome assembly approach resulted in one completely assembled chromosome and one complete plasmid for each strain, with a PacBio sequence coverage above 150-fold except for the plasmid of strain FAM22155 (Table <xref ref-type="table" rid="T2">2</xref>). This is most likely because this plasmid is relatively small (7.5 kbp) and thus selected against in the BluePippin size selection step (see Methods). MiSeq short read data (available for FAM8105 and FAM22155) were used to check the quality of the assembly and to eliminate potential single nucleotide mis-assemblies reported for PacBio data (Laehnemann et al., <xref ref-type="bibr" rid="B48">2015</xref>). Although no evidence for any mis-assembly was found for strain FAM22155, FAM8105 harbored a single nucleotide deletion in a homopolymer stretch in the chromosome and the plasmid (data not shown), which were corrected. This confirms that our assemblies are of a high quality. In addition, we used the MiSeq data to search for small plasmids potentially missed due to the size selection step during library preparation. However, no evidence for additional small plasmids could be found. Finally, a repeat analysis of the complete genomes of FAM8105, FAM22155, and FAM8627 classified all three as class II genomes (red circles, Figure <xref ref-type="fig" rid="F1">1A</xref>).</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Genome statistics of our three completely sequenced <italic>L. helveticus</italic> strains.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th/>
<th valign="top" align="left"><bold>FAM8105</bold></th>
<th valign="top" align="left"><bold>FAM22155</bold></th>
<th valign="top" align="left"><bold>FAM8627</bold></th>
</tr>
</thead>
<tbody>
<tr style="border-bottom: thin solid #000000;">
<td valign="top" align="left"><bold>No. of chromosomes</bold></td>
<td valign="top" align="left"><bold>1</bold></td>
<td valign="top" align="left"><bold>1</bold></td>
<td valign="top" align="left"><bold>1</bold></td>
</tr>
<tr>
<td valign="top" align="left">Length</td>
<td valign="top" align="left">2,209,387 bp</td>
<td valign="top" align="left">2,191,149 bp</td>
<td valign="top" align="left">2,035,631 bp</td>
</tr>
<tr>
<td valign="top" align="left">GC content</td>
<td valign="top" align="left">37.1%</td>
<td valign="top" align="left">37.1%</td>
<td valign="top" align="left">37.0%</td>
</tr>
<tr>
<td valign="top" align="left">Total genes</td>
<td valign="top" align="left">2,217</td>
<td valign="top" align="left">2,178</td>
<td valign="top" align="left">2,057</td>
</tr>
<tr>
<td valign="top" align="left">CDSs/RNAs</td>
<td valign="top" align="left">1,876/76</td>
<td valign="top" align="left">1,849/78</td>
<td valign="top" align="left">1,691/78</td>
</tr>
<tr>
<td valign="top" align="left">Pseudogenes</td>
<td valign="top" align="left">265</td>
<td valign="top" align="left">251</td>
<td valign="top" align="left">288</td>
</tr>
<tr>
<td valign="top" align="left">Transposases</td>
<td valign="top" align="left">194</td>
<td valign="top" align="left">192</td>
<td valign="top" align="left">130</td>
</tr>
<tr>
<td valign="top" align="left">Repeat info: max. length/No. repeat pairs (class)</td>
<td valign="top" align="left">5,367/341 (class II)</td>
<td valign="top" align="left">5,466/247 (class II)</td>
<td valign="top" align="left">5,466/198 (class II)</td>
</tr>
<tr>
<td valign="top" align="left">Average coverage PacBio</td>
<td valign="top" align="left">511 x</td>
<td valign="top" align="left">354 x</td>
<td valign="top" align="left">227 x</td>
</tr>
<tr>
<td valign="top" align="left">Average coverage MiSeq</td>
<td valign="top" align="left">152 x</td>
<td valign="top" align="left">456 x</td>
<td valign="top" align="left">(no MiSeq data)</td>
</tr>
<tr>
<td valign="top" align="left">No. of detected IS elements<xref ref-type="table-fn" rid="TN1"><sup>&#x0002A;</sup></xref> (TnpPred)</td>
<td valign="top" align="left">157</td>
<td valign="top" align="left">161</td>
<td valign="top" align="left">112</td>
</tr>
<tr>
<td valign="top" align="left">No. of CRISPR clusters</td>
<td valign="top" align="left">1</td>
<td valign="top" align="left">0</td>
<td valign="top" align="left">1</td>
</tr>
<tr>
<td valign="top" align="left">Cas proteins</td>
<td valign="top" align="left">Yes</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">Yes</td>
</tr>
<tr>
<td valign="top" align="left">Phages (Supplementary Table <xref ref-type="supplementary-material" rid="SM2">2</xref>)</td>
<td valign="top" align="left">2 intact, 1 questionable, 1 incomplete</td>
<td valign="top" align="left">1 intact</td>
<td valign="top" align="left">1 incomplete</td>
</tr>
<tr style="border-bottom: thin solid #000000;">
<td valign="top" align="left">GenBank accession</td>
<td valign="top" align="left"><ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="CP015496">CP015496</ext-link></td>
<td valign="top" align="left"><ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="CP015498">CP015498</ext-link></td>
<td valign="top" align="left"><ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="CP015444">CP015444</ext-link></td>
</tr>
<tr style="border-bottom: thin solid #000000;">
<td valign="top" align="left"><bold>No. of plasmids</bold></td>
<td valign="top" align="left"><bold>1</bold></td>
<td valign="top" align="left"><bold>1</bold></td>
<td valign="top" align="left"><bold>1</bold></td>
</tr> <tr>
<td valign="top" align="left">Length</td>
<td valign="top" align="left">45,858 bp</td>
<td valign="top" align="left">7,514 bp</td>
<td valign="top" align="left">13,399 bp</td>
</tr>
<tr>
<td valign="top" align="left">GC content</td>
<td valign="top" align="left">34.2%</td>
<td valign="top" align="left">35.1%</td>
<td valign="top" align="left">34.8%</td>
</tr>
<tr>
<td valign="top" align="left">Total genes (all CDSs)</td>
<td valign="top" align="left">43</td>
<td valign="top" align="left">10</td>
<td valign="top" align="left">14</td>
</tr>
<tr>
<td valign="top" align="left">Average coverage PacBio</td>
<td valign="top" align="left">659 x</td>
<td valign="top" align="left">38 x</td>
<td valign="top" align="left">155 x</td>
</tr>
<tr>
<td valign="top" align="left">Average coverage MiSeq</td>
<td valign="top" align="left">471 x</td>
<td valign="top" align="left">3316 x</td>
<td valign="top" align="left">(no MiSeq data)</td>
</tr>
<tr>
<td valign="top" align="left">Phages</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">No</td>
</tr>
<tr>
<td valign="top" align="left">GenBank accession</td>
<td valign="top" align="left"><ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="CP015497">CP015497</ext-link></td>
<td valign="top" align="left"><ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="CP015499">CP015499</ext-link></td>
<td valign="top" align="left"><ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="CP015445">CP015445</ext-link></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="TN1">
<label>&#x0002A;</label>
<p><italic>See Supplementary Results in Supplementary Data Sheet <xref ref-type="supplementary-material" rid="SM16">1</xref>, Supplementary Table <xref ref-type="supplementary-material" rid="SM4">4</xref>, Supplementary Figure <xref ref-type="supplementary-material" rid="SM15">4</xref></italic>.</p></fn>
</table-wrap-foot>
</table-wrap>
<p>The genomes were annotated at NCBI (see Methods). A detailed summary of their genome statistics is shown in Table <xref ref-type="table" rid="T2">2</xref>. An overview of their genome features, including CDSs, rRNAs, and tRNAs, is shown for chromosomes (Figure <xref ref-type="fig" rid="F2">2</xref>) and plasmids (Supplementary Figure <xref ref-type="supplementary-material" rid="SM13">2</xref>). The distribution of functional COG categories (see Methods; Table <xref ref-type="table" rid="T3">3</xref>) was similar for all three <italic>L. helveticus</italic> strains. However, compared to annotation projects we carried out in the past (data not shown), we observed a high number of CDS classified in category &#x0201C;L&#x0201D; (&#x0201C;Replication, recombination and repair&#x0201D;). This has been reported previously for the genus <italic>Lactobacillus</italic> (Lukjancenko et al., <xref ref-type="bibr" rid="B51">2012</xref>) and can, at least in part, be attributed to the many transposase sequences in the <italic>L. helveticus</italic> genomes which get classified as class &#x0201C;L.&#x0201D; For FAM8105, 56.9% of the class &#x0201C;L&#x0201D; hits were transposases (194/341), for FAM22155, 56.3% (192/341) and for FAM8627, 47.6% (130/273) (Table <xref ref-type="table" rid="T2">2</xref>). Accordingly, among the COG categories for which a function could be assigned, category L contained the largest number of genes (Table <xref ref-type="table" rid="T3">3</xref>).</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Circular genome map for FAM8105 <bold>(A)</bold>, FAM22155 <bold>(B)</bold>, and FAM8627 <bold>(C)</bold>, generated using CGView (Stothard and Wishart, <xref ref-type="bibr" rid="B77">2005</xref>). For every sub-figure, the following features are shown (moving from the outermost track inwards, origin of replication is positioned at 0 kbp): (1) CDS on forward strand colored according to COG category, (2) CDS (<italic>black</italic>), tRNA (<italic>green</italic>) and rRNA (<italic>orange</italic>) on forward strand, (3) black line representing genome sequence, (4) CDS (<italic>black</italic>), tRNA (<italic>green</italic>) and rRNA (<italic>orange</italic>) on reverse strand, (5) Intact prophages (<italic>red</italic>), questionable prophages (<italic>light red</italic>), incomplete prophages (yellow) and genomic islands (<italic>blue</italic>), (6) CDS on reverse strand colored according to COG category, (7) GC content (<italic>black</italic>), (8) positive and negative GC skew (<italic>green</italic> and <italic>purple</italic>, respectively), and (9) genome position in kbp.</p></caption>
<graphic xlink:href="fmicb-09-00063-g0002.tif"/>
</fig>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Number of genes associated with COG functional categories for all three sequenced strains, and for core, accessory and unique genome.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>Class</bold></th>
<th valign="top" align="left"><bold>FAM8105</bold></th>
<th valign="top" align="center"><bold>FAM22155</bold></th>
<th valign="top" align="center"><bold>FAM8627</bold></th>
<th valign="top" align="center"><bold>Core genome</bold></th>
<th valign="top" align="center"><bold>Accessory genome</bold></th>
<th valign="top" align="center"><bold>Unique genome</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">J, Translation, ribosomal structure and biogenesis</td>
<td valign="top" align="center">134<break/> (7.0%)</td>
<td valign="top" align="center">132<break/> (7.1%)</td>
<td valign="top" align="center">133<break/> (7.8%)</td>
<td valign="top" align="center">125<break/> (12.7%)</td>
<td valign="top" align="center">10<break/> (0.8%)</td>
<td valign="top" align="center">1<break/> (0.1%)</td>
</tr>
<tr>
<td valign="top" align="left">K, Transcription</td>
<td valign="top" align="center">116<break/> (6.0%)</td>
<td valign="top" align="center">104<break/> (5.6%)</td>
<td valign="top" align="center">104<break/> (6.1%)</td>
<td valign="top" align="center">61<break/> (6.2%)</td>
<td valign="top" align="center">90<break/> (7.0%)</td>
<td valign="top" align="center">63<break/> (5.9%)</td>
</tr>
<tr>
<td valign="top" align="left">L, Replication, recombination and repair</td>
<td valign="top" align="center">341<break/> (17.8%)</td>
<td valign="top" align="center">341<break/> (18.3%)</td>
<td valign="top" align="center">273<break/> (16.0%)</td>
<td valign="top" align="center">83<break/> (8.4%)</td>
<td valign="top" align="center">164<break/> (12.8%)</td>
<td valign="top" align="center">79<break/> (7.4%)</td>
</tr>
<tr>
<td valign="top" align="left">[Annotated as transposase]</td>
<td valign="top" align="left">[194<break/> (10.1%)]</td>
<td valign="top" align="center">[192<break/> (10.3%)]</td>
<td valign="top" align="center">[130<break/> (7.6%)]</td>
<td/>
<td/>
<td/>
</tr>
<tr>
<td valign="top" align="left">D, Cell cycle control, Cell division, chromos. partitioning</td>
<td valign="top" align="center">22<break/> (1.1%)</td>
<td valign="top" align="center">26<break/> (1.4%)</td>
<td valign="top" align="center">22<break/> (1.3%)</td>
<td valign="top" align="center">17<break/> (1.7%)</td>
<td valign="top" align="center">8<break/> (0.6%)</td>
<td valign="top" align="center">12<break/> (1.1%)</td>
</tr>
<tr>
<td valign="top" align="left">V, Defense mechanisms</td>
<td valign="top" align="center">42<break/> (2.2%)</td>
<td valign="top" align="center">47<break/> (2.5%)</td>
<td valign="top" align="center">29<break/> (1.7%)</td>
<td valign="top" align="center">11<break/> (1.1%)</td>
<td valign="top" align="center">58<break/> (4.5%)</td>
<td valign="top" align="center">51<break/> (4.8%)</td>
</tr>
<tr>
<td valign="top" align="left">T, Signal transduction mechanisms</td>
<td valign="top" align="center">29<break/> (1.5%)</td>
<td valign="top" align="center">30<break/> (1.6%)</td>
<td valign="top" align="center">28<break/> (1.6%)</td>
<td valign="top" align="center">21<break/> (2.1%)</td>
<td valign="top" align="center">13<break/> (1.0%)</td>
<td valign="top" align="center">11<break/> (1.0%)</td>
</tr>
<tr>
<td valign="top" align="left">M, Cell wall /membrane biogenesis</td>
<td valign="top" align="center">84<break/> (4.4%)</td>
<td valign="top" align="center">86<break/> (4.6%)</td>
<td valign="top" align="center">84<break/> (4.9%)</td>
<td valign="top" align="center">56<break/> (5.7%)</td>
<td valign="top" align="center">48<break/> (3.8%)</td>
<td valign="top" align="center">89<break/> (8.3%)</td>
</tr>
<tr>
<td valign="top" align="left">N, Cell motility</td>
<td valign="top" align="center">3<break/> (0.2%)</td>
<td valign="top" align="center">3<break/> (0.2%)</td>
<td valign="top" align="center">3<break/> (0.2%)</td>
<td valign="top" align="center">3<break/> (0.3%)</td>
<td valign="top" align="center">0<break/> (0.0%)</td>
<td valign="top" align="center">0<break/> (0.0%)</td>
</tr>
<tr>
<td valign="top" align="left">U, Intracellular trafficking and secretion</td>
<td valign="top" align="center">17<break/> (0.9%)</td>
<td valign="top" align="center">18<break/> (1.0%)</td>
<td valign="top" align="center">16<break/> (0.9%)</td>
<td valign="top" align="center">15<break/> (1.5%)</td>
<td valign="top" align="center">2<break/> (0.2%)</td>
<td valign="top" align="center">0<break/> (0.0%)</td>
</tr>
<tr>
<td valign="top" align="left">O, Posttranslational modification, protein turnover, chaperones</td>
<td valign="top" align="center">46<break/> (2.4%)</td>
<td valign="top" align="center">48<break/> (2.6%)</td>
<td valign="top" align="center">44<break/> (2.6%)</td>
<td valign="top" align="center">34<break/> (3.4%)</td>
<td valign="top" align="center">23<break/> (1.8%)</td>
<td valign="top" align="center">5<break/> (0.5%)</td>
</tr>
<tr>
<td valign="top" align="left">C, Energy production and conversion</td>
<td valign="top" align="center">47<break/> (2.4%)</td>
<td valign="top" align="center">54<break/> (2.9%)</td>
<td valign="top" align="center">47<break/> (2.8%)</td>
<td valign="top" align="center">38<break/> (3.8)</td>
<td valign="top" align="center">41<break/> (3.2%)</td>
<td valign="top" align="center">14<break/> (1.3%)</td>
</tr>
<tr>
<td valign="top" align="left">G, Carbohydrate transport and metabolism</td>
<td valign="top" align="center">103<break/> (5.4%)</td>
<td valign="top" align="center">97<break/> (5.2%)</td>
<td valign="top" align="center">85<break/> (5.0%)</td>
<td valign="top" align="center">63<break/> (6.4%)</td>
<td valign="top" align="center">78<break/> (6.1%)</td>
<td valign="top" align="center">25<break/> (2.3%)</td>
</tr>
<tr>
<td valign="top" align="left">E, Amino acid transport and metabolism</td>
<td valign="top" align="center">91<break/> (4.7%)</td>
<td valign="top" align="center">102<break/> (5.5%)</td>
<td valign="top" align="center">96<break/> (5.6%)</td>
<td valign="top" align="center">54<break/> (5.5%)</td>
<td valign="top" align="center">80<break/> (6.3%)</td>
<td valign="top" align="center">27<break/> (2.5%)</td>
</tr>
<tr>
<td valign="top" align="left">F, Nucleotide transport and metabolism</td>
<td valign="top" align="center">81<break/> (4.2%)</td>
<td valign="top" align="center">80<break/> (4.3%)</td>
<td valign="top" align="center">80<break/> (4.7%)</td>
<td valign="top" align="center">47<break/> (4.8%)</td>
<td valign="top" align="center">61<break/> (4.8%)</td>
<td valign="top" align="center">5<break/> (0.5%)</td>
</tr>
<tr>
<td valign="top" align="left">H, Coenzyme transport and metabolism</td>
<td valign="top" align="center">35<break/> (1.8%)</td>
<td valign="top" align="center">34<break/> (1.8%)</td>
<td valign="top" align="center">32<break/> (1.9%)</td>
<td valign="top" align="center">23<break/> (2.3%)</td>
<td valign="top" align="center">17<break/> (1.3%)</td>
<td valign="top" align="center">8<break/> (0.7%)</td>
</tr>
<tr>
<td valign="top" align="left">I, Lipid transport and metabolism</td>
<td valign="top" align="center">38<break/> (2.0%)</td>
<td valign="top" align="center">36<break/> (1.9%)</td>
<td valign="top" align="center">34<break/> (2.0%)</td>
<td valign="top" align="center">30<break/> (3.0%)</td>
<td valign="top" align="center">10<break/> (0.8%)</td>
<td valign="top" align="center">2<break/> (0.2%)</td>
</tr>
<tr>
<td valign="top" align="left">P, Inorganic ion transport and metabolism</td>
<td valign="top" align="center">64<break/> (3.3%)</td>
<td valign="top" align="center">70<break/> (3.8%)</td>
<td valign="top" align="center">65<break/> (3.8%)</td>
<td valign="top" align="center">47<break/> (4.8%)</td>
<td valign="top" align="center">35<break/> (2.7%)</td>
<td valign="top" align="center">6<break/> (0.6%)</td>
</tr>
<tr>
<td valign="top" align="left">Q, Secondary metabolites biosynthesis, transport and catabolism</td>
<td valign="top" align="center">6<break/> (0.3%)</td>
<td valign="top" align="center">4<break/> (0.2%)</td>
<td valign="top" align="center">4<break/> (0.2%)</td>
<td valign="top" align="center">2<break/> (0.2%)</td>
<td valign="top" align="center">5<break/> (0.4%)</td>
<td valign="top" align="center">1<break/> (0.1%)</td>
</tr>
<tr>
<td valign="top" align="left">S, Function unknown</td>
<td valign="top" align="center">487<break/> (25.4%)</td>
<td valign="top" align="center">444<break/> (23.9%)</td>
<td valign="top" align="center">408<break/> (23.9%)</td>
<td valign="top" align="center">229<break/> (23.2%)</td>
<td valign="top" align="center">383<break/> (30.0%)</td>
<td valign="top" align="center">383<break/> (35.8%)</td>
</tr>
<tr>
<td valign="top" align="left">Not in COG category</td>
<td valign="top" align="center">133<break/> (6.9%)</td>
<td valign="top" align="center">103<break/> (5.5%)</td>
<td valign="top" align="center">118<break/> (6.9%)</td>
<td valign="top" align="center">29<break/> (2.9%)</td>
<td valign="top" align="center">152<break/> (11.9%)</td>
<td valign="top" align="center">287<break/> (26.8%)</td>
</tr>
<tr>
<td valign="top" align="left">Total CDS</td>
<td valign="top" align="center">1,919<break/> (100%)</td>
<td valign="top" align="center">1,859<break/> (100%)</td>
<td valign="top" align="center">1,705<break/> (100%)</td>
<td valign="top" align="center">988<break/> (100%)</td>
<td valign="top" align="center">1,278<break/> (100%)</td>
<td valign="top" align="center">1,069<break/> (100%)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><italic>COG categories which contained no genes/gene clusters are not shown. Numbers in brackets show percentage as fraction of total number of CDSs</italic>.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec>
<title>Comparison of short and long read-based assemblies with respect to gene coverage</title>
<p>Despite the value of complete genome sequences, they are still highly underrepresented in public sequence repositories. This is also true for LAB: Among the 213 strains recently analyzed, almost all the genome assemblies (&#x0003E; 96%) were fragmented (Sun et al., <xref ref-type="bibr" rid="B78">2015</xref>). The number of contigs reported for the LAB strains ranged from 8 to 964 (a median of about 90). A comparison of the FAM8105 and FAM22155 genomes assembled only from Illumina MiSeq data using SPAdes vs. the respective final PacBio-based reference genomes provided an insight into the extent of such differences (Figure <xref ref-type="fig" rid="F3">3</xref>): About 230 regions with an average size of roughly 1,000 bp were not covered by the MiSeq assemblies, which amounted to 10&#x02013;12% of the actual genome of these two strains (genome size around 2.2 Mbp; Table <xref ref-type="table" rid="T2">2</xref>). Notably, for the repeat-rich <italic>L. helveticus</italic> genomes, around 10% of the annotated CDSs would be missed in addition to the rRNA operons that represent the largest repeats in the genome, i.e., roughly 200 genes (Table <xref ref-type="table" rid="T2">2</xref>). For pseudogenes, the percentage of missed cases was even higher, surpassing 20% of all pseudogenes.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Genomic regions missed in the assembly based on MiSeq reads. A circular plot of FAM8105 and FAM22155 is shown, where the outer ring (1) represents the complete PacBio assembly of FAM8105 (light blue) and FAM22155 (dark blue) along with their plasmids. Moving inwards, the contigs of the MiSeq assembly are shown next (gray boxes, 2), then lines that illustrate the coverage gaps in the MiSeq assemblies (black, 3), and finally genes which are affected by the gaps in the MiSeq assembly (4; their colors correspond to the legend in the top left).</p></caption>
<graphic xlink:href="fmicb-09-00063-g0003.tif"/>
</fig>
</sec>
<sec>
<title>Phylogenetic tree with a focus on <italic>L. helveticus</italic></title>
<p>To ensure reliable phylogenetic placement of the newly sequenced strains, we selected several taxonomically diverse LAB genomes on top of all completely sequenced <italic>L. helveticus</italic> genomes from NCBI (Supplementary Table <xref ref-type="supplementary-material" rid="SM3">3</xref>). The maximum likelihood phylogenetic tree generated from these genomes is based on 107 known housekeeping genes and has very good bootstrap support (Figure <xref ref-type="fig" rid="F4">4</xref>). The three newly sequenced strains form a monophyletic clade with the other nine <italic>L. helveticus</italic> strains (Figure <xref ref-type="fig" rid="F4">4A</xref>), which is clearly separated from the other groups and is characterized by a relatively low degree of intra-clade variation. As reported previously, the genus <italic>Lactobacillus</italic> is not monophyletic, but paraphyletic (Mayr and Bock, <xref ref-type="bibr" rid="B54">2002</xref>; Sun et al., <xref ref-type="bibr" rid="B78">2015</xref>); i.e., the <italic>Lactobacillus</italic> clade also includes <italic>Leuconostoc, Oenococcus</italic>, and <italic>Pediococcus</italic> species.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Maximum likelihood phylogenetic tree of completely sequenced <italic>L. helveticus</italic> strains, in the context of several key LAB strains. Phylogenetic tree was constructed using a concatenated alignment of 107 known housekeeping genes (Dupont et al., <xref ref-type="bibr" rid="B22">2012</xref>). <bold>(A)</bold> The collapsed <italic>L. helveticus</italic> clade relative to other LAB bacteria. Bootstrap scores for all nodes are shown (percentage of 100 bootstrap runs). The bar at the bottom represents the number of amino acid substitutions per site. <italic>Bacillus subtilis</italic> subsp<italic>. subtilis</italic> 168 served as outgroup. <bold>(B)</bold> Expanded <italic>L. helveticus</italic> clade based on the same calculation as above. The three strains of this study are shown in bold. The <italic>L. helveticus</italic> clade has a 100 times higher resolution than the complete tree which is reflected by the bar. The symbol showing crossed scissors indicates two strains without a CRISPR/Cas system; the numbers in a black box indicate how many CRISPR spacers were detected in total per strain.</p></caption>
<graphic xlink:href="fmicb-09-00063-g0004.tif"/>
</fig>
<p>The three FAM strains fall into two different subgroups within the <italic>L. helveticus</italic> clade (Figure <xref ref-type="fig" rid="F4">4B</xref>). FAM8627 is close to CNRZ 32 (Christiansen et al., <xref ref-type="bibr" rid="B15">2008</xref>; Broadbent et al., <xref ref-type="bibr" rid="B7">2013</xref>), a strain used as a starter culture and for the production of bioactive peptides in milk. FAM8105 and FAM22155 are more closely related to DPC 4571 (Hannon et al., <xref ref-type="bibr" rid="B37">2003</xref>; Callanan et al., <xref ref-type="bibr" rid="B12">2008</xref>), another strain known to be beneficial in cheese production. Of note, strains FAM22155 and CNRZ 32 seem to lack a CRISPR/Cas system as neither CRISPR repeats nor Cas proteins were detected (Figure <xref ref-type="fig" rid="F4">4B</xref>).</p>
</sec>
<sec>
<title>Comparative genomics of 12 <italic>L. helveticus</italic> strains</title>
<p>To the best of our knowledge and as noted in a recent review (Stefanovic et al., <xref ref-type="bibr" rid="B75">2017</xref>), no pan-core genome study has been reported for <italic>L. helveticus</italic>. Thus, we carried out such an analysis on the 12 complete genomes using Roary (Page et al., <xref ref-type="bibr" rid="B61">2015</xref>). The pan genome is generally defined as the sum of all genes in a species, whereas the core genome is defined as the orthologous genes that are present in all strains of a species (Medini et al., <xref ref-type="bibr" rid="B56">2005</xref>). Furthermore, the accessory genome comprises orthologous gene clusters (orthologous genes from here on are referred to as &#x0201C;gene clusters&#x0201D; or &#x0201C;genes,&#x0201D; depending on the context) that, in our example, are found in at least two and up to 11 strains. Finally, we also determined genes that occur in only one of the strains; these genes represent the unique or &#x0201C;strain-specific&#x0201D; genome.</p>
<p>As more genomes were added, the size of the core genome diminished and reached 988 gene clusters for all 12 genomes (Figure <xref ref-type="fig" rid="F5">5A</xref>). The curve of the pan genome hints at a still &#x0201C;open&#x0201D; pan genome, comparable to results from a similar-sized pan-core genome study for 17 <italic>L. casei</italic> genomes (Broadbent et al., <xref ref-type="bibr" rid="B8">2012</xref>). Overall, we identified a pan genome of 3,335 gene clusters (Figure <xref ref-type="fig" rid="F5">5B</xref>). They could be further divided into a core genome of 988 gene clusters (29.6%) present in all 12 strains, an accessory genome of 1,278 gene clusters (38.3%) present in a subset of the 12 strains (Supplementary Figure <xref ref-type="supplementary-material" rid="SM14">3</xref>), and a total of 1,069 strain-specific genes (32.1%). The number of strain-specific gene clusters ranged from 29 genes only found in strain H9 up to 225 gene clusters unique to strain R0052 (Figure <xref ref-type="fig" rid="F5">5B</xref>). Notably, among the CDSs missed in the two MiSeq assemblies (Figure <xref ref-type="fig" rid="F3">3</xref>), not only accessory genes were missed, but also around 30 core genes, i.e., roughly 3% of the close to 1,000 core genes identified.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Statistics of the <italic>L. helveticus</italic> pan- and core genome based on 12 completely sequenced strains. <bold>(A)</bold> Number of gene clusters for pan (dashed line) and core genome profile (straight line) when successively adding genomes. The curves show mean values of 10 iterations where genomes were randomly added and the pan and core genome was calculated successively for all 12 steps. <bold>(B)</bold> Number of gene clusters (reflected by the size of the respective areas) for core, accessory, pan and unique genome.</p></caption>
<graphic xlink:href="fmicb-09-00063-g0005.tif"/>
</fig>
<p>To identify whether gene clusters of certain functional groups are preferentially found in all 12 strains (core genome) or if they diverge more between the strains, COG functional categories for core, accessory and unique genome of all 12 <italic>L. helveticus</italic> strains were explored (Table <xref ref-type="table" rid="T3">3</xref>; see Methods). As expected, the COG functional categories of some highly conserved biological processes, such as &#x0201C;Translation, ribosomal structure, and biogenesis&#x0201D; (Class J), and &#x0201C;Intracellular trafficking and secretion&#x0201D; (Class U), were enriched among the core genes and depleted in the unique genome. In contrast, the category K &#x0201C;Transcription&#x0201D; was equally present in the core, accessory and unique genomes, indicative of specific transcriptional regulatory mechanisms in some of the strains. For the class &#x0201C;Replication, recombination, and repair&#x0201D; (L), enrichment in the accessory genome was observed. As mentioned, this is mainly caused by transposases which are often found in more than one of the 12 strains but not in all. Enrichment among accessory and unique genes was observed for &#x0201C;Defense mechanisms&#x0201D; (Class V). Importantly, for the unique genes, we observed enrichment of genes in the COG functional category &#x0201C;Cell wall/membrane biogenesis&#x0201D; (Class M). Finally, genes of unknown function (Class S) were enriched in the unique genome, as well as genes not assigned to any COG class. For the latter, the largest difference between their presence in the core genome (very low) vs. the unique genome could be detected (Table <xref ref-type="table" rid="T3">3</xref>).</p>
</sec>
<sec>
<title>Genome mining for pseudogenes and gene families of interest</title>
<sec>
<title>GO term enrichment analysis of pseudogenes</title>
<p>Enrichment of certain genomic features can help to elucidate lifestyle adaptations, such as the adaptation to a new biological niche (D&#x00027;Souza and Kost, <xref ref-type="bibr" rid="B21">2016</xref>). To analyze whether certain biological processes were overrepresented among the large number of annotated pseudogenes (Table <xref ref-type="table" rid="T2">2</xref>), we tested for enrichment of GO terms in the pseudogenes vs. a background distribution of GO terms in the CDSs plus pseudogenes of the three FAM <italic>L. helveticus</italic> genomes (see Methods).</p>
<p>GO terms associated with amino acid membrane transport, carbohydrate and lipid metabolic processes were overrepresented among the pseudogenes of the three FAM strains (Table <xref ref-type="table" rid="T4">4</xref>), which can be explained with an adaptation to the nutrient-rich milk environment. The inactivation of genes involved in amino acid biosynthesis in such an environment is well-known (Makarova et al., <xref ref-type="bibr" rid="B52">2006</xref>; Callanan et al., <xref ref-type="bibr" rid="B12">2008</xref>; Christiansen et al., <xref ref-type="bibr" rid="B15">2008</xref>; Cremonesi et al., <xref ref-type="bibr" rid="B17">2013</xref>). The genes for 6-phospho-beta-glucosidases are grouped under the GO term &#x0201C;carbohydrate metabolic process&#x0201D; (<ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="GO:0005975">GO:0005975</ext-link>); their encoded enzymes are associated with the hydrolysis of glucosidic bonds and contain the InterPro domain &#x0201C;glycoside hydrolase family 1.&#x0201D; All five share high similarity (Supplementary Table <xref ref-type="supplementary-material" rid="SM6">6</xref>): Three gene products are just below 500 amino acids in length, one member is annotated as a pseudogene that covers the N-terminal 211 aa, while the last member encodes a short 48 aa protein covering the C-terminal part of the three longer proteins. Exploring these five members in all 12 complete genomes, we found that three were inactivated not only in the three FAM strains, but also in several other completely sequenced <italic>L. helveticus</italic> strains (Supplementary Table <xref ref-type="supplementary-material" rid="SM6">6</xref>).</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>GO terms overrepresented among the pseudogenes in our three <italic>L. helveticus</italic> strains.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>FAM8105</bold></th>
<th valign="top" align="left"><bold>FAM22155</bold></th>
<th valign="top" align="left"><bold>FAM8627</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Amino acid transmembrane transport;<break/> <italic>p</italic>-value &#x0003D; <bold>0.012</bold><break/> (GO:0003333)</td>
<td valign="top" align="left">Carbohydrate metabolic process;<break/> <italic>p</italic>-value &#x0003D; <bold>0.0031</bold><break/> (GO:0005975)</td>
<td valign="top" align="left">Amino acid transmembrane transport;<break/> <italic>p</italic>-value &#x0003D; <bold>0.0025</bold><break/> (GO:0003333)</td>
</tr>
<tr>
<td valign="top" align="left">Lipid metabolic process;<break/> <italic>p</italic>-value &#x0003D; <bold>0.026</bold><break/> (GO:0006629)</td>
<td valign="top" align="left">Amino acid transmembrane transport;<break/> <italic>p</italic>-value &#x0003D; <bold>0.0483</bold><break/> (GO:0003333)</td>
<td valign="top" align="left">Transposition, DNA-mediated;<break/> <italic>p</italic>-value &#x0003D; <bold>0.0081</bold><break/> (GO:0006313)</td>
</tr>
<tr>
<td valign="top" align="left">Transmembrane transport;<break/> <italic>p</italic>-value &#x0003D; <bold>0.026</bold><break/> (GO:0055085)</td>
<td valign="top" align="left">Lipid metabolic process;<break/> <italic>p</italic>-value &#x0003D; 0.0735<break/> (GO:0006629)</td>
<td valign="top" align="left">Lipid metabolic process;<break/> <italic>p</italic>-value &#x0003D; <bold>0.0431</bold><break/> (GO:0006629)</td>
</tr>
<tr>
<td valign="top" align="left">Carbohydrate metabolic process;<break/> <italic>p</italic>-value &#x0003D; 0.098<break/> (GO:0005975)</td>
<td valign="top" align="left">Regulation of transcription; DNA-templated;<break/> <italic>p</italic>-value &#x0003D; 0.0970<break/> (GO:0006355)</td>
<td valign="top" align="left">DNA recombination;<break/> <italic>p</italic>-value &#x0003D; 0.0579<break/> (GO:0006310)</td>
</tr>
<tr>
<td valign="top" align="left">Glutamine metabolic process;<break/> <italic>p</italic>-value &#x0003D; 0.099<break/> (GO:0006541)</td>
<td valign="top" align="left">Glutamine metabolic process;<break/> <italic>p</italic>-value &#x0003D; 0.097<break/> (GO:0006541)</td>
<td valign="top" align="left">Phosphoenolpyruvate-dependent sugar phosphotransferase;<break/> <italic>p</italic>-value &#x0003D; 0.0711<break/> (GO:0009401)</td>
</tr>
<tr>
<td valign="top" align="left">Terpenoid biosynthetic process;<break/> <italic>p</italic>-value &#x0003D; 0.099<break/> (GO:0016114)</td>
<td valign="top" align="left">Terpenoid biosynthetic process;<break/> <italic>p</italic>-value &#x0003D; 0.097<break/> (GO:0016114)</td>
<td valign="top" align="left">Isoprenoid biosynthetic process<break/> <italic>p</italic>-value &#x0003D; 0.0795<break/> (GO:0008299)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><italic>The six terms with lowest p-values (see Methods) are shown for each strain. Values below a significance level of 0.05 are shown in bold</italic>.</p>
</table-wrap-foot>
</table-wrap>
<p>Finally, enrichment for lipid metabolic processes could hint at changes in enzymes that may affect the lipid composition of the cell membrane. An overview of four selected gene products belonging to this class is shown in Supplementary Table <xref ref-type="supplementary-material" rid="SM7">7</xref>, three of which were inactivated in a subset of the strains.</p>
</sec>
<sec>
<title>Genes relevant for cheese ripening</title>
<p>Protein degradation, amino acid catabolism, and autolysis are major biochemical processes taking place during cheese ripening. Thus, we took a closer look at genes associated with these processes and analyzed the three newly sequenced FAM strains, four strains that broadly cover the <italic>L. helveticus</italic> clade (including DPC 4571, CNRZ 32, H10, and R0052; Figure <xref ref-type="fig" rid="F4">4B</xref>) and the <italic>L. acidophilus</italic> strain NCFM (Supplementary Tables <xref ref-type="supplementary-material" rid="SM8">8</xref>, <xref ref-type="supplementary-material" rid="SM9">9</xref>).</p>
<p>All three FAM strains and the dairy isolates DPC 4571 and CNRZ 32 harbored the gene for the cell membrane-localized CEP PrtH3 (Supplementary Table <xref ref-type="supplementary-material" rid="SM8">8</xref>). However, a frameshift that likely renders the protein inactive was present in the N-terminal coding region of <italic>prtH3</italic> in FAM8627 (Figure <xref ref-type="fig" rid="F6">6A</xref>). This strain encodes two additional CEPs named PrtH1 and PrtH4. These CEPs are also present in CNRZ 32, which possesses a total of four CEP-encoding genes, but they are absent in the other strains (Supplementary Table <xref ref-type="supplementary-material" rid="SM8">8</xref>). To address which of the CEPs could possess an important function in <italic>L. helveticus</italic>, we measured proteolytic activity in the cell wall fraction (see Supplementary Methods in Supplementary Data Sheet <xref ref-type="supplementary-material" rid="SM16">1</xref>) of the three FAM strains. FAM8627 did not exhibit any detectable proteolytic activity, indicating that PrtH3 was the predominant CEP present in the cell wall extracts used here (Supplementary Table <xref ref-type="supplementary-material" rid="SM5">5</xref>; Supplementary Methods in Supplementary Data Sheet <xref ref-type="supplementary-material" rid="SM16">1</xref>) and providing a genotype-phenotype link (Figure <xref ref-type="fig" rid="F6">6A</xref>). The <italic>prtH3</italic> gene of DPC 4571 carried a stop codon in the C-terminal coding region, and thus, likely encodes a functional protein lacking the S-layer domains and an immunoglobulin domain, potentially affecting additional functions of the protein or interaction(s) with other proteins (Figure <xref ref-type="fig" rid="F6">6A</xref>).</p>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>Overview of predicted InterPro protein domains of selected cell membrane proteins of interest. <bold>(A)</bold> Protein domains for CEP PrtH3 (blue bars; prediction source on the left). The green bar represents PrtH3 of strain CNRZ 32. The two predicted <italic>prtH3</italic> pseudogenes from DPC 4571 (premature stop codon predicted close to C-Terminus, 99.3% aa identity until stop codon to DPC 4571) and FAM8627 (100% aa identity until frameshift close to N-terminus) are shown in orange. <bold>(B)</bold> Protein domains for a PGH gene inactivated only in <italic>L. helveticus</italic> strain H9. The green bar represents the intact PGH with a M23 family peptidase domain (Locus tag LHV_RS06490) of strain DPC 4571. The frameshift in the N-terminus (aa position 64) of the ortholog in strain H9 is predicted to inactivate protein function as the protein no longer contains the peptidase M23 domain. <bold>(C)</bold> Protein domains for a PGH gene inactivated only in strain CNRZ 32. The green bar represents the intact PGH with a glycoside hydrolase domain (Locus tag LHV_RS07135) of strain DPC 4571. The frameshift occurs at aa position 274 and thus likely inactivates the original function of the protein since the two SLAP domains are in the frame-shifted region.</p></caption>
<graphic xlink:href="fmicb-09-00063-g0006.tif"/>
</fig>
<p>The GO term enrichment analysis of the pseudogenes had revealed that amino acid membrane transport was one process affected in the FAM strains (Table <xref ref-type="table" rid="T4">4</xref>), i.e., a function not needed in a nutrient-rich milk environment. This observation is further supported by the fact that two or more copies of several peptide transporter operons are encoded in the respective genomes (Supplementary Table <xref ref-type="supplementary-material" rid="SM8">8</xref>). All analyzed strains (except FAM8105 and H10) carried genes encoding several oligopeptide transporter operons and the peptidase complement (Supplementary Table <xref ref-type="supplementary-material" rid="SM8">8</xref>). Amino acid proto- and auxotrophies have been reported for strain CNRZ 32 (Christiansen et al., <xref ref-type="bibr" rid="B15">2008</xref>). Therefore, we compared the gene products associated with amino acid metabolism of this strain with the eight strains mentioned above (Supplementary Table <xref ref-type="supplementary-material" rid="SM9">9A</xref>). The analyzed strains did not possess complete biosynthetic pathways for Arg, Glu, His, Ile, Leu, Lys, Phe, Pro, Thr (except strain H10), Trp, Tyr, and Val.</p>
<p>Researchers have suggested that variations in the autolytic potential of <italic>L. helveticus</italic> strains are linked to either differences in the set of peptidoglycan hydrolases (PGHs) (Jebava et al., <xref ref-type="bibr" rid="B43">2011</xref>) and/or the cell wall composition (Vinogradov et al., <xref ref-type="bibr" rid="B81">2013</xref>). Based on PCR experiments, Jebava and colleagues proposed that nine peptidoglycan hydrolases are ubiquitous genes in <italic>L. helveticus</italic>. In contrast to these results, we identified only five of these nine genes in the functional core genome (Table <xref ref-type="table" rid="T5">5</xref>). These five gene products were intact in all strains and were highly conserved to their DPC 4571 ortholog (&#x0003E;98% average pairwise aa identity, Table <xref ref-type="table" rid="T5">5</xref>). In contrast, the remaining genes were inactivated in one or several strains; therefore, those genes were designated as pseudogenes in the annotation and thus, are not part of the core genome. Due to frameshifts, two genes were annotated as pseudogenes in one of the 12 strains. The ortholog of LHV_RS06490 in strain H9 contained a frameshift in the N-terminal third of the encoded protein, which is predicted to abrogate its function (Figure <xref ref-type="fig" rid="F6">6B</xref>). For the ortholog of LHV_RS07135 in strain CNRZ 32, the frameshift in the C-terminal third would leave out two S-layer protein domains, and thus, may have a more subtle effect (Figure <xref ref-type="fig" rid="F6">6C</xref>). However, the frameshift could also affect important interactions with other proteins or carbohydrates and affect attachment to the cell envelope (Hyn&#x000F6;nen and Palva, <xref ref-type="bibr" rid="B42">2013</xref>). The two remaining genes (LHV_RS06550 and LHV_RS10160) showed more variability among the strains: Mutations that likely lead to inactive gene products were observed in four (LHV_RS06550) and nine strains (LHV_RS10160), respectively (Table <xref ref-type="table" rid="T5">5</xref>). In summary, the data indicate that there is considerable variability concerning the relevant genes, including proteins at the cell surface.</p>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>Overview of peptidoglycan hydrolase (PGH) genes in complete <italic>L. helveticus</italic> genomes.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>Strain</bold></th>
<th valign="top" align="left"><bold>LHV_RS00930 N-acetylmuramidase</bold></th>
<th valign="top" align="left"><bold>LHV_RS00935 Amidase</bold></th>
<th valign="top" align="left"><bold>LHV_RS02820 N-acetylmuramidase</bold></th>
<th valign="top" align="left"><bold>LHV_RS03290 Lysozyme</bold></th>
<th valign="top" align="left"><bold>LHV_RS05260 N-acetylmuramidase</bold></th>
<th valign="top" align="left"><bold>LHV_RS06490 M23 family peptidase</bold></th>
<th valign="top" align="left"><bold>LHV_RS07135 Lysin</bold></th>
<th valign="top" align="left"><bold>LHV_RS06550 M23 family peptidase</bold></th>
<th valign="top" align="left"><bold>LHV_RS10160 <xref ref-type="table-fn" rid="TN2"><sup>&#x0002A;1</sup></xref> Lysin</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">FAM8105</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">P (FS)</td>
</tr>
<tr>
<td valign="top" align="left">FAM22155</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">P (FS)</td>
</tr>
<tr>
<td valign="top" align="left">FAM8627</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">P (FS)</td>
</tr>
<tr>
<td valign="top" align="left">CAUH18</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">P (ST)</td>
<td valign="top" align="left">&#x02713;</td>
</tr>
<tr>
<td valign="top" align="left">CNRZ 32</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">P (FS)</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">P (FS)</td>
</tr>
<tr>
<td valign="top" align="left">D76</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">P (FS)</td>
</tr>
<tr>
<td valign="top" align="left">DPC 4571</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">P (FS)</td>
</tr>
<tr>
<td valign="top" align="left">H10</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">P (ST)</td>
<td valign="top" align="left">&#x02713;</td>
</tr>
<tr>
<td valign="top" align="left">H9</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">P (FS)</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">P (FS)</td>
<td valign="top" align="left">P (FS)</td>
</tr>
<tr>
<td valign="top" align="left">KLDS1.8701</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">P (FS)</td>
</tr>
<tr>
<td valign="top" align="left">MB2-1</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">P (FS)</td>
<td valign="top" align="left">P (FS)</td>
</tr>
<tr>
<td valign="top" align="left">R0052</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="left">&#x02713;</td>
</tr>
<tr>
<td valign="top" align="left">In core genome</td>
<td valign="top" align="left">Yes</td>
<td valign="top" align="left">Yes</td>
<td valign="top" align="left">Yes</td>
<td valign="top" align="left">Yes</td>
<td valign="top" align="left">Yes</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">No</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><italic>The locus tags of nine PGH genes of strain DPC 4571 are shown on top along with the annotation of the encoded protein. The nine genes were selected based on a detailed study of PGHs (Jebava et al., <xref ref-type="bibr" rid="B43">2011</xref>). PGHs annotated as functional are marked with a tick mark (&#x02713;), pseudogenes are marked with &#x0201C;P&#x0201D;, the reason for the pseudogene annotation is shown in brackets. Five of the nine PGH genes belonged to the core genome and were highly conserved (&#x0003E;98% average pairwise aa identity, using BLSM62 substitution-scoring matrix)</italic>.</p>
<p><italic>P (ST): Pseudogene with premature stop codon</italic>.</p>
<p><italic>P (FS): Pseudogene with frameshift</italic>.</p>
<fn id="TN2">
<label>&#x0002A;1</label>
<p><italic>Frameshift according to RefSeq annotation</italic>.</p></fn>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec>
<title>Metagenome analysis of cheese starter cultures</title>
<p>For the production of Gruy&#x000E8;re cheese, the use of NWCs, where <italic>L. helveticus</italic> is a predominant species, is regulated. For quality management, therefore, determining the composition of NWCs at the species, and preferably at the strain level, is very relevant. To test whether the complete genomes of the three newly sequenced strains could contribute to a better assignment of the species composition of NWCs, and in particular, whether the genomes could help to distinguish different strains of a species (Smid et al., <xref ref-type="bibr" rid="B73">2014</xref>), a metagenomic analysis of a Gruy&#x000E8;re whey starter culture from a cheese factory in Switzerland was performed using the Illumina HiSeq platform.</p>
<p>First, the species composition of the NWC was determined. Using a reference-based approach, the Illumina HiSeq reads were mapped to the genomes of the strains that were able to explain 95% of the reads of the metagenome sample (see Methods). About 57% of the reads originated from <italic>L. helveticus</italic>, 34% from <italic>S. thermophilus</italic> and 5% from <italic>L. delbrueckii</italic>, respectively (Figure <xref ref-type="fig" rid="F7">7</xref>, pie chart). Another 5% of the reads could not be mapped. However, this last percentage might possibly decrease if the&#x02014;based on the mapped reads&#x02014;&#x0201C;rare&#x0201D; species in the sample were also considered, which was beyond the scope of this analysis. The percentages from this whole genome sequencing-based metagenomics approach are comparable to those reported in a reverse transcriptase length heterogeneity PCR-based analysis of the composition of NWCs in Grana Padano (Rossetti et al., <xref ref-type="bibr" rid="B67">2008</xref>), where the domain A of the variable 16S rRNA gene was assessed. <italic>L. helveticus</italic> strains were always dominant, while the percentages of <italic>S. thermophilus</italic> and <italic>L. delbrueckii</italic> seemed to vary in these cheese whey starters.</p>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>Metagenomics of NWCs benefits from the availability of complete genome data. The metagenomic analysis of a NWC reveals the qualitative species level composition (pie chart). A mapping of the reads against all 12 <italic>L. helveticus</italic> genomes provides a rough separation of reads that can map to several of the NCBI complete genomes vs. reads assigned to our three FAM strains (upper right panel; see Methods). A focus on sequences that uniquely map to only one <italic>L. helveticus</italic> strain is shown in the lower panel.</p></caption>
<graphic xlink:href="fmicb-09-00063-g0007.tif"/>
</fig>
<p>Next, the relevance of the three FAM strains in relation to existing NCBI RefSeq <italic>L. helveticus</italic> strains was assessed. To achieve this, the mapping information above was filtered for two groups: (1) reads mapping only to FAM strains and (2) reads mapping only to NCBI RefSeq strains (Figure <xref ref-type="fig" rid="F7">7</xref>, right upper panel). Although this information provides only a &#x0201C;semi-quantitative&#x0201D; picture, nevertheless, it emphasizes the relevance of the FAM strains for this NWC from a Gruy&#x000E8;re cheese.</p>
<p>Finally, we determined the numbers of all uniquely mapped HiSeq reads, i.e., reads that mapped exclusively to one target sequence among the <italic>L. helveticus</italic> strains. The result of this analysis showed a high relevance of FAM8105 and FAM22155 compared to the RefSeq <italic>L. helveticus</italic> reference strains. In contrast, strain FAM8627 had virtually no unique mappings (Figure <xref ref-type="fig" rid="F7">7</xref>, right lower panel).</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<p>Since the introduction in 2009, the average read length of the PacBio third-generation NGS technology has been steadily increasing (Eid et al., <xref ref-type="bibr" rid="B25">2009</xref>). Together with modern genome assembly algorithms (Koren et al., <xref ref-type="bibr" rid="B45">2012</xref>; Chin et al., <xref ref-type="bibr" rid="B14">2013</xref>), this technology is revolutionizing the ability to sequence microbial genomes and to subsequently study their function. Our data demonstrated that by using long read technologies, even repeat-rich class II genomes can readily be <italic>de novo</italic> assembled into complete and highly accurate genome sequences. The comparison of the MiSeq- and PacBio-based assemblies emphasized that existing strain differences can be overlooked, and that not only accessory genes but also core genes can be missed by the fragmented assemblies based on short reads. PacBio data, thus, are particularly suited to describe full genomes of LAB, which harbor a large number of repeats. Although more expensive, a complete genome sequence based on long read technologies represents an optimal basis for subsequent accurate and in-depth genome annotation (Omasits et al., <xref ref-type="bibr" rid="B60">2017</xref>). Furthermore, complete genomes enable researchers to study genome rearrangements and evolution over time (Callanan et al., <xref ref-type="bibr" rid="B12">2008</xref>), and provide the basis for using strain-specific sequences for diagnostic purposes (Ercolini, <xref ref-type="bibr" rid="B27">2013</xref>), to create accurate, genome-scale metabolic models (Stefanovic et al., <xref ref-type="bibr" rid="B75">2017</xref>), and to carry out functional genomics studies relying on condition-specific gene or protein expression data (Omasits et al., <xref ref-type="bibr" rid="B59">2013</xref>).</p>
<p>Compared to a large phylogenomic profiling study of LAB (Sun et al., <xref ref-type="bibr" rid="B78">2015</xref>) which included two <italic>L. helveticus</italic> strains, we provided a more detailed phylogeny of <italic>L. helveticus</italic> strains, which formed a distinct clade among LAB. Notably, two strains from different subclades (CNRZ 32 and FAM22155) lacked a detectable CRISPR/Cas system. An assessment of other LAB using pre-computed datasets from CRISPRfinder (<ext-link ext-link-type="uri" xlink:href="http://crispr.i2bc.paris-saclay.fr/crispr/">http://crispr.i2bc.paris-saclay.fr/crispr/</ext-link>) indicated that this system was also lost in a few of the analyzed <italic>L. acidophilus</italic> and <italic>L. delbrueckii</italic> strains. In contrast, loss among <italic>L. johnsonii</italic> strains was more prominent, as three out of five (60%) lacked the system. Thus, the CRISPR/Cas system may not be as useful for strain typing as previously proposed (Selle and Barrangou, <xref ref-type="bibr" rid="B71">2015</xref>). These strains have likely acquired other defense mechanisms against phage attack, which is quite frequent in the dairy environment (Samson and Moineau, <xref ref-type="bibr" rid="B69">2013</xref>). Some of these mechanisms can likely be found among the enriched COG class of &#x0201C;defense mechanisms&#x0201D; (V) among accessory and unique genes (see below).</p>
<p>The results of the first pan-core genome analysis for <italic>L. helveticus</italic> based on 12 complete genomes indicated that the ratio of pan and core genes was comparable to that reported for <italic>L. casei</italic> (Broadbent et al., <xref ref-type="bibr" rid="B8">2012</xref>). Although the core genome was enriched in genes of known function, 132 of the 988 core gene clusters (13.2%) were annotated as &#x0201C;hypothetical protein.&#x0201D; Particularly interesting was the enrichment of the functional COG categories &#x0201C;cell wall and membrane biogenesis&#x0201D; (M) among the unique genes and &#x0201C;defense mechanisms&#x0201D; (V) among the accessory and unique genes. The COG class &#x0201C;unknown function&#x0201D; (S) was also overrepresented among unique genes, indicating that significant effort will be required to unravel the putative functions carried out by the unique genes on top of the known role of surface-localized proteins. Furthermore, some of the strain-specific sequences included insertion elements (IE) and transposons (Supplementary Results in Supplementary Data Sheet <xref ref-type="supplementary-material" rid="SM16">1</xref>), which could be exploited for diagnostic purposes.</p>
<p>The pseudogene analysis also supported the observation that the nutrient-rich conditions encountered by <italic>L. helveticus</italic> strains in their natural habitat favor the accumulation of repeats and insertion sequences and that their genomes are undergoing reductive genome evolution (Callanan et al., <xref ref-type="bibr" rid="B12">2008</xref>; Broadbent et al., <xref ref-type="bibr" rid="B7">2013</xref>). Consistent with this, it has been recently shown in <italic>Escherichia coli</italic> that gene loss in nutrient-rich environments can serve as a significant fitness advantage for auxotrophic mutants (D&#x00027;Souza and Kost, <xref ref-type="bibr" rid="B21">2016</xref>). Moreover, the pseudogene analysis indicated that genes for lipid metabolic processes were affected in all three FAM strains, in particular genes involved in isoprenoid biosynthesis. This observation can be explained as an adaptation to a low pH environment. During milk fermentation, the pH naturally drops due to lactic acid production. To counteract this stress factor, <italic>L. helveticus</italic> could have evolved to preferentially use acetyl-CoA for the biosynthesis of saturated fatty acids instead of isoprenoids to stabilize the cell membrane. This hypothesis is in line with a proteomic study by (Fernandez et al., <xref ref-type="bibr" rid="B30">2008</xref>), who found that <italic>L. delbrueckii</italic> subsp. <italic>bulgaricus</italic> repressed enzymes involved in isoprenoid biosynthesis during acid stress.</p>
<p>Our analysis of amino acid metabolism genes based on KEGG pathways suggested that all strains are auxotrophs for at least 12 amino acids, which is in accordance with strain CNRZ 32 that was described to be auxotroph for 14 amino acids (Christiansen et al., <xref ref-type="bibr" rid="B15">2008</xref>). Moreover, amino acid transport systems are often found among pseudogenes. This seems to be an evolutionary consequence of the low amount of free amino acids present in milk. The presence of oligopeptide transport systems and a broad peptidase complement in <italic>L. helveticus</italic> suggests that all essential amino acids are supplied by the internal breakdown of peptides in this species.</p>
<p>The genome mining effort provided direct evidence for a genotype to phenotype link for PrtH3, a member of the CEPs, which correlated with the biochemical analysis for strain FAM8627. The genome mining also suggested an indirect link for the PGH gene family that will require further experiments. The PGH complement and the cell wall composition, have been postulated to represent factors that contribute to different autolytic potential of <italic>L. helveticus</italic> strains (Jebava et al., <xref ref-type="bibr" rid="B43">2011</xref>; Vinogradov et al., <xref ref-type="bibr" rid="B81">2013</xref>). As Jebava et al. reported that 24 <italic>L. helveticus</italic> strains expressed all nine genes, differential gene expression does not seem to be related to different autolytic properties. The present data&#x02014;in contrast to Jebava et al.&#x00027;s PCR data&#x02014;suggested several differences in the PGH complement among the strains, and only five of nine genes were present in the core genome. The remaining four genes were mutated in at least one of the strains likely resulting in at least partial loss of their function. These <italic>in silico</italic> analyses should thus ideally be further complemented by in-depth proteomics profiling experiments (Ahrens et al., <xref ref-type="bibr" rid="B1">2010</xref>), such as a comprehensive analysis of the subcellular localization data of condition-specific proteomes including rich surface proteomes (Stekhoven et al., <xref ref-type="bibr" rid="B76">2014</xref>), to explore a potential correlation between differential protein expression and varying autolytic properties. However, in line with the alternative hypothesis that the cell wall composition is a key factor for autolysis (Vinogradov et al., <xref ref-type="bibr" rid="B81">2013</xref>), we observed that the COG category for &#x0201C;cell wall/membrane biogenesis&#x0201D; was enriched in the unique genes of <italic>L. helveticus</italic> strains indicating that cell wall composition may vary between strains. Thus, further studies on not only membrane proteins but also the chemical composition and the biochemical synthesis of the cell wall are needed to help unravel the molecular mechanisms of autolysis in <italic>L. helveticus</italic>.</p>
<p>The availability of more complete genomes is highly relevant to study the composition of metagenomes in more detail and beyond 16S rRNA analysis (De Filippis et al., <xref ref-type="bibr" rid="B18">2014</xref>; Ellegaard and Engel, <xref ref-type="bibr" rid="B26">2016</xref>). The present whole genome sequencing-based metagenome analysis of an NWC demonstrated that complete genome sequences can help to decipher the strain composition in moderately complex metagenomes (Erkus et al., <xref ref-type="bibr" rid="B28">2013</xref>), such as those observed in raw milk or cheese starter cultures (Smid et al., <xref ref-type="bibr" rid="B73">2014</xref>). Particularly promising is the potential to assemble the genomes of different strains directly from such moderately complex mixtures (Sangwan et al., <xref ref-type="bibr" rid="B70">2016</xref>). Although this assembly will be challenging when closely related genomes are present in the mixture (Brown, <xref ref-type="bibr" rid="B9">2015</xref>), complete genome information is one of the key factors to further exploit the exceptional potential of lactobacilli for various biotechnological applications.</p>
</sec>
<sec id="s5">
<title>Data access</title>
<p>The genome sequences of the three <italic>L. helveticus</italic> strains are available from NCBI GenBank under accession numbers <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="CP015496">CP015496</ext-link> &#x00026; <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="CP015497">CP015497</ext-link> (FAM8105), <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="CP015498">CP015498</ext-link> &#x00026; <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="CP015499">CP015499</ext-link> (FAM22155), and <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="CP015444">CP015444</ext-link> &#x00026; <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="CP015445">CP015445</ext-link> (FAM8627) (Table <xref ref-type="table" rid="T1">1</xref>). Furthermore, raw sequence data (and methylation analysis) has been submitted to the NCBI Sequence Read Archive (SRA): <ext-link ext-link-type="NCBI:sra" xlink:href="SRX1725197">SRX1725197</ext-link> (FAM8105), <ext-link ext-link-type="NCBI:sra" xlink:href="SRX1726542">SRX1726542</ext-link> (FAM2155), <ext-link ext-link-type="NCBI:sra" xlink:href="SRX1726359">SRX1726359</ext-link> (FAM8607), see also Supplementary Table <xref ref-type="supplementary-material" rid="SM1">1</xref>.</p>
</sec>
<sec id="s6">
<title>Author contributions</title>
<p>DM, AV, and MS: assembled genomes, MS: carried out bioinformatic analyses, genome annotation, and created figures, JM: mined the genomes for genes of interest; AM: cultivated <italic>L. helveticus</italic> strains, extracted gDNA and created light microscopy images; AW: set up the SMRT portal and developed the repeat analysis web server together with MB; VS: explored differences between short read and PacBio based assemblies; CW: performed enzymatic assays; JF and EE-M: participated in study design and data interpretation; SI: oversaw the culturing, biochemical analyses, and selection of genes of interest; CA: conceived the study, oversaw bioinformatics analyses, repeat server functionality; MS and CA: wrote the paper.</p>
<sec>
<title>Conflict of interest statement</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
</sec>
</body>
<back>
<ack><p>The authors acknowledge support of the Agroscope Research Program Microbial BioDiversity. DM was jointly financed by the ZHAW and Agroscope. The authors thank Sergey Koren for advice to create the repeat analysis, Florian Freimoser for feedback on the manuscript, and Tosso Leeb of the NGS platform of the University of Bern, for library prep and sequencing of whey cultures. CA acknowledges support for AV from the Swiss National Science Foundation (SNSF) under grant 31003A-156320.</p>
</ack>
<sec sec-type="supplementary-material" id="s7">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fmicb.2018.00063/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fmicb.2018.00063/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Table1.DOCX" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Table 1</label>
<caption><p>Metrics for the PacBio SMRT sequencing runs and corresponding NCBI Sequence Read Archive (SRA) accession numbers.</p></caption></supplementary-material>
<supplementary-material xlink:href="Table2.DOCX" id="SM2" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Table 2</label>
<caption><p>Genomic positions of predicted genomic islands and prophages.</p></caption></supplementary-material>
<supplementary-material xlink:href="Table3.DOCX" id="SM3" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Table 3</label>
<caption><p>Overview of bacterial strains used for phylogenetic analyses in this study.</p></caption></supplementary-material>
<supplementary-material xlink:href="Table4.DOCX" id="SM4" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Table 4</label>
<caption><p>IS elements identified in the three <italic>L. helveticus</italic> genomes. Additional to the analysis using TnpPred, the IS elements were also identified using ISfinder (<ext-link ext-link-type="uri" xlink:href="https://www-is.biotoul.fr">https://www-is.biotoul.fr</ext-link>). For our three FAM strains, this resulted in more IS elements than the TnpPred analysis. Some of the IS elements were specific to one <italic>L. helveticus</italic> FAM strain and thus might be used for diagnostic applications. Pseudogenes which are attributed to IS sequences are noted in round brackets. IS sequences localized in plasmids are noted in square brackets (they only occurred in FAM8105).</p></caption></supplementary-material>
<supplementary-material xlink:href="Table5.DOCX" id="SM5" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Table 5</label>
<caption><p>CEP activity of <italic>L. helveticus</italic> FAM8105, FAM22155 and FAM8627.</p></caption></supplementary-material>
<supplementary-material xlink:href="Table6.DOCX" id="SM6" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Table 6</label>
<caption><p>Overview of predicted 6-phospho-beta-glucosidase gene products in complete <italic>L. helveticus</italic> genomes. Presence/absence table for five 6-phospho-beta-glucosidase CDSs detected in the twelve completely sequenced strains. Tick marks (&#x02713;) represent genes which are detected and predicted to be functional. &#x0201C;P&#x0201D; marks genes that were predicted as pseudogenes by the NCBI annotation. In the first row the accession number of a representative NCBI RefSeq protein is given for every group.</p></caption></supplementary-material>
<supplementary-material xlink:href="Table7.DOCX" id="SM7" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Table 7</label>
<caption><p>Overview of genes related to lipid metabolism for all complete <italic>L. helveticus</italic> genomes. Presence/absence table for four genes related to lipid metabolism detected either as intact or pseudogene in all completely sequenced strains. Tick marks (&#x02713;) represent genes which are detected and predicted to be functional. &#x0201C;P&#x0201D; marks genes that were predicted as pseudogenes by the NCBI annotation pipeline.</p></caption></supplementary-material>
<supplementary-material xlink:href="Table8.DOCX" id="SM8" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Table 8</label>
<caption><p>Analysis of the presence of peptide transporters, proteinases and peptidases in selected Lactobacillus strains. White, yellow and green table cells indicate absence, single and multiple genes, respectively.</p></caption></supplementary-material>
<supplementary-material xlink:href="Table9.DOCX" id="SM9" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Table 9A</label>
<caption><p><italic>In silico</italic> analysis of the amino acid biosynthetic capabilities of various Lactobacillus strains based on KEGG pathway annotation.</p></caption></supplementary-material>
<supplementary-material xlink:href="Table9.DOCX" id="SM10" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Table 9B</label>
<caption><p>Details of <italic>in silico</italic> analysis for amino acid metabolism based on KEGG pathway.</p></caption></supplementary-material>
<supplementary-material xlink:href="Table10.DOCX" id="SM11" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Table 10</label>
<caption><p>Additional result files of the pan-core genome analysis of 12 <italic>L. helveticus</italic> genomes. We provide these files as a resource to the community; most files are text files (csv &#x00026; faa) and are using Linux style formatted line breaks. In addition, we provide HMM profiles for core genome clusters.</p></caption></supplementary-material>
<supplementary-material xlink:href="Image1.JPEG" id="SM12" mimetype="image/jpeg" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Figure 1</label>
<caption><p>Light microscopic images of the three <italic>L. helveticus</italic> strains. <bold>(A)</bold> FAM8105, <bold>(B)</bold> FAM22155, and <bold>(C)</bold> FAM8627. In agreement with earlier reports, <italic>L. helveticus</italic> cells are predominantly rods or coccobacilli (Claesson et al., <xref ref-type="bibr" rid="B16">2007</xref>).</p></caption></supplementary-material>
<supplementary-material xlink:href="Image2.JPEG" id="SM13" mimetype="image/jpeg" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Figure 2</label>
<caption><p>Circular maps for plasmids of FAM8105, FAM22155, and FAM8627. The plots were generated using CGview (Stothard and Wishart, <xref ref-type="bibr" rid="B77">2005</xref>). For each subfigure <bold>(A)</bold> FAM8105, <bold>(B)</bold> FAM22155, and <bold>(C)</bold> FAM8627, the following features are shown (moving from the outermost track inwards): (1) CDS on forward strand colored according to COG category, (2) CDS (<italic>black</italic>) on forward strand, (3) black line representing genome sequence, (4) CDS (<italic>black</italic>) on reverse strand, (5) CDS on reverse strand colored according to COG category, (6) GC content (<italic>black</italic>), (7) positive and negative GC skew (<italic>green</italic> and <italic>purple</italic>, respectively) and (8) genome position in kbp.</p></caption></supplementary-material>
<supplementary-material xlink:href="Image3.JPEG" id="SM14" mimetype="image/jpeg" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Figure 3</label>
<caption><p>Barplot showing the distribution of core, accessory and unique gene clusters among the 12 <italic>L. helveticus</italic> strains. The y-axis shows the number of gene clusters for every category, the x-axis shows how many strains contribute to the respective clusters. On the leftmost position (&#x0201C;1&#x0201D;) the number of clusters with gene(s) from only one strain is shown (&#x0201C;Unique genome&#x0201D;). On the rightmost position (&#x0201C;12&#x0201D;) the same is shown for the core genome (genes present in all strains). Everything in between (&#x0201C;2&#x0201D; &#x02013; &#x0201C;11&#x0201D;) corresponds to gene clusters of the accessory genome.</p></caption></supplementary-material>
<supplementary-material xlink:href="Image4.JPEG" id="SM15" mimetype="image/jpeg" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Figure 4</label>
<caption><p>Distribution of the occurrence of eleven insertion sequence (IS) families among 12 <italic>L. helveticus</italic> strains shown as a heatmap. The background color corresponds to the number of ISs detected using TnpPred for an IS family for the respective strain (reflecting log values used for clustering). Strains and IS are clustered (hierarchical clustering using average linkage and euclidean distance based on log values) and the dendrogram is shown on top for IS and on the left for the strains. For FAM8105, the ISs detected on the plasmid are shown in brackets. White boxes at the bottom and at the right show the total for IS families and strains, respectively. For the remaining eight families (IS1, IS1380, IS21, IS481, IS630, IS91, ISAs1, Tn3), no hits were observed.</p></caption></supplementary-material>
<supplementary-material xlink:href="DataSheet1.DOCX" id="SM16" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Data Sheet 1</label>
<caption><p>Supplementary Material and Methods &#x00026; Results.</p></caption></supplementary-material>
<supplementary-material xlink:href="DataSheet2.ZIP" id="SM17" mimetype="application/zip" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Data Sheet 2</label>
<caption><p>Result files of pan-core genome analysis. For descriptions, see Supplementary Table <xref ref-type="supplementary-material" rid="SM11">10</xref>.</p></caption></supplementary-material>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ahrens</surname> <given-names>C. H.</given-names></name> <name><surname>Brunner</surname> <given-names>E.</given-names></name> <name><surname>Qeli</surname> <given-names>E.</given-names></name> <name><surname>Basler</surname> <given-names>K.</given-names></name> <name><surname>Aebersold</surname> <given-names>R.</given-names></name></person-group> (<year>2010</year>). <article-title>Generating and navigating proteome maps using mass spectrometry</article-title>. <source>Nat. Rev. Mol. Cell Biol.</source> <volume>11</volume>, <fpage>789</fpage>&#x02013;<lpage>801</lpage>. <pub-id pub-id-type="doi">10.1038/nrm2973</pub-id><pub-id pub-id-type="pmid">20944666</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alexa</surname> <given-names>A.</given-names></name> <name><surname>Rahnenf&#x000FC;hrer</surname> <given-names>J.</given-names></name> <name><surname>Lengauer</surname> <given-names>T.</given-names></name></person-group> (<year>2006</year>). <article-title>Improved scoring of functional groups from gene expression data by decorrelating GO graph structure</article-title>. <source>Bioinformatics</source> <volume>22</volume>, <fpage>1600</fpage>&#x02013;<lpage>1607</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btl140</pub-id><pub-id pub-id-type="pmid">16606683</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ankenbrand</surname> <given-names>M. J.</given-names></name> <name><surname>Keller</surname> <given-names>A.</given-names></name></person-group> (<year>2016</year>). <article-title>bcgTree: automatized phylogenetic tree building from bacterial core genomes</article-title>. <source>Genome</source> <volume>59</volume>, <fpage>783</fpage>&#x02013;<lpage>791</lpage>. <pub-id pub-id-type="doi">10.1139/gen-2015-0175</pub-id><pub-id pub-id-type="pmid">27603265</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Arndt</surname> <given-names>D.</given-names></name> <name><surname>Grant</surname> <given-names>J. R.</given-names></name> <name><surname>Marcu</surname> <given-names>A.</given-names></name> <name><surname>Sajed</surname> <given-names>T.</given-names></name> <name><surname>Pon</surname> <given-names>A.</given-names></name> <name><surname>Liang</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>PHASTER: a better, faster version of the PHAST phage search tool</article-title>. <source>Nucleic Acids Res</source>. <volume>44</volume>, <fpage>W16</fpage>&#x02013;<lpage>W21</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkw387</pub-id><pub-id pub-id-type="pmid">27141966</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bankevich</surname> <given-names>A.</given-names></name> <name><surname>Nurk</surname> <given-names>S.</given-names></name> <name><surname>Antipov</surname> <given-names>D.</given-names></name> <name><surname>Gurevich</surname> <given-names>A. A.</given-names></name> <name><surname>Dvorkin</surname> <given-names>M.</given-names></name> <name><surname>Kulikov</surname> <given-names>A. S.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>SPAdes: a new genome assembly algorithm and its applications to single-cell sequencing</article-title>. <source>J. Comput. Biol</source>. <volume>19</volume>, <fpage>455</fpage>&#x02013;<lpage>477</lpage>. <pub-id pub-id-type="doi">10.1089/cmb.2012.0021</pub-id><pub-id pub-id-type="pmid">22506599</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bolotin</surname> <given-names>A.</given-names></name> <name><surname>Quinquis</surname> <given-names>B.</given-names></name> <name><surname>Renault</surname> <given-names>P.</given-names></name> <name><surname>Sorokin</surname> <given-names>A.</given-names></name> <name><surname>Ehrlich</surname> <given-names>S. D.</given-names></name> <name><surname>Kulakauskas</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2004</year>). <article-title>Complete sequence and comparative genome analysis of the dairy bacterium <italic>Streptococcus thermophilus</italic></article-title>. <source>Nat. Biotechnol</source>. <volume>22</volume>, <fpage>1554</fpage>&#x02013;<lpage>1558</lpage>. <pub-id pub-id-type="doi">10.1038/nbt1034</pub-id><pub-id pub-id-type="pmid">15543133</pub-id></citation></ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Broadbent</surname> <given-names>J. R.</given-names></name> <name><surname>Hughes</surname> <given-names>J. E.</given-names></name> <name><surname>Welker</surname> <given-names>D. L.</given-names></name> <name><surname>Tompkins</surname> <given-names>T. A.</given-names></name> <name><surname>Steele</surname> <given-names>J. L.</given-names></name></person-group> (<year>2013</year>). <article-title>Complete genome sequence for <italic>Lactobacillus helveticus</italic> CNRZ 32, an industrial cheese starter and cheese flavor adjunct</article-title>. <source>Genome Announc</source>. <volume>1</volume>:<fpage>e00590</fpage>-<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1128/genomeA.00590-13</pub-id><pub-id pub-id-type="pmid">23969047</pub-id></citation></ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Broadbent</surname> <given-names>J. R.</given-names></name> <name><surname>Neeno-Eckwall</surname> <given-names>E. C.</given-names></name> <name><surname>Stahl</surname> <given-names>B.</given-names></name> <name><surname>Tandee</surname> <given-names>K.</given-names></name> <name><surname>Cai</surname> <given-names>H.</given-names></name> <name><surname>Morovic</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>Analysis of the <italic>Lactobacillus casei</italic> supragenome and its influence in species evolution and lifestyle adaptation</article-title>. <source>BMC Genomics</source> <volume>13</volume>:<fpage>533</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2164-13-533</pub-id><pub-id pub-id-type="pmid">23035691</pub-id></citation></ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brown</surname> <given-names>C. T.</given-names></name></person-group> (<year>2015</year>). <article-title>Strain recovery from metagenomes</article-title>. <source>Nat. Biotechnol</source>. <volume>33</volume>, <fpage>1041</fpage>&#x02013;<lpage>1043</lpage>. <pub-id pub-id-type="doi">10.1038/nbt.3375</pub-id><pub-id pub-id-type="pmid">26448087</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cahill</surname> <given-names>M. J.</given-names></name> <name><surname>K&#x000F6;ser</surname> <given-names>C. U.</given-names></name> <name><surname>Ross</surname> <given-names>N. E.</given-names></name> <name><surname>Archer</surname> <given-names>J. A. C.</given-names></name></person-group> (<year>2010</year>). <article-title>Read length and repeat resolution: exploring prokaryote genomes using next-generation sequencing technologies</article-title>. <source>PLoS ONE</source> <volume>5</volume>:<fpage>e11518</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0011518</pub-id><pub-id pub-id-type="pmid">20634954</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cai</surname> <given-names>H.</given-names></name> <name><surname>Thompson</surname> <given-names>R.</given-names></name> <name><surname>Budinich</surname> <given-names>M. F.</given-names></name> <name><surname>Broadbent</surname> <given-names>J. R.</given-names></name> <name><surname>Steele</surname> <given-names>J. L.</given-names></name></person-group> (<year>2009</year>). <article-title>Genome sequence and comparative genome analysis of <italic>Lactobacillus casei</italic>: insights into their niche-associated evolution</article-title>. <source>Genome Biol. Evol</source>. <volume>1</volume>, <fpage>239</fpage>&#x02013;<lpage>257</lpage>. <pub-id pub-id-type="doi">10.1093/gbe/evp019</pub-id><pub-id pub-id-type="pmid">20333194</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Callanan</surname> <given-names>M.</given-names></name> <name><surname>Kaleta</surname> <given-names>P.</given-names></name> <name><surname>O&#x00027;Callaghan</surname> <given-names>J.</given-names></name> <name><surname>O&#x00027;Sullivan</surname> <given-names>O.</given-names></name> <name><surname>Jordan</surname> <given-names>K.</given-names></name> <name><surname>McAuliffe</surname> <given-names>O.</given-names></name> <etal/></person-group>. (<year>2008</year>). <article-title>Genome sequence of <italic>Lactobacillus helveticus</italic>, an organism distinguished by selective gene loss and insertion sequence element expansion</article-title>. <source>J. Bacteriol</source>. <volume>190</volume>, <fpage>727</fpage>&#x02013;<lpage>735</lpage>. <pub-id pub-id-type="doi">10.1128/JB.01295-07</pub-id><pub-id pub-id-type="pmid">17993529</pub-id></citation></ref>
<ref id="B13">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chai</surname> <given-names>G.</given-names></name> <name><surname>Yu</surname> <given-names>M.</given-names></name> <name><surname>Jiang</surname> <given-names>L.</given-names></name> <name><surname>Duan</surname> <given-names>Y.</given-names></name> <name><surname>Huang</surname> <given-names>J.</given-names></name></person-group> (<year>2017</year>). <article-title>HMMCAS: a web tool for the identification and domain annotations of Cas proteins</article-title>, in <source>IEEE/ACM Transactions on Computational Biology and Bioinformatics</source> (<publisher-loc>New York, NY</publisher-loc>). <pub-id pub-id-type="doi">10.1109/TCBB.2017.2665542</pub-id></citation></ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chin</surname> <given-names>C.-S.</given-names></name> <name><surname>Alexander</surname> <given-names>D. H.</given-names></name> <name><surname>Marks</surname> <given-names>P.</given-names></name> <name><surname>Klammer</surname> <given-names>A. A.</given-names></name> <name><surname>Drake</surname> <given-names>J.</given-names></name> <name><surname>Heiner</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>Nonhybrid, finished microbial genome assemblies from long-read SMRT sequencing data</article-title>. <source>Nat. Methods</source> <volume>10</volume>, <fpage>563</fpage>&#x02013;<lpage>569</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.2474</pub-id><pub-id pub-id-type="pmid">23644548</pub-id></citation></ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Christiansen</surname> <given-names>J. K.</given-names></name> <name><surname>Hughes</surname> <given-names>J. E.</given-names></name> <name><surname>Welker</surname> <given-names>D. L.</given-names></name> <name><surname>Rodr&#x000ED;guez</surname> <given-names>B. T.</given-names></name> <name><surname>Steele</surname> <given-names>J. L.</given-names></name> <name><surname>Broadbent</surname> <given-names>J. R.</given-names></name></person-group> (<year>2008</year>). <article-title>Phenotypic and genotypic analysis of amino acid auxotrophy in <italic>Lactobacillus helveticus</italic> CNRZ 32</article-title>. <source>Appl. Environ. Microbiol</source>. <volume>74</volume>, <fpage>416</fpage>&#x02013;<lpage>423</lpage>. <pub-id pub-id-type="doi">10.1128/AEM.01174-07</pub-id><pub-id pub-id-type="pmid">17993552</pub-id></citation></ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Claesson</surname> <given-names>M. J.</given-names></name> <name><surname>van Sinderen</surname> <given-names>D.</given-names></name> <name><surname>O&#x00027;Toole</surname> <given-names>P. W.</given-names></name></person-group> (<year>2007</year>). <article-title>The genus Lactobacillus - a genomic basis for understanding its diversity</article-title>. <source>FEMS Microbiol. Lett</source>. <volume>269</volume>, <fpage>22</fpage>&#x02013;<lpage>28</lpage>. <pub-id pub-id-type="doi">10.1111/j.1574-6968.2006.00596.x</pub-id><pub-id pub-id-type="pmid">17343688</pub-id></citation></ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cremonesi</surname> <given-names>P.</given-names></name> <name><surname>Chessa</surname> <given-names>S.</given-names></name> <name><surname>Castiglioni</surname> <given-names>B.</given-names></name></person-group> (<year>2013</year>). <article-title>Genome sequence and analysis of <italic>Lactobacillus helveticus</italic></article-title>. <source>Front. Microbiol</source>. <volume>3</volume>:<fpage>435</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2012.00435</pub-id><pub-id pub-id-type="pmid">23335916</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>De Filippis</surname> <given-names>F.</given-names></name> <name><surname>La Storia</surname> <given-names>A.</given-names></name> <name><surname>Stellato</surname> <given-names>G.</given-names></name> <name><surname>Gatti</surname> <given-names>M.</given-names></name> <name><surname>Ercolini</surname> <given-names>D.</given-names></name></person-group> (<year>2014</year>). <article-title>A selected core microbiome drives the early stages of three popular italian cheese manufactures</article-title>. <source>PLoS ONE</source> <volume>9</volume>:<fpage>e89680</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0089680</pub-id><pub-id pub-id-type="pmid">24586960</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>De Man</surname> <given-names>J. C.</given-names></name> <name><surname>Rogosa</surname> <given-names>M.</given-names></name> <name><surname>Sharpe</surname> <given-names>E. M.</given-names></name></person-group> (<year>1960</year>). <article-title>A medium for the cultivation of Lactobacilli</article-title>. <source>J. Appl. Bacteriol</source>. <volume>23</volume>, <fpage>130</fpage>&#x02013;<lpage>135</lpage>. <pub-id pub-id-type="doi">10.1111/j.1365-2672.1960.tb00188.x</pub-id></citation></ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dhillon</surname> <given-names>B. K.</given-names></name> <name><surname>Laird</surname> <given-names>M. R.</given-names></name> <name><surname>Shay</surname> <given-names>J. A.</given-names></name> <name><surname>Winsor</surname> <given-names>G. L.</given-names></name> <name><surname>Lo</surname> <given-names>R.</given-names></name> <name><surname>Nizam</surname> <given-names>F.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>IslandViewer 3: more flexible, interactive genomic island discovery, visualization and analysis</article-title>. <source>Nucleic Acids Res</source>. <volume>43</volume>, <fpage>W104</fpage>&#x02013;<lpage>W108</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkv401</pub-id><pub-id pub-id-type="pmid">25916842</pub-id></citation></ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>D&#x00027;Souza</surname> <given-names>G.</given-names></name> <name><surname>Kost</surname> <given-names>C.</given-names></name></person-group> (<year>2016</year>). <article-title>Experimental evolution of metabolic dependency in bacteria</article-title>. <source>PLoS Genet</source>. <volume>12</volume>:<fpage>e1006364</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pgen.1006364</pub-id><pub-id pub-id-type="pmid">27814362</pub-id></citation></ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dupont</surname> <given-names>C. L.</given-names></name> <name><surname>Rusch</surname> <given-names>D. B.</given-names></name> <name><surname>Yooseph</surname> <given-names>S.</given-names></name> <name><surname>Lombardo</surname> <given-names>M.-J.</given-names></name> <name><surname>Richter</surname> <given-names>R. A.</given-names></name> <name><surname>Valas</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>Genomic insights to SAR86, an abundant and uncultivated marine bacterial lineage</article-title>. <source>ISME J</source>. <volume>6</volume>, <fpage>1186</fpage>&#x02013;<lpage>1199</lpage>. <pub-id pub-id-type="doi">10.1038/ismej.2011.189</pub-id><pub-id pub-id-type="pmid">22170421</pub-id></citation></ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Edgar</surname> <given-names>R. C.</given-names></name></person-group> (<year>2004</year>). <article-title>MUSCLE: multiple sequence alignment with high accuracy and high throughput</article-title>. <source>Nucleic Acids Res</source>. <volume>32</volume>, <fpage>1792</fpage>&#x02013;<lpage>1797</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkh340</pub-id><pub-id pub-id-type="pmid">15034147</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Edgar</surname> <given-names>R. C.</given-names></name></person-group> (<year>2007</year>). <article-title>PILER-CR: fast and accurate identification of CRISPR repeats</article-title>. <source>BMC Bioinformatics</source> <volume>8</volume>:<fpage>18</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-8-18</pub-id><pub-id pub-id-type="pmid">17239253</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Eid</surname> <given-names>J.</given-names></name> <name><surname>Fehr</surname> <given-names>A.</given-names></name> <name><surname>Gray</surname> <given-names>J.</given-names></name> <name><surname>Luong</surname> <given-names>K.</given-names></name> <name><surname>Lyle</surname> <given-names>J.</given-names></name> <name><surname>Otto</surname> <given-names>G.</given-names></name> <etal/></person-group>. (<year>2009</year>). <article-title>Real-time DNA sequencing from single polymerase molecules</article-title>. <source>Science</source> <volume>323</volume>, <fpage>133</fpage>&#x02013;<lpage>138</lpage>. <pub-id pub-id-type="doi">10.1126/science.1162986</pub-id><pub-id pub-id-type="pmid">19023044</pub-id></citation></ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ellegaard</surname> <given-names>K. M.</given-names></name> <name><surname>Engel</surname> <given-names>P.</given-names></name></person-group> (<year>2016</year>). <article-title>Beyond 16S rRNA community profiling: intra-species diversity in the gut microbiota</article-title>. <source>Front. Microbiol</source>. <volume>7</volume>:<fpage>1475</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2016.01475</pub-id><pub-id pub-id-type="pmid">27708630</pub-id></citation></ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ercolini</surname> <given-names>D.</given-names></name></person-group> (<year>2013</year>). <article-title>High-throughput sequencing and metagenomics: moving forward in the culture-independent analysis of food microbial ecology</article-title>. <source>Appl. Environ. Microbiol</source>. <volume>79</volume>, <fpage>3148</fpage>&#x02013;<lpage>3155</lpage>. <pub-id pub-id-type="doi">10.1128/AEM.00256-13</pub-id><pub-id pub-id-type="pmid">23475615</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Erkus</surname> <given-names>O.</given-names></name> <name><surname>de Jager</surname> <given-names>V. C. L.</given-names></name> <name><surname>Spus</surname> <given-names>M.</given-names></name> <name><surname>van Alen-Boerrigter</surname> <given-names>I. J.</given-names></name> <name><surname>van Rijswijck</surname> <given-names>I. M. H.</given-names></name> <name><surname>Hazelwood</surname> <given-names>L.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>Multifactorial diversity sustains microbial community stability</article-title>. <source>ISME J</source>. <volume>7</volume>, <fpage>2126</fpage>&#x02013;<lpage>2136</lpage>. <pub-id pub-id-type="doi">10.1038/ismej.2013.108</pub-id><pub-id pub-id-type="pmid">23823494</pub-id></citation></ref>
<ref id="B29">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Eugster-Meier</surname> <given-names>E.</given-names></name> <name><surname>Fr&#x000F6;hlich-Wyder</surname> <given-names>M. T.</given-names></name> <name><surname>Jakob</surname> <given-names>E.</given-names></name> <name><surname>Wechsler</surname> <given-names>D.</given-names></name></person-group> (<year>2017</year>). <article-title>Le Gruy&#x000E8;re PDO-Switzerland</article-title>, in <source>Global Cheesemaking Technology: Cheese Quality and Characteristics</source>, eds <person-group person-group-type="editor"><name><surname>Papademas</surname> <given-names>P.</given-names></name> <name><surname>Bintsis</surname> <given-names>T.</given-names></name></person-group> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>John Wiley &#x00026; Sons</publisher-name>), <fpage>228</fpage>&#x02013;<lpage>230</lpage>.</citation></ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fernandez</surname> <given-names>A.</given-names></name> <name><surname>Ogawa</surname> <given-names>J.</given-names></name> <name><surname>Penaud</surname> <given-names>S.</given-names></name> <name><surname>Boudebbouze</surname> <given-names>S.</given-names></name> <name><surname>Ehrlich</surname> <given-names>D.</given-names></name> <name><surname>van de Guchte</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2008</year>). <article-title>Rerouting of pyruvate metabolism during acid adaptation in <italic>Lactobacillus bulgaricus</italic></article-title>. <source>Proteomics</source> <volume>8</volume>, <fpage>3154</fpage>&#x02013;<lpage>3163</lpage>. <pub-id pub-id-type="doi">10.1002/pmic.200700974</pub-id><pub-id pub-id-type="pmid">18615427</pub-id></citation></ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Finn</surname> <given-names>R. D.</given-names></name> <name><surname>Attwood</surname> <given-names>T. K.</given-names></name> <name><surname>Babbitt</surname> <given-names>P. C.</given-names></name> <name><surname>Bateman</surname> <given-names>A.</given-names></name> <name><surname>Bork</surname> <given-names>P.</given-names></name> <name><surname>Bridge</surname> <given-names>A. J.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>InterPro in 2017-beyond protein family and domain annotations</article-title>. <source>Nucleic Acids Res</source>. <volume>45</volume>, <fpage>D190</fpage>&#x02013;<lpage>D199</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkw1107</pub-id><pub-id pub-id-type="pmid">27899635</pub-id></citation></ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Finn</surname> <given-names>R. D.</given-names></name> <name><surname>Coggill</surname> <given-names>P.</given-names></name> <name><surname>Eberhardt</surname> <given-names>R. Y.</given-names></name> <name><surname>Eddy</surname> <given-names>S. R.</given-names></name> <name><surname>Mistry</surname> <given-names>J.</given-names></name> <name><surname>Mitchell</surname> <given-names>A. L.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>The Pfam protein families database: towards a more sustainable future</article-title>. <source>Nucleic Acids Res</source>. <volume>44</volume>, <fpage>D279</fpage>&#x02013;<lpage>285</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkv1344</pub-id><pub-id pub-id-type="pmid">26673716</pub-id></citation></ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Geissler</surname> <given-names>A. J.</given-names></name> <name><surname>Behr</surname> <given-names>J.</given-names></name> <name><surname>Vogel</surname> <given-names>R. F.</given-names></name></person-group> (<year>2016</year>). <article-title>Multiple genome sequences of the important beer-spoiling species <italic>Lactobacillus backii</italic></article-title>. <source>Genome Announc</source>. <volume>4</volume>:<fpage>e00826</fpage>-<lpage>16</lpage>. <pub-id pub-id-type="doi">10.1128/genomeA.00826-16</pub-id><pub-id pub-id-type="pmid">27563041</pub-id></citation></ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Giraffa</surname> <given-names>G.</given-names></name> <name><surname>Chanishvili</surname> <given-names>N.</given-names></name> <name><surname>Widyastuti</surname> <given-names>Y.</given-names></name></person-group> (<year>2010</year>). <article-title>Importance of lactobacilli in food and feed biotechnology</article-title>. <source>Res. Microbiol</source>. <volume>161</volume>, <fpage>480</fpage>&#x02013;<lpage>487</lpage>. <pub-id pub-id-type="doi">10.1016/j.resmic.2010.03.001</pub-id><pub-id pub-id-type="pmid">20302928</pub-id></citation></ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Grissa</surname> <given-names>I.</given-names></name> <name><surname>Vergnaud</surname> <given-names>G.</given-names></name> <name><surname>Pourcel</surname> <given-names>C.</given-names></name></person-group> (<year>2007</year>). <article-title>CRISPRFinder: a web tool to identify clustered regularly interspaced short palindromic repeats</article-title>. <source>Nucleic Acids Res</source>. <volume>35</volume>, <fpage>W52</fpage>&#x02013;<lpage>W57</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkm360</pub-id><pub-id pub-id-type="pmid">17537822</pub-id></citation></ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hannon</surname> <given-names>J. A.</given-names></name> <name><surname>Kilcawley</surname> <given-names>K. N.</given-names></name> <name><surname>Wilkinson</surname> <given-names>M. G.</given-names></name> <name><surname>Delahunty</surname> <given-names>C. M.</given-names></name> <name><surname>Beresford</surname> <given-names>T. P.</given-names></name></person-group> (<year>2007</year>). <article-title>Flavour precursor development in Cheddar cheese due to lactococcal starters and the presence and lysis of <italic>Lactobacillus helveticus</italic></article-title>. <source>Int. Dairy J</source>. <volume>17</volume>, <fpage>316</fpage>&#x02013;<lpage>327</lpage>. <pub-id pub-id-type="doi">10.1016/j.idairyj.2006.03.001</pub-id></citation></ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hannon</surname> <given-names>J. A.</given-names></name> <name><surname>Wilkinson</surname> <given-names>M. G.</given-names></name> <name><surname>Delahunty</surname> <given-names>C. M.</given-names></name> <name><surname>Wallace</surname> <given-names>J. M.</given-names></name> <name><surname>Morrissey</surname> <given-names>P. A.</given-names></name> <name><surname>Beresford</surname> <given-names>T. P.</given-names></name></person-group> (<year>2003</year>). <article-title>Use of autolytic starter systems to accelerate the ripening of Cheddar cheese</article-title>. <source>Int. Dairy J</source>. <volume>13</volume>, <fpage>313</fpage>&#x02013;<lpage>323</lpage>. <pub-id pub-id-type="doi">10.1016/S0958-6946(02)00178-4</pub-id></citation></ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hornischer</surname> <given-names>K.</given-names></name> <name><surname>H&#x000E4;u&#x000DF;ler</surname> <given-names>S.</given-names></name></person-group> (<year>2016</year>). <article-title>Diagnostics and resistance profiling of bacterial pathogens</article-title>. <source>Curr. Top. Microbiol. Immunol</source>. <volume>398</volume>, <fpage>89</fpage>&#x02013;<lpage>102</lpage>. <pub-id pub-id-type="doi">10.1007/82_2016_494</pub-id><pub-id pub-id-type="pmid">27474081</pub-id></citation></ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huber</surname> <given-names>W.</given-names></name> <name><surname>Carey</surname> <given-names>V. J.</given-names></name> <name><surname>Gentleman</surname> <given-names>R.</given-names></name> <name><surname>Anders</surname> <given-names>S.</given-names></name> <name><surname>Carlson</surname> <given-names>M.</given-names></name> <name><surname>Carvalho</surname> <given-names>B. S.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Orchestrating high-throughput genomic analysis with Bioconductor</article-title>. <source>Nat. Methods</source> <volume>12</volume>, <fpage>115</fpage>&#x02013;<lpage>121</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.3252</pub-id><pub-id pub-id-type="pmid">25633503</pub-id></citation></ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huerta-Cepas</surname> <given-names>J.</given-names></name> <name><surname>Szklarczyk</surname> <given-names>D.</given-names></name> <name><surname>Forslund</surname> <given-names>K.</given-names></name> <name><surname>Cook</surname> <given-names>H.</given-names></name> <name><surname>Heller</surname> <given-names>D.</given-names></name> <name><surname>Walter</surname> <given-names>M. C.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>eggNOG 4.5: a hierarchical orthology framework with improved functional annotations for eukaryotic, prokaryotic and viral sequences</article-title>. <source>Nucleic Acids Res</source>. <volume>44</volume>, <fpage>D286</fpage>&#x02013;<lpage>D293</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkv1248</pub-id><pub-id pub-id-type="pmid">26582926</pub-id></citation></ref>
<ref id="B41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hunt</surname> <given-names>M.</given-names></name> <name><surname>De Silva</surname> <given-names>N.</given-names></name> <name><surname>Otto</surname> <given-names>T. D.</given-names></name> <name><surname>Parkhill</surname> <given-names>J.</given-names></name> <name><surname>Keane</surname> <given-names>J. A.</given-names></name> <name><surname>Harris</surname> <given-names>S. R.</given-names></name></person-group> (<year>2015</year>). <article-title>Circlator: automated circularization of genome assemblies using long sequencing reads</article-title>. <source>Genome Biol</source>. <volume>16</volume>, <fpage>294</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-015-0849-0</pub-id><pub-id pub-id-type="pmid">26714481</pub-id></citation></ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hyn&#x000F6;nen</surname> <given-names>U.</given-names></name> <name><surname>Palva</surname> <given-names>A.</given-names></name></person-group> (<year>2013</year>). <article-title>Lactobacillus surface layer proteins: structure, function and applications</article-title>. <source>Appl. Microbiol. Biotechnol</source>. <volume>97</volume>, <fpage>5225</fpage>&#x02013;<lpage>5243</lpage>. <pub-id pub-id-type="doi">10.1007/s00253-013-4962-2</pub-id><pub-id pub-id-type="pmid">23677442</pub-id></citation></ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jebava</surname> <given-names>I.</given-names></name> <name><surname>Plockova</surname> <given-names>M.</given-names></name> <name><surname>Lortal</surname> <given-names>S.</given-names></name> <name><surname>Valence</surname> <given-names>F.</given-names></name></person-group> (<year>2011</year>). <article-title>The nine peptidoglycan hydrolases genes in <italic>Lactobacillus helveticus</italic> are ubiquitous and early transcribed</article-title>. <source>Int. J. Food Microbiol</source>. <volume>148</volume>, <fpage>1</fpage>&#x02013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1016/j.ijfoodmicro.2011.04.015</pub-id><pub-id pub-id-type="pmid">21571387</pub-id></citation></ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Koren</surname> <given-names>S.</given-names></name> <name><surname>Harhay</surname> <given-names>G. P.</given-names></name> <name><surname>Smith</surname> <given-names>T. P. L.</given-names></name> <name><surname>Bono</surname> <given-names>J. L.</given-names></name> <name><surname>Harhay</surname> <given-names>D. M.</given-names></name> <name><surname>Mcvey</surname> <given-names>S. D.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>Reducing assembly complexity of microbial genomes with single-molecule sequencing</article-title>. <source>Genome Biol</source>. <volume>14</volume>:<fpage>R101</fpage>. <pub-id pub-id-type="doi">10.1186/gb-2013-14-9-r101</pub-id><pub-id pub-id-type="pmid">24034426</pub-id></citation></ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Koren</surname> <given-names>S.</given-names></name> <name><surname>Schatz</surname> <given-names>M. C.</given-names></name> <name><surname>Walenz</surname> <given-names>B. P.</given-names></name> <name><surname>Martin</surname> <given-names>J.</given-names></name> <name><surname>Howard</surname> <given-names>J. T.</given-names></name> <name><surname>Ganapathy</surname> <given-names>G.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>Hybrid error correction and <italic>de novo</italic> assembly of single-molecule sequencing reads</article-title>. <source>Nat. Biotechnol</source>. <volume>30</volume>, <fpage>693</fpage>&#x02013;<lpage>700</lpage>. <pub-id pub-id-type="doi">10.1038/nbt.2280</pub-id><pub-id pub-id-type="pmid">22750884</pub-id></citation></ref>
<ref id="B46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krzywinski</surname> <given-names>M.</given-names></name> <name><surname>Schein</surname> <given-names>J.</given-names></name> <name><surname>Birol</surname> <given-names>I.</given-names></name> <name><surname>Connors</surname> <given-names>J.</given-names></name> <name><surname>Gascoyne</surname> <given-names>R.</given-names></name> <name><surname>Horsman</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2009</year>). <article-title>Circos: an information aesthetic for comparative genomics</article-title>. <source>Genome Res</source>. <volume>19</volume>, <fpage>1639</fpage>&#x02013;<lpage>1645</lpage>. <pub-id pub-id-type="doi">10.1101/gr.092759.109</pub-id><pub-id pub-id-type="pmid">19541911</pub-id></citation></ref>
<ref id="B47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kurtz</surname> <given-names>S.</given-names></name> <name><surname>Phillippy</surname> <given-names>A.</given-names></name> <name><surname>Delcher</surname> <given-names>A. L.</given-names></name> <name><surname>Smoot</surname> <given-names>M.</given-names></name> <name><surname>Shumway</surname> <given-names>M.</given-names></name> <name><surname>Antonescu</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2004</year>). <article-title>Versatile and open software for comparing large genomes</article-title>. <source>Genome Biol</source>. <volume>5</volume>:<fpage>R12</fpage>. <pub-id pub-id-type="doi">10.1186/gb-2004-5-2-r12</pub-id><pub-id pub-id-type="pmid">14759262</pub-id></citation></ref>
<ref id="B48">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Laehnemann</surname> <given-names>D.</given-names></name> <name><surname>Borkhardt</surname> <given-names>A.</given-names></name> <name><surname>McHardy</surname> <given-names>A. C.</given-names></name></person-group> (<year>2015</year>). <article-title>Denoising DNA deep sequencing data -high-throughput sequencing errors and their correction</article-title>. <source>Brief. Bioinform</source>. <volume>17</volume>, <fpage>154</fpage>&#x02013;<lpage>179</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbv029</pub-id><pub-id pub-id-type="pmid">26026159</pub-id></citation></ref>
<ref id="B49">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Leroy</surname> <given-names>F.</given-names></name> <name><surname>De Vuyst</surname> <given-names>L.</given-names></name></person-group> (<year>2004</year>). <article-title>Lactic acid bacteria as functional starter cultures for the food fermentation industry</article-title>. <source>Trends Food Sci. Technol</source>. <volume>15</volume>, <fpage>67</fpage>&#x02013;<lpage>78</lpage>. <pub-id pub-id-type="doi">10.1016/j.tifs.2003.09.004</pub-id></citation></ref>
<ref id="B50">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>H.</given-names></name></person-group> (<year>2013</year>). <source>Aligning Sequence Reads, Clone Sequences and Assembly Contigs with BWA-MEM</source>. arXiv [q-bio.GN]. Available online at: <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/1303.3997">http://arxiv.org/abs/1303.3997</ext-link></citation></ref>
<ref id="B51">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lukjancenko</surname> <given-names>O.</given-names></name> <name><surname>Ussery</surname> <given-names>D. W.</given-names></name> <name><surname>Wassenaar</surname> <given-names>T. M.</given-names></name></person-group> (<year>2012</year>). <article-title>Comparative genomics of Bifidobacterium, Lactobacillus and related probiotic genera</article-title>. <source>Microb. Ecol</source>. <volume>63</volume>, <fpage>651</fpage>&#x02013;<lpage>673</lpage>. <pub-id pub-id-type="doi">10.1007/s00248-011-9948-y</pub-id><pub-id pub-id-type="pmid">22031452</pub-id></citation></ref>
<ref id="B52">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Makarova</surname> <given-names>K.</given-names></name> <name><surname>Slesarev</surname> <given-names>A.</given-names></name> <name><surname>Wolf</surname> <given-names>Y.</given-names></name> <name><surname>Sorokin</surname> <given-names>A.</given-names></name> <name><surname>Mirkin</surname> <given-names>B.</given-names></name> <name><surname>Koonin</surname> <given-names>E.</given-names></name> <etal/></person-group>. (<year>2006</year>). <article-title>Comparative genomics of the lactic acid bacteria</article-title>. <source>Proc. Natl. Acad. Sci. U.S.A</source>. <volume>103</volume>, <fpage>15611</fpage>&#x02013;<lpage>15616</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.0607117103</pub-id><pub-id pub-id-type="pmid">17030793</pub-id></citation></ref>
<ref id="B53">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mavromatis</surname> <given-names>K.</given-names></name> <name><surname>Land</surname> <given-names>M. L.</given-names></name> <name><surname>Brettin</surname> <given-names>T. S.</given-names></name> <name><surname>Quest</surname> <given-names>D. J.</given-names></name> <name><surname>Copeland</surname> <given-names>A.</given-names></name> <name><surname>Clum</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>The fast changing landscape of sequencing technologies and their impact on microbial genome assemblies and annotation</article-title>. <source>PLoS ONE</source> <volume>7</volume>:<fpage>e48837</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0048837</pub-id><pub-id pub-id-type="pmid">23251337</pub-id></citation></ref>
<ref id="B54">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mayr</surname> <given-names>E.</given-names></name> <name><surname>Bock</surname> <given-names>W. J.</given-names></name></person-group> (<year>2002</year>). <article-title>Classifications and other ordering systems</article-title>. <source>J. Zoolog. Syst. Evol. Res</source>. <volume>40</volume>, <fpage>169</fpage>&#x02013;<lpage>194</lpage>. <pub-id pub-id-type="doi">10.1046/j.1439-0469.2002.00211.x</pub-id></citation></ref>
<ref id="B55">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>McSweeney</surname> <given-names>P. L. H.</given-names></name></person-group> (<year>2011</year>). <article-title>Biochemistry of cheese ripening</article-title>, in <source>Encyclopedia Dairy Science, 2nd Edn.</source>, eds <person-group person-group-type="editor"><name><surname>Fuquay</surname> <given-names>J. W.</given-names></name> <name><surname>Fox</surname> <given-names>P. F.</given-names></name> <name><surname>McSweeney</surname> <given-names>P. L. H.</given-names></name></person-group> (<publisher-loc>San Diego, CA</publisher-loc>: <publisher-name>Academic Press</publisher-name>), <fpage>667</fpage>&#x02013;<lpage>674</lpage>.</citation></ref>
<ref id="B56">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Medini</surname> <given-names>D.</given-names></name> <name><surname>Donati</surname> <given-names>C.</given-names></name> <name><surname>Tettelin</surname> <given-names>H.</given-names></name> <name><surname>Masignani</surname> <given-names>V.</given-names></name> <name><surname>Rappuoli</surname> <given-names>R.</given-names></name></person-group> (<year>2005</year>). <article-title>The microbial pan-genome</article-title>. <source>Curr. Opin. Genet. Dev</source>. <volume>15</volume>, <fpage>589</fpage>&#x02013;<lpage>594</lpage>. <pub-id pub-id-type="doi">10.1016/j.gde.2005.09.006</pub-id><pub-id pub-id-type="pmid">16185861</pub-id></citation></ref>
<ref id="B57">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Moser</surname> <given-names>A.</given-names></name> <name><surname>Berthoud</surname> <given-names>H.</given-names></name> <name><surname>Eugster</surname> <given-names>E.</given-names></name> <name><surname>Meile</surname> <given-names>L.</given-names></name> <name><surname>Irmler</surname> <given-names>S.</given-names></name></person-group> (<year>2017</year>). <article-title>Detection and enumeration of <italic>Lactobacillus helveticus</italic> in dairy products</article-title>. <source>Int. Dairy J</source>. <volume>68</volume>, <fpage>52</fpage>&#x02013;<lpage>59</lpage>. <pub-id pub-id-type="doi">10.1016/j.idairyj.2016.12.007</pub-id></citation></ref>
<ref id="B58">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>O&#x00027;Leary</surname> <given-names>N. A.</given-names></name> <name><surname>Wright</surname> <given-names>M. W.</given-names></name> <name><surname>Brister</surname> <given-names>J. R.</given-names></name> <name><surname>Ciufo</surname> <given-names>S.</given-names></name> <name><surname>Haddad</surname> <given-names>D.</given-names></name> <name><surname>McVeigh</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>Reference sequence (RefSeq) database at NCBI: current status, taxonomic expansion, and functional annotation</article-title>. <source>Nucleic Acids Res</source>. <volume>44</volume>, <fpage>D733</fpage>&#x02013;<lpage>D745</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkv1189</pub-id><pub-id pub-id-type="pmid">26553804</pub-id></citation></ref>
<ref id="B59">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Omasits</surname> <given-names>U.</given-names></name> <name><surname>Quebatte</surname> <given-names>M.</given-names></name> <name><surname>Stekhoven</surname> <given-names>D. J.</given-names></name> <name><surname>Fortes</surname> <given-names>C.</given-names></name> <name><surname>Roschitzki</surname> <given-names>B.</given-names></name> <name><surname>Robinson</surname> <given-names>M. D.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>Directed shotgun proteomics guided by saturated RNA-seq identifies a complete expressed prokaryotic proteome</article-title>. <source>Genome Res</source>. <volume>23</volume>, <fpage>1916</fpage>&#x02013;<lpage>1927</lpage>. <pub-id pub-id-type="doi">10.1101/gr.151035.112</pub-id><pub-id pub-id-type="pmid">23878158</pub-id></citation></ref>
<ref id="B60">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Omasits</surname> <given-names>U.</given-names></name> <name><surname>Varadarajan</surname> <given-names>A. R.</given-names></name> <name><surname>Schmid</surname> <given-names>M.</given-names></name> <name><surname>Goetze</surname> <given-names>S.</given-names></name> <name><surname>Melidis</surname> <given-names>D.</given-names></name> <name><surname>Bourqui</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>An integrative strategy to identify the entire protein coding potential of prokaryotic genomes by proteogenomics</article-title>. <source>Genome Res</source>. <volume>27</volume>, <fpage>2083</fpage>&#x02013;<lpage>2095</lpage>. <pub-id pub-id-type="doi">10.1101/gr.218255.116</pub-id><pub-id pub-id-type="pmid">29141959</pub-id></citation></ref>
<ref id="B61">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Page</surname> <given-names>A. J.</given-names></name> <name><surname>Cummins</surname> <given-names>C. A.</given-names></name> <name><surname>Hunt</surname> <given-names>M.</given-names></name> <name><surname>Wong</surname> <given-names>V. K.</given-names></name> <name><surname>Reuter</surname> <given-names>S.</given-names></name> <name><surname>Holden</surname> <given-names>M. T. G.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Roary: rapid large-scale prokaryote pan genome analysis</article-title>. <source>Bioinformatics</source> <volume>31</volume>, <fpage>3691</fpage>&#x02013;<lpage>3693</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btv421</pub-id><pub-id pub-id-type="pmid">26198102</pub-id></citation></ref>
<ref id="B62">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Prajapati</surname> <given-names>J. B.</given-names></name> <name><surname>Khedkar</surname> <given-names>C. D.</given-names></name> <name><surname>Chitra</surname> <given-names>J.</given-names></name> <name><surname>Suja</surname> <given-names>S.</given-names></name> <name><surname>Mishra</surname> <given-names>V.</given-names></name> <name><surname>Sreeja</surname> <given-names>V.</given-names></name> <etal/></person-group>. (<year>2011</year>). <article-title>Whole-genome shotgun sequencing of an Indian-origin <italic>Lactobacillus helveticus</italic> strain, MTCC 5463, with probiotic potential</article-title>. <source>J. Bacteriol</source>. <volume>193</volume>, <fpage>4282</fpage>&#x02013;<lpage>4283</lpage>. <pub-id pub-id-type="doi">10.1128/JB.05449-11</pub-id><pub-id pub-id-type="pmid">21705605</pub-id></citation></ref>
<ref id="B63">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Reddy</surname> <given-names>T. B. K.</given-names></name> <name><surname>Thomas</surname> <given-names>A. D.</given-names></name> <name><surname>Stamatis</surname> <given-names>D.</given-names></name> <name><surname>Bertsch</surname> <given-names>J.</given-names></name> <name><surname>Isbandi</surname> <given-names>M.</given-names></name> <name><surname>Jansson</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>The Genomes OnLine Database (GOLD) v.5: a metadata management system based on a four level (meta)genome project classification</article-title>. <source>Nucleic Acids Res</source>. <volume>43</volume>, <fpage>D1099</fpage>&#x02013;<lpage>D1106</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gku950</pub-id><pub-id pub-id-type="pmid">25348402</pub-id></citation></ref>
<ref id="B64">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Remus-Emsermann</surname> <given-names>M. N. P.</given-names></name> <name><surname>Schmid</surname> <given-names>M.</given-names></name> <name><surname>Gekenidis</surname> <given-names>M.-T.</given-names></name> <name><surname>Pelludat</surname> <given-names>C.</given-names></name> <name><surname>Frey</surname> <given-names>J. E.</given-names></name> <name><surname>Ahrens</surname> <given-names>C. H.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>Complete genome sequence of Pseudomonas citronellolis P3B5, a candidate for microbial phyllo-remediation of hydrocarbon-contaminated sites</article-title>. <source>Stand. Genomic Sci</source>. <volume>11</volume>, <fpage>75</fpage>. <pub-id pub-id-type="doi">10.1186/s40793-016-0190-6</pub-id><pub-id pub-id-type="pmid">28300228</pub-id></citation></ref>
<ref id="B65">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rice</surname> <given-names>P.</given-names></name> <name><surname>Longden</surname> <given-names>I.</given-names></name> <name><surname>Bleasby</surname> <given-names>A.</given-names></name></person-group> (<year>2000</year>). <article-title>EMBOSS: the European Molecular Biology Open Software Suite</article-title>. <source>Trends Genet</source>. <volume>16</volume>, <fpage>276</fpage>&#x02013;<lpage>277</lpage>. <pub-id pub-id-type="doi">10.1016/S0168-9525(00)02024-2</pub-id><pub-id pub-id-type="pmid">10827456</pub-id></citation></ref>
<ref id="B66">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ricker</surname> <given-names>N.</given-names></name> <name><surname>Qian</surname> <given-names>H.</given-names></name> <name><surname>Fulthorpe</surname> <given-names>R. R.</given-names></name></person-group> (<year>2012</year>). <article-title>The limitations of draft assemblies for understanding prokaryotic adaptation and evolution</article-title>. <source>Genomics</source> <volume>100</volume>, <fpage>167</fpage>&#x02013;<lpage>175</lpage>. <pub-id pub-id-type="doi">10.1016/j.ygeno.2012.06.009</pub-id><pub-id pub-id-type="pmid">22750556</pub-id></citation></ref>
<ref id="B67">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rossetti</surname> <given-names>L.</given-names></name> <name><surname>Fornasari</surname> <given-names>M. E.</given-names></name> <name><surname>Gatti</surname> <given-names>M.</given-names></name> <name><surname>Lazzi</surname> <given-names>C.</given-names></name> <name><surname>Neviani</surname> <given-names>E.</given-names></name> <name><surname>Giraffa</surname> <given-names>G.</given-names></name></person-group> (<year>2008</year>). <article-title>Grana Padano cheese whey starters: microbial composition and strain distribution</article-title>. <source>Int. J. Food Microbiol</source>. <volume>127</volume>, <fpage>168</fpage>&#x02013;<lpage>171</lpage>. <pub-id pub-id-type="doi">10.1016/j.ijfoodmicro.2008.06.005</pub-id><pub-id pub-id-type="pmid">18620769</pub-id></citation></ref>
<ref id="B68">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Salvetti</surname> <given-names>E.</given-names></name> <name><surname>Torriani</surname> <given-names>S.</given-names></name> <name><surname>Felis</surname> <given-names>G. E.</given-names></name></person-group> (<year>2012</year>). <article-title>The genus Lactobacillus: a taxonomic update</article-title>. <source>Probiotics Antimicrob. Proteins</source> <volume>4</volume>, <fpage>217</fpage>&#x02013;<lpage>226</lpage>. <pub-id pub-id-type="doi">10.1007/s12602-012-9117-8</pub-id><pub-id pub-id-type="pmid">26782181</pub-id></citation></ref>
<ref id="B69">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Samson</surname> <given-names>J. E.</given-names></name> <name><surname>Moineau</surname> <given-names>S.</given-names></name></person-group> (<year>2013</year>). <article-title>Bacteriophages in food fermentations: new frontiers in a continuous arms race</article-title>. <source>Annu. Rev. Food Sci. Technol</source>. <volume>4</volume>, <fpage>347</fpage>&#x02013;<lpage>368</lpage>. <pub-id pub-id-type="doi">10.1146/annurev-food-030212-182541</pub-id><pub-id pub-id-type="pmid">23244395</pub-id></citation></ref>
<ref id="B70">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sangwan</surname> <given-names>N.</given-names></name> <name><surname>Xia</surname> <given-names>F.</given-names></name> <name><surname>Gilbert</surname> <given-names>J. A.</given-names></name></person-group> (<year>2016</year>). <article-title>Recovering complete and draft population genomes from metagenome datasets</article-title>. <source>Microbiome</source> <volume>4</volume>, <fpage>8</fpage>. <pub-id pub-id-type="doi">10.1186/s40168-016-0154-5</pub-id><pub-id pub-id-type="pmid">26951112</pub-id></citation></ref>
<ref id="B71">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Selle</surname> <given-names>K.</given-names></name> <name><surname>Barrangou</surname> <given-names>R.</given-names></name></person-group> (<year>2015</year>). <article-title>CRISPR-based technologies and the future of food science</article-title>. <source>J. Food Sci</source>. <volume>80</volume>, <fpage>R2367</fpage>&#x02013;<lpage>R2372</lpage>. <pub-id pub-id-type="doi">10.1111/1750-3841.13094</pub-id><pub-id pub-id-type="pmid">26444151</pub-id></citation></ref>
<ref id="B72">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Slattery</surname> <given-names>L.</given-names></name> <name><surname>O&#x00027;Callaghan</surname> <given-names>J.</given-names></name> <name><surname>Fitzgerald</surname> <given-names>G. F.</given-names></name> <name><surname>Beresford</surname> <given-names>T.</given-names></name> <name><surname>Ross</surname> <given-names>R. P.</given-names></name></person-group> (<year>2010</year>). <article-title>Invited review: <italic>Lactobacillus helveticus</italic>- a thermophilic dairy starter related to gut bacteria</article-title>. <source>J. Dairy Sci</source>. <volume>93</volume>, <fpage>4435</fpage>&#x02013;<lpage>4454</lpage>. <pub-id pub-id-type="doi">10.3168/jds.2010-3327</pub-id><pub-id pub-id-type="pmid">20854978</pub-id></citation></ref>
<ref id="B73">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Smid</surname> <given-names>E. J.</given-names></name> <name><surname>Erkus</surname> <given-names>O.</given-names></name> <name><surname>Spus</surname> <given-names>M.</given-names></name> <name><surname>Wolkers-Rooijackers</surname> <given-names>J. C. M.</given-names></name> <name><surname>Alexeeva</surname> <given-names>S.</given-names></name> <name><surname>Kleerebezem</surname> <given-names>M.</given-names></name></person-group> (<year>2014</year>). <article-title>Functional implications of the microbial community structure of undefined mesophilic starter cultures</article-title>. <source>Microb. Cell Fact</source>. <volume>13</volume>(<supplement>Suppl 1.</supplement>):<fpage>S2</fpage>. <pub-id pub-id-type="doi">10.1186/1475-2859-13-S1-S2</pub-id><pub-id pub-id-type="pmid">25185941</pub-id></citation></ref>
<ref id="B74">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Stamatakis</surname> <given-names>A.</given-names></name></person-group> (<year>2014</year>). <article-title>RAxML version 8: a tool for phylogenetic analysis and post-analysis of large phylogenies</article-title>. <source>Bioinformatics</source> <volume>30</volume>, <fpage>1312</fpage>&#x02013;<lpage>1313</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btu033</pub-id><pub-id pub-id-type="pmid">24451623</pub-id></citation></ref>
<ref id="B75">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Stefanovic</surname> <given-names>E.</given-names></name> <name><surname>Fitzgerald</surname> <given-names>G.</given-names></name> <name><surname>McAuliffe</surname> <given-names>O.</given-names></name></person-group> (<year>2017</year>). <article-title>Advances in the genomics and metabolomics of dairy lactobacilli: a review</article-title>. <source>Food Microbiol</source>. <volume>61</volume>, <fpage>33</fpage>&#x02013;<lpage>49</lpage>. <pub-id pub-id-type="doi">10.1016/j.fm.2016.08.009</pub-id><pub-id pub-id-type="pmid">27697167</pub-id></citation></ref>
<ref id="B76">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Stekhoven</surname> <given-names>D. J.</given-names></name> <name><surname>Omasits</surname> <given-names>U.</given-names></name> <name><surname>Quebatte</surname> <given-names>M.</given-names></name> <name><surname>Dehio</surname> <given-names>C.</given-names></name> <name><surname>Ahrens</surname> <given-names>C. H.</given-names></name></person-group> (<year>2014</year>). <article-title>Proteome-wide identification of predominant subcellular protein localizations in a bacterial model organism</article-title>. <source>J. Proteomics</source> <volume>99</volume>, <fpage>123</fpage>&#x02013;<lpage>137</lpage>. <pub-id pub-id-type="doi">10.1016/j.jprot.2014.01.015</pub-id><pub-id pub-id-type="pmid">24486812</pub-id></citation></ref>
<ref id="B77">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Stothard</surname> <given-names>P.</given-names></name> <name><surname>Wishart</surname> <given-names>D. S.</given-names></name></person-group> (<year>2005</year>). <article-title>Circular genome visualization and exploration using CGView</article-title>. <source>Bioinformatics</source> <volume>21</volume>, <fpage>537</fpage>&#x02013;<lpage>539</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bti054</pub-id><pub-id pub-id-type="pmid">15479716</pub-id></citation></ref>
<ref id="B78">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>Z.</given-names></name> <name><surname>Harris</surname> <given-names>H. M. B.</given-names></name> <name><surname>McCann</surname> <given-names>A.</given-names></name> <name><surname>Guo</surname> <given-names>C.</given-names></name> <name><surname>Argim&#x000F3;n</surname> <given-names>S.</given-names></name> <name><surname>Zhang</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Expanding the biotechnology potential of lactobacilli through comparative genomics of 213 strains and associated genera</article-title>. <source>Nat. Commun</source>. <volume>6</volume>, <fpage>8322</fpage>. <pub-id pub-id-type="doi">10.1038/ncomms9322</pub-id><pub-id pub-id-type="pmid">26415554</pub-id></citation></ref>
<ref id="B79">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tatusova</surname> <given-names>T.</given-names></name> <name><surname>DiCuccio</surname> <given-names>M.</given-names></name> <name><surname>Badretdin</surname> <given-names>A.</given-names></name> <name><surname>Chetvernin</surname> <given-names>V.</given-names></name> <name><surname>Nawrocki</surname> <given-names>E. P.</given-names></name> <name><surname>Zaslavsky</surname> <given-names>L.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>NCBI prokaryotic genome annotation pipeline</article-title>. <source>Nucleic Acids Res</source>. <volume>44</volume>, <fpage>6614</fpage>&#x02013;<lpage>6624</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkw569</pub-id><pub-id pub-id-type="pmid">27342282</pub-id></citation></ref>
<ref id="B80">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Taverniti</surname> <given-names>V.</given-names></name> <name><surname>Guglielmetti</surname> <given-names>S.</given-names></name></person-group> (<year>2012</year>). <article-title>Health-promoting properties of <italic>Lactobacillus helveticus</italic></article-title>. <source>Front. Microbiol</source>. <volume>3</volume>:<fpage>392</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2012.00392</pub-id><pub-id pub-id-type="pmid">23181058</pub-id></citation></ref>
<ref id="B81">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vinogradov</surname> <given-names>E.</given-names></name> <name><surname>Valence</surname> <given-names>F.</given-names></name> <name><surname>Maes</surname> <given-names>E.</given-names></name> <name><surname>Jebava</surname> <given-names>I.</given-names></name> <name><surname>Chuat</surname> <given-names>V.</given-names></name> <name><surname>Lortal</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>Structural studies of the cell wall polysaccharides from three strains of <italic>Lactobacillus helveticus</italic> with different autolytic properties: DPC4571, BROI, and LH1</article-title>. <source>Carbohydr. Res</source>. <volume>379</volume>, <fpage>7</fpage>&#x02013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1016/j.carres.2013.05.020</pub-id><pub-id pub-id-type="pmid">23831635</pub-id></citation></ref>
<ref id="B82">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>W.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Sun</surname> <given-names>Z.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Zhou</surname> <given-names>Z.</given-names></name> <name><surname>Sun</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2011</year>). <article-title>Complete genome sequence of <italic>Lactobacillus helveticus</italic> H10</article-title>. <source>J. Bacteriol</source>. <volume>193</volume>, <fpage>2666</fpage>&#x02013;<lpage>2667</lpage>. <pub-id pub-id-type="doi">10.1128/JB.00166-11</pub-id><pub-id pub-id-type="pmid">21398542</pub-id></citation></ref>
</ref-list>
</back>
</article>