<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Microbiol.</journal-id>
<journal-title>Frontiers in Microbiology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Microbiol.</abbrev-journal-title>
<issn pub-type="epub">1664-302X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmicb.2020.01527</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Microbiology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Comparative Genomics Discloses the Uniqueness and the Biosynthetic Potential of the Marine Cyanobacterium <italic>Hyella patelloides</italic></article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Brito</surname> <given-names>&#x00C2;ngela</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1003639/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Vieira</surname> <given-names>Jorge</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Vieira</surname> <given-names>Cristina P.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/719234/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhu</surname> <given-names>Tao</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Le&#x00E3;o</surname> <given-names>Pedro N.</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/218823/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Ramos</surname> <given-names>Vitor</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/90217/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Lu</surname> <given-names>Xuefeng</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/87231/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Vasconcelos</surname> <given-names>Vitor M.</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/162986/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Gugger</surname> <given-names>Muriel</given-names></name>
<xref ref-type="aff" rid="aff7"><sup>7</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/191217/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Tamagnini</surname> <given-names>Paula</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/91974/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>i3S &#x2013; Instituto de Investiga&#x00E7;&#x00E3;o e Inova&#x00E7;&#x00E3;o em Sa&#x00FA;de, Universidade do Porto</institution>, <addr-line>Porto</addr-line>, <country>Portugal</country></aff>
<aff id="aff2"><sup>2</sup><institution>IBMC &#x2013; Instituto de Biologia Molecular e Celular, Universidade do Porto</institution>, <addr-line>Porto</addr-line>, <country>Portugal</country></aff>
<aff id="aff3"><sup>3</sup><institution>Key Laboratory of Biofuels, Shandong Provincial Key Laboratory of Synthetic Biology, Qingdao Institute of Bioenergy and Bioprocess Technology, Chinese Academy of Sciences</institution>, <addr-line>Qingdao</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>Interdisciplinary Centre of Marine and Environmental Research (CIIMAR/CIMAR), University of Porto</institution>, <addr-line>Matosinhos</addr-line>, <country>Portugal</country></aff>
<aff id="aff5"><sup>5</sup><institution>Laboratory for Marine Biology and Biotechnology, Qingdao National Laboratory for Marine Science and Technology</institution>, <addr-line>Qingdao</addr-line>, <country>China</country></aff>
<aff id="aff6"><sup>6</sup><institution>Departamento de Biologia, Faculdade de Ci&#x00EA;ncias, Universidade do Porto</institution>, <addr-line>Porto</addr-line>, <country>Portugal</country></aff>
<aff id="aff7"><sup>7</sup><institution>Institut Pasteur, Collection des Cyanobact&#x00E9;ries</institution>, <addr-line>Paris</addr-line>, <country>France</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Frank T. Robb, University of Maryland, Baltimore, United States</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Yuu Hirose, Toyohashi University of Technology, Japan; Francisco (Paco) Barona-Gomez, Center for Research and Advanced Studies of the National Polytechnic Institute, Mexico</p></fn>
<corresp id="c001">&#x002A;Correspondence: Paula Tamagnini, <email>pmtamagn@ibmc.up.pt</email></corresp>
<fn fn-type="other" id="fn004"><p>This article was submitted to Evolutionary and Genomic Microbiology, a section of the journal Frontiers in Microbiology</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>07</day>
<month>07</month>
<year>2020</year>
</pub-date>
<pub-date pub-type="collection">
<year>2020</year>
</pub-date>
<volume>11</volume>
<elocation-id>1527</elocation-id>
<history>
<date date-type="received">
<day>09</day>
<month>01</month>
<year>2020</year>
</date>
<date date-type="accepted">
<day>12</day>
<month>06</month>
<year>2020</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2020 Brito, Vieira, Vieira, Zhu, Le&#x00E3;o, Ramos, Lu, Vasconcelos, Gugger and Tamagnini.</copyright-statement>
<copyright-year>2020</copyright-year>
<copyright-holder>Brito, Vieira, Vieira, Zhu, Le&#x00E3;o, Ramos, Lu, Vasconcelos, Gugger and Tamagnini</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Baeocytous cyanobacteria (Pleurocapsales/Subsection II) can thrive in a wide range of habitats on Earth but, compared to other cyanobacterial lineages, they remain poorly studied at genomic level. In this study, we sequenced the first genome from a member of the <italic>Hyella</italic> genus &#x2013; <italic>H. patelloides</italic> LEGE 07179, a recently described species isolated from the Portuguese foreshore. This genome is the largest of the thirteen baeocyte-forming cyanobacterial genomes sequenced so far, and diverges from the most closely related strains. Comparative analysis revealed strain-specific genes and horizontal gene transfer events between <italic>H. patelloides</italic> and its closest relatives. Moreover, <italic>H</italic>. <italic>patelloides</italic> genome is distinctive by the number and diversity of natural product biosynthetic gene clusters (BGCs). The majority of these clusters are strain-specific BGCs with a high probability of synthesizing novel natural products. One BGC was identified as being putatively involved in the production of terminal olefin. Our results showed that, <italic>H</italic>. <italic>patelloides</italic> produces hydrocarbon with C<sub>15</sub> chain length, and synthesizes C<sub>14</sub>, C<sub>16</sub>, and C<sub>18</sub> fatty acids exceeding 4% of the dry cell weight. Overall, our data contributed to increase the information on baeocytous cyanobacteria, and shed light on <italic>H. patelloides</italic> evolution, phylogeny and natural product biosynthetic potential.</p>
</abstract>
<kwd-group>
<kwd>biosynthetic gene clusters</kwd>
<kwd>cyanobacteria</kwd>
<kwd>genome</kwd>
<kwd><italic>Hyella</italic></kwd>
<kwd>natural products</kwd>
</kwd-group>
<contract-sponsor id="cn001">Funda&#x00E7;&#x00E3;o para a Ci&#x00EA;ncia e a Tecnologia<named-content content-type="fundref-id">10.13039/501100001871</named-content></contract-sponsor>
<counts>
<fig-count count="8"/>
<table-count count="2"/>
<equation-count count="0"/>
<ref-count count="48"/>
<page-count count="15"/>
<word-count count="0"/>
</counts>
</article-meta>
</front>
<body>
<sec id="S1">
<title>Introduction</title>
<p>Cyanobacteria are a monophyletic group of Gram negative bacteria with the ability to perform oxygenic photosynthesis and nowadays contribute up to 30% of the annual oxygen production on Earth (<xref ref-type="bibr" rid="B14">Deruyter and Fromme, 2008</xref>). They are found in a broad range of habitats (from fresh to salt water, soils and extreme environments) contributing significantly to the global primary production, mainly in nutrient-limited environments (<xref ref-type="bibr" rid="B19">Garcia-Pichel et al., 2003</xref>; <xref ref-type="bibr" rid="B18">Flombaum et al., 2013</xref>; <xref ref-type="bibr" rid="B15">D&#x00ED;ez et al., 2016</xref>). In addition, the diazotrophic cyanobacteria constitute the major source of biological nitrogen in the open ocean (<xref ref-type="bibr" rid="B47">Zehr, 2011</xref>). Cyanobacteria are also known to produce a wealth of natural products (NPs) with a wide spectrum of noteworthy biological activities such as anticancer, antibacterial, antiviral and antifungal (<xref ref-type="bibr" rid="B32">Nunnery et al., 2010</xref>; <xref ref-type="bibr" rid="B40">Sivonen et al., 2010</xref>). The importance of cyanobacteria in ecosystems&#x2019; equilibrium and the interest in their bioproducts are contributing to enlarge their genomic representation, still largely biased toward marine picocyanobacterial genera. The unicellular strains that divide by multiple fission producing small daughter cells, baeocytes (Subsection II/Pleurocapsales) (<xref ref-type="bibr" rid="B11">Castenholz, 2001</xref>), are clearly underrepresented at genomic level. Phylogenetic studies have shown that baeocyte-forming cyanobacteria are distributed in one clade composed by different genera (the major baeocystous clade), plus two separated branches containing <italic>Pleurocapsa</italic> sp. PCC 7327 and <italic>Chroococcidiopsis thermalis</italic> PCC 7203, respectively (<xref ref-type="bibr" rid="B38">Shih et al., 2013</xref>; <xref ref-type="bibr" rid="B10">Calteau et al., 2014</xref>). Baeocyte-forming strains have an ubiquitous distribution and can be found in terrestrial and desert habitats, in freshwater and marine environments as well as in the intertidal zones (<xref ref-type="bibr" rid="B11">Castenholz, 2001</xref>). Most are epilithic or endolithic, and some are true endoliths with the capacity to dig and grow into calcium carbonates or sequestrate Ca-carbonates in their baeocytes (<xref ref-type="bibr" rid="B20">Garcia-Pichel et al., 2010</xref>; <xref ref-type="bibr" rid="B5">Benzerara et al., 2014</xref>; <xref ref-type="bibr" rid="B21">Guida and Garcia-Pichel, 2016</xref>) leading to marine and terrestrial carbonate erosion and deleterious effects on coral reef and bivalve ecology. <italic>Hyella</italic> is a euendolithic baeocytous genus characterized by cells surrounded by a firm sheath, and the thalli often form branching pseudofilaments that can grow on calcium carbonate substrates (<xref ref-type="bibr" rid="B3">Al-Thukair and Golubic, 1991</xref>; <xref ref-type="bibr" rid="B2">Al-Thukair, 2011</xref>; <xref ref-type="bibr" rid="B8">Brito et al., 2017</xref>). To date, thirteen genomes representing five baeocyte-forming genera have been reported (NCBI and IMG/JGI databases) but none from <italic>Hyella</italic> is available so far.</p>
<p>Beyond evolution and classification of the organisms, the growing interest in genomics of cyanobacteria is driven by the discovery of new drugs (<xref ref-type="bibr" rid="B23">Kleigrewe et al., 2015</xref>; <xref ref-type="bibr" rid="B30">Moss et al., 2016</xref>). A genome-mining study revealed that up to 70% of cyanobacterial genomes contain polyketide synthase (PKS) and non-ribosomal peptide synthetase (NRPS) pathways or hybrids of these two (<xref ref-type="bibr" rid="B10">Calteau et al., 2014</xref>). Molecules derived from these pathways constitute the majority of known cyanobacterial NPs, but only 20% of the gene clusters of the PKS and NRPS pathways could be assigned to known compounds (<xref ref-type="bibr" rid="B10">Calteau et al., 2014</xref>), highlighting a large number of orphan clusters, with products yet to be discovered. In addition, gene clusters involved in the ribosome-dependent synthesis and post-translationally modified peptides (RiPPs) are also present throughout the phylum (<xref ref-type="bibr" rid="B38">Shih et al., 2013</xref>). RiPPs, PKS and NRPS gene clusters are well represented within the few available baeocyte-forming cyanobacterial genomes, but little is known about these biosynthetic pathways and their products compared to any other cyanobacterial lineages investigated (<xref ref-type="bibr" rid="B16">Dittmann et al., 2015</xref>).</p>
<p>Previously, we isolated and characterized a new marine <italic>Hyella</italic> strain &#x2013; <italic>H. patelloides</italic> LEGE 07179 &#x2013; from a rocky beach on the North of Portugal (<xref ref-type="bibr" rid="B9">Brito et al., 2012</xref>, <xref ref-type="bibr" rid="B8">2017</xref>), and a preliminary metabolomic analysis revealed the potential of this cyanobacterium to produce compounds related to neopeptin and antanapeptin as well as novel ones (<xref ref-type="bibr" rid="B7">Brito et al., 2015</xref>). The present study aimed at enlarging the baeocytous cyanobacterial genomic representation by sequencing for the first time the genome of <italic>Hyella</italic>. Moreover, a comprehensive comparative study with the available genomes of its closest relatives was performed to evaluate the distinctive characteristics of the genus. In addition, extensive analyses regarding NPs biosynthetic gene clusters were carried out to obtain an overview of their diversity and putative products.</p>
</sec>
<sec id="S2">
<title>Results</title>
<p><italic>Hyella patelloides</italic> LEGE 07179 (hereafter referred to <italic>H. patelloides</italic>) (<xref ref-type="fig" rid="F1">Figures 1A<sub>1</sub>&#x2013;A<sub>3</sub></xref>) was isolated from a <italic>Patella</italic> sp. shell collected from the intertidal zone of a rocky beach in the North of Portugal (<xref ref-type="fig" rid="F1">Figures 1B<sub>1</sub>&#x2013;B<sub>3</sub></xref>). This species displays club-shaped or cylindrical cells that divide by multiple binary fission originating baeocytes, and cells/colonies are surrounded by a multi-layered sheath from which pseudofilaments can emerge, especially when growing in solid medium (<xref ref-type="fig" rid="F1">Figures 1A<sub>1</sub>&#x2013;A<sub>3</sub></xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption><p>Light <bold>(A<sub>1</sub>,A<sub>2</sub>)</bold> and transmission electron <bold>(A<sub>3</sub>)</bold> micrographs of <italic>Hyella patelloides</italic> LEGE 07179. Geographical location <bold>(B<sub>1</sub>)</bold>, graphical representation <bold>(B<sub>2</sub>)</bold> and image of the sampling site <bold>(B<sub>3</sub>)</bold>. Scale bars: <bold>(A<sub>1</sub>,A<sub>2</sub>)</bold>, 20 &#x03BC;m; <bold>(A<sub>3</sub>)</bold>, 0.5 &#x03BC;m.</p></caption>
<graphic xlink:href="fmicb-11-01527-g001.tif"/>
</fig>
<sec id="S2.SS1">
<title>Genome Properties, Phylogeny and Comparative Analysis</title>
<p>The draft genome of <italic>H. patelloides</italic> has an estimated size of about 8.1 Mb assembled in 675 contigs, with a 37.57% GC content and a coverage of 107.8 times. The average contig size is 11946.2 bp and the N50 is 20427. Gene annotation revealed 8104 CDS with three rRNA genes and 50 tRNA genes. Despite using both 200 and 600-base-pair-read libraries, the genome, with a high proportion of repetitive sequences, could not be assembled into a single scaffold. In addition, <italic>H</italic>. <italic>patelloides</italic> genome was evaluated for its completeness and contamination level using CheckM (<xref ref-type="bibr" rid="B33">Parks et al., 2015</xref>). The results obtained (completeness &#x2013; 99.2% and contamination level &#x2013; 1.9%) highlight its quality (according to <xref ref-type="bibr" rid="B33">Parks et al., 2015</xref>).</p>
<p>The main characteristics of <italic>H</italic>. <italic>patelloides</italic> genome were compared to all baeocyte-forming cyanobacterial genomes available (<xref ref-type="supplementary-material" rid="TS1">Supplementary Table S1</xref>). The genome sizes of these strains range from 4.9 to 8.1 Mb with a low GC content (between 35 and 45%), despite their various origins (marine, freshwater, soil and hot and mineral springs). The majority of these genomes (obtained from axenic or non-axenic strains) remain in draft form, notably the largest ones (&#x003E;10 scaffolds) (<xref ref-type="supplementary-material" rid="TS1">Supplementary Table S1</xref>).</p>
<p>Based on the previous 16S rRNA gene phylogenetic study, <italic>H. patelloides</italic> is placed within the major baeocystous clade (<xref ref-type="bibr" rid="B8">Brito et al., 2017</xref>). Thus, we selected the following <italic>H</italic>. <italic>patelloides</italic> closest strains for the comparative and phylogenomic analysis: <italic>Chroococcidiopsis</italic> sp. PCC 6712, <italic>Xenococcus</italic> sp. PCC 7305, <italic>Myxosarcina</italic> sp. GI1, <italic>Pleurocapsa</italic> sp. PCC 7319, <italic>Stanieria cyanosphaera</italic> PCC 7437 and <italic>Stanieria</italic> sp. NIES 3757 as well as <italic>Cyanothece</italic> sp. PCC 8802 and <italic>Moorea producens</italic> 3L as outgroups. The phylogenomic analysis was based on 1209 orthologous genes (<xref ref-type="supplementary-material" rid="TS2">Supplementary Table S2</xref> and <xref ref-type="supplementary-material" rid="DS6">Supplementary File S1</xref>), shared between the selected strains, and three sub-clusters appear clearly defined in the Bayesian phylogenetic tree: one composed by <italic>H</italic>. <italic>patelloides</italic>, <italic>Chroococcidiopsis</italic> sp. PCC 6712 and <italic>Xenococcus</italic> sp. PCC 7305 (S1), another by <italic>Myxosarcina</italic> sp. GI1 and <italic>Pleurocapsa</italic> sp. PCC 7319 (S2), and a third one including the two <italic>Stanieria</italic> strains (NIES 3757 and PCC 7437) (S3) (<xref ref-type="fig" rid="F2">Figure 2</xref>). These sub-clusters are highly congruent with the previous 16S rRNA phylogeny (<xref ref-type="bibr" rid="B8">Brito et al., 2017</xref>), and are in line with other phylogenetic studies based on the concatenation of conserved proteins (<xref ref-type="bibr" rid="B38">Shih et al., 2013</xref>; <xref ref-type="bibr" rid="B10">Calteau et al., 2014</xref>). The two <italic>Stanieria</italic> are the most distantly related to <italic>H</italic>. <italic>patelloides</italic>, whereas the most closely related cyanobacterium is the freshwater <italic>Chroococcidiopsis</italic> sp. PCC 6712, however this strain is still highly divergent from <italic>Hyella</italic>. Indeed, using the same alignment of these 1209 orthologous genes, the synonymous changes per synonymous position (<italic>Ks</italic>) and non-synonymous changes per non-synonymous position (<italic>Ka</italic>) values between these two strains were estimated as 0.9043 (253971.3 synonymous positions analyzed) and 0.0888 (842237.6 non-synonymous positions analyzed), respectively. These values are between those obtained when comparing the two <italic>Stanieria</italic> species [<italic>Ks</italic> = 0.2828 (255486.8 synonymous positions analyzed) and <italic>Ka</italic> = 0.0259 (840722.1 non-synonymous positions analyzed)] and those obtained when comparing <italic>Myxosarcina</italic> sp. G1 and <italic>Pleurocapsa</italic> sp. PCC 7319 [(<italic>Ks</italic> = 1.8362 (258426.0 synonymous positions analyzed) and <italic>Ka</italic> = 0.1575 (837783.0 non-synonymous positions analyzed)] (<xref ref-type="supplementary-material" rid="TS3">Supplementary Table S3</xref>). Nevertheless, it is difficult to translate such divergence values into time (million years) since according to the Tajima&#x2019;s relative rate tests (compares evolutionary rates between species), the different lineages are clearly evolving at different rates. The pattern is more evident when analyzing third codon position than amino acid differences, thus significant changes in mutation rate are the most likely explanation for our observations (<xref ref-type="supplementary-material" rid="TS4">Supplementary Table S4</xref>). It should also be noted that for most comparisons involving third codon positions only, saturation could have an important impact on the inferences, especially if different species have different codon preferences. When performing the test using amino acid differences, significantly different rates are observed in four comparisons, three of them involving <italic>H</italic>. <italic>patelloides</italic> (<xref ref-type="supplementary-material" rid="TS4">Supplementary Table S4</xref>). The lineage leading to <italic>H</italic>. <italic>patelloides</italic> is accumulating amino acid differences faster than the lineage leading to <italic>Stanieria</italic>, but slower than the lineages leading to <italic>Xenococcus</italic> sp. PCC 7305, thus analysis of the <italic>H</italic>. <italic>patelloides</italic> biology and genome could reveal new insights into cyanobacterial biology and diversity.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption><p>Bayesian phylogenetic tree based on 1209 concatenated orthologous genes shared between <italic>Hyella patelloides</italic> LEGE 07179 and the cyanobacterial strains selected for this analysis. <italic>Moorea producens</italic> 3L and <italic>Cyanothece</italic> sp. PCC 8802 were used to root the tree. Numbers along branches indicate posterior credibility probability values. The strains&#x2019; habitat is highlighted by different colors: Blue &#x2013; marine; Green &#x2013; freshwater; Brown &#x2013; soil; Pink &#x2013; rice field.</p></caption>
<graphic xlink:href="fmicb-11-01527-g002.tif"/>
</fig>
</sec>
<sec id="S2.SS2">
<title>Homologous Genes</title>
<p>When looking to the orthologous and paralogous genes shared between <italic>H</italic>. <italic>patelloides</italic> and the selected cyanobacterial strains, between 3244 (40%) and 3880 (47.9%) of <italic>H</italic>. <italic>patelloides</italic> genes have orthologs in another strain (<xref ref-type="table" rid="T1">Table 1</xref>). If both orthologs and paralogs are considered, in between 5480 and 5998 <italic>H</italic>. <italic>patelloides</italic> genes are recognized in the other strains studied (<xref ref-type="table" rid="T1">Table 1</xref>). The percentage of <italic>H</italic>. <italic>patelloides</italic> duplicated genes (27%) is higher than the ones observed for any other baeocyte-genomes, notably the ones of the same clade, <italic>Chroococcidiopsis</italic> sp. PCC 6712 (838 genes; 16.2% of the <italic>Chroococcidiopsis</italic> genes) and <italic>Xenococcus</italic> sp. PCC 7305 (1090 genes; 20.1% of the <italic>Xenococcus</italic> genes). However, it is closer to the percentage observed for <italic>Myxosarcina</italic> sp. GI1 (1593 genes; 24.4% of the <italic>Myxosarcina</italic> genes) and <italic>Pleurocapsa</italic> sp. PCC 7319 (1640 genes; 24.3% of the <italic>Pleurocapsa</italic> genes) (<xref ref-type="supplementary-material" rid="TS5">Supplementary Table S5</xref>). This is compatible with an independent loss of duplicated genes in the <italic>Chroococcidiopsis</italic> and <italic>Xenococcus</italic> lineages, a gain of duplicated genes in the <italic>H</italic>. <italic>patelloides</italic> lineage, but also with other more complex scenarios, involving, for instance, horizontal gene transfer (HGT). Remarkably, <italic>H</italic>. <italic>patelloides</italic> shares more genes with the more distantly related marine strains of the S2 clade (<italic>Pleurocapsa</italic> sp. PCC 7319 and <italic>Myxosarcina</italic> sp. GI1) than with the more closely related marine strain of the clade S1 (<italic>Xenococcus</italic> sp. PCC 7305). This could indicate a significant gene loss in the <italic>Xenococcus</italic> lineage but it is also compatible with HGT. In addition, there is a high number of <italic>H</italic>. <italic>patelloides</italic> genes (2106&#x2013;2624) with no similarity to the genes described for each of the strains studied (<xref ref-type="table" rid="T1">Table 1</xref>). However, if all strains are considered as if they were a single one, only 794 <italic>H. patelloides</italic> genes do not show similarity to the other baeocyte-forming cyanobacterial strains studied, as if, by chance alone, <italic>H. patelloides</italic> retains genes that are lost in the other lineages.</p>
<table-wrap position="float" id="T1">
<label>TABLE 1</label>
<caption><p>Number/percentage of orthologous and paralogous genes shared by <italic>Hyella patelloides</italic> LEGE 07179 and the selected baeocyte-forming cyanobacterial strains, as well as the number/percentage of <italic>H</italic>. <italic>patelloides</italic> genes with no similarity with the ones from those strains.</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<thead>
<tr>
<td valign="top" align="left"><bold>Cyanobacterial strains</bold></td>
<td valign="top" align="center"><bold>Orthologous</bold></td>
<td valign="top" align="center"><bold>Paralogous</bold></td>
<td valign="top" align="center"><bold>No similarity</bold></td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><italic>Chroococcidiopsis</italic> sp. PCC 6712</td>
<td valign="top" align="center">3619 (44.7%)</td>
<td valign="top" align="center">2256 (27.8%)</td>
<td valign="top" align="center">2229 (27.5%)</td>
</tr>
<tr>
<td valign="top" align="left"><italic>Xenococcus</italic> sp. PCC 7305</td>
<td valign="top" align="center">3320 (41.0%)</td>
<td valign="top" align="center">2200 (27.1%)</td>
<td valign="top" align="center">2584 (31.9%)</td>
</tr>
<tr>
<td valign="top" align="left"><italic>Myxosarcina</italic> sp. GI1</td>
<td valign="top" align="center">3573 (44.1%)</td>
<td valign="top" align="center">2291 (28.3%)</td>
<td valign="top" align="center">2240 (27.6%)</td>
</tr>
<tr>
<td valign="top" align="left"><italic>Pleurocapsa</italic> sp. PCC 7319</td>
<td valign="top" align="center">3880 (47.9%)</td>
<td valign="top" align="center">2118 (26.1%)</td>
<td valign="top" align="center">2106 (26.0%)</td>
</tr>
<tr>
<td valign="top" align="left"><italic>Stanieria cyanosphaera</italic> PCC 7437</td>
<td valign="top" align="center">3282 (40.5%)</td>
<td valign="top" align="center">2303 (28.4%)</td>
<td valign="top" align="center">2519 (31.1%)</td>
</tr>
<tr>
<td valign="top" align="left"><italic>Stanieria</italic> sp. NIES 3757</td>
<td valign="top" align="center">3244 (40.0%)</td>
<td valign="top" align="center">2236 (27.6%)</td>
<td valign="top" align="center">2624 (32.4%)</td>
</tr>
</tbody>
</table></table-wrap>
</sec>
<sec id="S2.SS3">
<title>Horizontal Gene Transfer</title>
<p>The high number of putative orthologs shared between <italic>H</italic>. <italic>patelloides</italic> and the other baeocyte-forming strains studied (<xref ref-type="table" rid="T1">Table 1</xref>) suggests horizontal gene transfer (HGT). To evaluate this transfer, the <italic>Ks</italic> values for a set of orthologs identified in all strains (&#x201C;All&#x201D;; <xref ref-type="supplementary-material" rid="TS2">Supplementary Table S2</xref>) were compared against the set of orthologs found in <italic>H. patelloides</italic> and another strain only (&#x201C;Pair&#x201D;; <xref ref-type="supplementary-material" rid="TS6">Supplementary Table S6</xref>). The dataset for &#x201C;All&#x201D; is the 1209 concatenated gene alignments with an average of 261835.7 synonymous sites, while the datasets for the <italic>Ks</italic> distributions under the label &#x201C;Pair&#x201D; for <italic>Chroococcidiopsis</italic> sp. PCC 6712, <italic>Xenococcus</italic> sp. PCC 7305, <italic>Myxosarcina</italic> sp. GI1, <italic>Pleurocapsa</italic> sp. PCC 7319, <italic>Stanieria cyanosphaera</italic> PCC 743, and <italic>Stanieria</italic> sp. NIES 3757 were based on 126, 62, 99, 185, 26, and 19 concatenated gene alignments, respectively (with 20274.8, 8366.7, 14767, 30910.2, 4798.08, and 1581.6, synonymous sites, respectively). If the pattern was created by HGT, the <italic>Ks</italic> value estimated for a &#x201C;Pair&#x201D; will be lower than the <italic>Ks</italic> estimated for &#x201C;All.&#x201D; In fact, this tendency is observed between <italic>H</italic>. <italic>patelloides</italic> and all marine strains (<italic>Xenococcus</italic> sp. PCC 7305, <italic>Myxosarcina</italic> sp. GI1 and <italic>Pleurocapsa</italic> sp. PCC 7319) and <italic>Stanieria cyanosphaera</italic> PCC 7437 (<xref ref-type="fig" rid="F3">Figure 3</xref>), supporting the high probability of occurring HGT events between these strains. Then, we investigated if the putative orthologs shared between <italic>H</italic>. <italic>patelloides</italic> and the other baeocyte-forming strains have a recognizable function, since they could be &#x201C;false gene predictions.&#x201D; Therefore, the annotated genes were divided into different categories (conserved protein of unknown function/protein with recognizable function, transposase, and non-conserved protein of unknown function), and distributed into different classes taken into account their presence/absence in one or more of the strains studied [classes 1 to 6 correspond to genes that are absent in one (class 1), two (class 2), three (class 3), four (class 4), five (class 5), and six (class 6) strains]. Thus, it was possible to observe that a high percentage of these genes are annotated as conserved protein of unknown function or protein with recognizable function (notably genes that are only absent in one or two strains and so, present in the majority of them &#x2013; Classes 1 and 2). However, when we look to the genes that are only present in three or less strains, this percentage slightly decreases (Classes 3 to 6) (<xref ref-type="fig" rid="F4">Figure 4</xref> and <xref ref-type="supplementary-material" rid="TS7">Supplementary Tables S7</xref>, <xref ref-type="supplementary-material" rid="TS8">S8</xref>). In addition, some genes are identified as &#x201C;transposase&#x201D; or &#x201C;non-conserved protein of unknown function.&#x201D; Taken into account the low percentage of &#x201C;transposase&#x201D; genes identified, the high number of putative orthologs found is not an artifact created by annotating mobile elements as <italic>H</italic>. <italic>patelloides</italic> genes (<xref ref-type="fig" rid="F4">Figure 4</xref> and <xref ref-type="supplementary-material" rid="TS7">Supplementary Tables S7</xref>, <xref ref-type="supplementary-material" rid="TS8">S8</xref>). Concerning the genes absent in six strains and thus only present in <italic>H</italic>. <italic>patelloides</italic> (Class 6), the majority of them are annotated as non-conserved protein of unknown function, but still there is a reasonable percentage of strain-specific genes identified as conserved protein/protein with a recognizable function (<xref ref-type="fig" rid="F4">Figure 4</xref> and <xref ref-type="supplementary-material" rid="TS7">Supplementary Tables S7</xref>, <xref ref-type="supplementary-material" rid="TS8">S8</xref>). Closely related species are expected to have more genes in common than distantly related ones. Therefore, we looked at whether when a gene is only absent in one or few strains, that strain(s) is usually the same one or not. Therefore, as can be observed in <xref ref-type="fig" rid="F5">Figure 5</xref>, when a gene is absent just in one strain (Class 1), in the majority of the cases <italic>Xenococcus</italic> sp. PCC 7305 is that strain (light blue bar, <xref ref-type="fig" rid="F5">Figure 5</xref> and <xref ref-type="supplementary-material" rid="TS8">Supplementary Table S8</xref>). In addition, <italic>Stanieria</italic> strains and <italic>Xenococcus</italic> sp. PCC 7305 are the ones for which more genes are absent in the remaining classes (Classes 2 to 5). Although, this was expected for the <italic>Stanieria</italic> strains, since they are the ones more distantly related to <italic>H</italic>. <italic>patelloides</italic>, <italic>Xenococcus</italic> sp. PCC 7305 appeared as the strain with a higher number of &#x201C;missing genes&#x201D; within the group of baeocyte-forming strains studied. Interestingly, in the Class 5 (genes absent in five strains and thus, only present in <italic>H</italic>. <italic>patelloides</italic> and another strain), <italic>Pleurocapsa</italic> sp. PCC 7319 is the strain with the lowest frequency followed by the <italic>H</italic>. <italic>patelloides</italic> closest strain <italic>Chroococcidiopsis</italic> sp. PCC 6712 (<xref ref-type="fig" rid="F5">Figure 5</xref> and <xref ref-type="supplementary-material" rid="TS8">Supplementary Table S8</xref>). Thus, for the majority of the cases where <italic>H. patelloides</italic> genes are missing in other baeocyte-forming strains, they are still present in <italic>Pleurocapsa</italic> sp. PCC 7319.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption><p>Comparison of the synonymous rate (Ks) distributions based on sets of 500 non-overlapping synonymous sites obtained using concatenated gene alignments. All &#x2013; sets of genes for which a putative orthologous gene was found in all species; Pair &#x2013; genes for which the ortholog was found only in <italic>Hyella patelloides</italic> LEGE 07179 and the analyzed strain.</p></caption>
<graphic xlink:href="fmicb-11-01527-g003.tif"/>
</fig>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption><p>Percentage of <italic>Hyella patelloides</italic> LEGE 07179 genes identified as conserved protein of unknown function/protein with recognizable function, transposase or non-conserved protein of unknown function absent in one or more of the baeocyte-forming cyanobacteria studied. Class 1 &#x2013; genes absent in one strain; Class 2 &#x2013; genes absent in two strains; Class 3 &#x2013; genes absent in three strains; Class 4 &#x2013; genes absent in four strains; Class 5 &#x2013; genes absent in five strains; Class 6 &#x2013; genes absent in six strains (only present in <italic>H</italic>. <italic>patelloides</italic>).</p></caption>
<graphic xlink:href="fmicb-11-01527-g004.tif"/>
</fig>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption><p>Frequency of genes from the different classes analyzed (1 to 5) per baeocyte-forming cyanobacterial strain studied. Class 1 &#x2013; genes absent in one strain; Class 2 &#x2013; genes absent in two strains; Class 3 &#x2013; genes absent in three strains; Class 4 &#x2013; genes absent in four strains; Class 5 &#x2013; genes absent in five strains.</p></caption>
<graphic xlink:href="fmicb-11-01527-g005.tif"/>
</fig>
<p>In addition, we addressed if the HGT events could be mainly related to plasmid transfer. Among the studied strains, only <italic>Stanieria</italic> spp. have their genomes assigned in chromosome and plasmids (<italic>Stanieria cyanosphaera</italic> PCC 7437: Chr &#x2013; 5.04 Mb and 5 plasmids &#x2013; 0.50 Mb (9.1%); <italic>Stanieria</italic> sp. NIES-3757: Chr &#x2013; 5.32 Mb and 1 plasmid &#x2013; 0.14 Mb (2.6%). Therefore, plasmid sequences in the <italic>H</italic>. <italic>patelloides</italic> genome, as well as in the other strains studied, were predicted using PlasFlow (default parameters, threshold = 0.7) (<xref ref-type="supplementary-material" rid="TS9">Supplementary Table S9</xref>). For both <italic>Stanieria</italic> strains, the fraction (base pairs) predicted to be located in plasmids is in line with the values reported in the databases. However, this is only true if the category &#x201C;unclassified&#x201D; is assumed to be &#x201C;chromosome.&#x201D; Therefore, for the remaining genomes, the same assumption was followed. Concerning to <italic>H</italic>. <italic>patelloides</italic>, 35% of its genome is predicted to be constituted by plasmids. This value is rather high, when compared to those from the other strains (0.4 to 21%), namely compared to <italic>Stanieria</italic> (less than 10%). Thus, in order to have a value for <italic>H</italic>. <italic>patelloides</italic>, in the same range of that reported for <italic>Stanieria</italic>, a threshold as high as 0.85 would have to be used (<xref ref-type="fig" rid="F6">Figure 6</xref>). Genes for which an ortholog has been identified in all strains are often located in sequences predicted to be of plasmid origin, even when a high PlasFlow threshold value is used (<xref ref-type="supplementary-material" rid="TS10">Supplementary Table S10</xref>). This is an odd observation, since the same plasmid is unlikely to be present in all species, and the use of the traditional reverse BLAST for ortholog identification should not produce so many erroneous orthologous identifications. In fact, none of the 1209 genes from <italic>Stanieria</italic> sp. NIES-3757 that are present in the other strains, and were used in the phylogenetic analyses, are located in the single plasmid reported for this strain. Moreover, only two out of the 1209 genes from <italic>Stanieria cyanosphaera</italic> PCC 7437 are located in plasmid sequences [one from plasmid 1 (2503797950) and another from plasmid 3 (2503797623)] (<xref ref-type="supplementary-material" rid="TS2">Supplementary Table S2</xref>). Hence, the traditional reverse BLAST for ortholog identification seems to work well since, as expected, less than 0.17% of the genes present in all strains are located in <italic>Stanieria</italic> plasmids.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption><p>Fraction of <italic>Hyella patelloides</italic> LEGE 07179, <italic>Stanieria cyanosphaera</italic> PCC 7437 and <italic>Stanieria</italic> sp. NIES 3757 genomes constituted by plasmids, using PlasFlow with different cut-offs (number between brackets).</p></caption>
<graphic xlink:href="fmicb-11-01527-g006.tif"/>
</fig>
<p>Furthermore, we looked into the genes present in <italic>H</italic>. <italic>patelloides</italic> and another strain only, and according to PlasFlow some of these genes (29 to 53%) are predicted to be located in plasmids, but the majority are in the &#x201C;unclassified&#x201D; category (<xref ref-type="table" rid="T2">Table 2</xref>). When considering the set of 19 genes predicted to be present in <italic>H. patelloides</italic> and <italic>Stanieria</italic> sp. NIES 3757, two genes are located in plasmid 1 (gene 879 predicted to be a pseudogene and WP_096388421.1_4869). When considering the 26 genes predicted to be present in <italic>H. patelloides</italic> and <italic>Stanieria cyanosphaera</italic> PCC 7437, nine genes are from plasmids (2503802167, 2503797825, 2503798028, 2503797810, 2503797825, and 2503798007 on plasmid 1; 2503797602, 2503797607, and 2503797643 on plasmid 3) (<xref ref-type="supplementary-material" rid="TS6">Supplementary Table S6</xref>). Therefore, as expected, since plasmids are unlikely to be present in all species, genes that are found in <italic>H. patelloides</italic> and another strain only, seem to have a much higher chance of being located in plasmids than the ones present in all species (<italic>P</italic> &#x003C; 0.0005 and <italic>P</italic> &#x003C; 0.000001 for <italic>Stanieria</italic> sp. NIES 3757 and <italic>Stanieria cyanosphaera</italic> PCC 7437, respectively; Fisher exact test). However, most sequences identified as of &#x201C;plasmid origin&#x201D; by PlasFlow show genes that are present in all species, as well as in <italic>H. patelloides</italic> and in a single strain (<xref ref-type="supplementary-material" rid="TS10">Supplementary Table S10</xref>). Moreover, <italic>H. patelloides</italic> genes that have orthologs in <italic>Stanieria</italic> plasmids can be located in scaffolds that also contain genes identified in all species. Therefore, it is not easy to be sure that the sequences identified as of plasmid origin by PlasFlow are correctly identified. However, these findings should be regarded as indicative rather than definitive, and support the inference that <italic>H</italic>. <italic>patelloides</italic> genome is constituted by chromosomes and plasmids and some of the genes present in this strain and another one only (from the set studied), are located in plasmids.</p>
<table-wrap position="float" id="T2">
<label>TABLE 2</label>
<caption><p>Number of genes present in <italic>Hyella patelloides</italic> LEGE 07179 and in another baeocyte-forming cyanobacterial strain only, and their location in plasmid or chromosomal according to the PlasFlow prediction.</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<thead>
<tr>
<td valign="top" align="left"><bold>Cyanobacterial strains</bold></td>
<td valign="top" align="center"><bold>Chromosome</bold></td>
<td valign="top" align="center"><bold>Plasmid</bold></td>
<td valign="top" align="center"><bold>Unclassified</bold></td>
<td valign="top" align="center"><bold>Total</bold></td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><italic>Chroococcidiopsis</italic> sp. PCC 6712</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">37</td>
<td valign="top" align="center">84</td>
<td valign="top" align="center">126</td>
</tr>
<tr>
<td valign="top" align="left"><italic>Xenococcus</italic> sp. PCC 7305</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">21</td>
<td valign="top" align="center">39</td>
<td valign="top" align="center">62</td>
</tr>
<tr>
<td valign="top" align="left"><italic>Myxosarcina</italic> sp. GI1</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">33</td>
<td valign="top" align="center">62</td>
<td valign="top" align="center">99</td>
</tr>
<tr>
<td valign="top" align="left"><italic>Pleurocapsa</italic> sp. PCC 7319</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">94</td>
<td valign="top" align="center">86</td>
<td valign="top" align="center">185</td>
</tr>
<tr>
<td valign="top" align="left"><italic>Stanieria</italic> sp. NIES-3757</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">10</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">19</td>
</tr>
<tr>
<td valign="top" align="left"><italic>Stanieria cyanosphaera</italic> PCC 7437</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">9</td>
<td valign="top" align="center">16</td>
<td valign="top" align="center">26</td>
</tr>
</tbody>
</table></table-wrap>
</sec>
<sec id="S2.SS4">
<title>COG Categories Distribution</title>
<p>The distribution of genes among the Clusters of Orthologous Groups (COG) functional categories, allowed to assign 58.9% of the <italic>H</italic>. <italic>patelloides</italic> genes (<xref ref-type="supplementary-material" rid="TS11">Supplementary Table S11</xref>), slightly less than those observed for the other baeocyte-forming cyanobacterial strains analyzed (between 64 and 68%). Interestingly, the seven genomes exhibit similarities in the COG functional category distribution. Thus, the percentage of genes assigned to the different functional categories is similar for all strains analyzed, independently of their genome size.</p>
</sec>
<sec id="S2.SS5">
<title>Natural Product Biosynthetic Gene Clusters and Hydrocarbons Production</title>
<p>A total of 21 biosynthetic gene clusters (BGCs) involved in the production of natural products were identified in the <italic>H</italic>. <italic>patelloides</italic> genome. Interestingly, <italic>H</italic>. <italic>patelloides</italic> and <italic>Pleurocapsa</italic> sp. PCC 7319, both isolated from a marine organism shell (intertidal zone), harbor the highest number of predicted BGCs (21 and 23, respectively) (<xref ref-type="supplementary-material" rid="TS1">Supplementary Table S1</xref>). The <italic>H</italic>. <italic>patelloides</italic> BGCs include three polyketide synthase (PKS), four non-ribosomal peptide synthetase (NRPS), five hybrid PKS/NRPS, six RiPPs, and three terpenes. Remarkably, most of these BGCs displayed low similarity with the ones available in the antiSMASH database (<xref ref-type="supplementary-material" rid="TS12">Supplementary Table S12</xref>). However, two of <italic>H</italic>. <italic>patelloides</italic> BGCs have similarities with the ones identified in other cyanobacterial strains. The first one (PKS gene cluster) shares 13 genes in synteny with an orphan cluster identified in <italic>Pleurocapsa</italic> sp. PCC 7319. This shared cluster is constituted by one module of PKS (with a dehydratase domain in PCC 7319) and eleven diversified tailoring enzymes with putative function of methyl- and sulfotransferases and transporters (<xref ref-type="fig" rid="F7">Figure 7</xref> and <xref ref-type="supplementary-material" rid="TS13">Supplementary Table S13</xref>). Considering the high similarity of these two clusters (59%) and the close relationship between these strains, the characterization of the putative compound can be undertaken using both strains. The second BGC is a PKS gene cluster involved in the production of hydrocarbons, a terminal olefin synthase pathway (OLS pathway), which is also present in the other baeocyte-forming strains studied (<xref ref-type="bibr" rid="B48">Zhu et al., 2018</xref>). Afterward, the hydrocarbon composition of <italic>H. patelloides</italic> was further investigated, revealing that this cyanobacterial strain produces hydrocarbons with C<sub>15</sub> chain length, such as 1-pentadecene (C<sub>15:1</sub>, &#x0394;1) and 2-pentadecene (C<sub>15:1</sub>, &#x0394;2) (<xref ref-type="fig" rid="F8">Figure 8A</xref>). Furthermore, the investigation of the fatty acid substrates for the OLS pathway showed that this strain synthesizes C<sub>14</sub>, C<sub>16</sub>, and C<sub>18</sub> fatty acids, being C<sub>16</sub> the most abundant. In total, the amount of fatty acid in <italic>H</italic>. <italic>patelloides</italic> exceeds 4% of the Dry Cell Weight (DCW) (<xref ref-type="fig" rid="F8">Figure 8B</xref>).</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption><p>PKS Gene cluster shared by <italic>Hyella patelloides</italic> LEGE 07179 and <italic>Pleurocapsa</italic> sp. PCC 7319. The genes are color coded according to their putative function: green for PKS (with the domains listed), dark green for 3-oxoacyl synthase, orange for cytochrome P450, pink for NAD(P)-binding domain-containing protein, blue for sulfotransferase, red for transporter, black for methyltransferase and gray for unknown proteins. KS, ketosynthase; AT, acyltransferase; ACP, acyl carrier protein; TE, thioesterase; DH, dehydratase. Further details provided in <xref ref-type="supplementary-material" rid="TS13">Supplementary Table S13</xref>.</p></caption>
<graphic xlink:href="fmicb-11-01527-g007.tif"/>
</fig>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption><p>Hydrocarbon profile of <italic>Hyella patelloides</italic> LEGE 07179. Terminal olefins with chain length of C<sub>15</sub> were identified <bold>(A)</bold>, and fatty acid composition of <italic>Hyella patelloides</italic> with chain lengths of C<sub>14</sub>, C<sub>16</sub>, and C<sub>18</sub> were identified <bold>(B)</bold>. Number after the colon indicates the number of double bonds, while the number after the triangle indicates the position of the double bound. Data are the means of three biological replicates, and error bars represent standard deviations.</p></caption>
<graphic xlink:href="fmicb-11-01527-g008.tif"/>
</fig>
<p>As stated above, the majority of the <italic>H</italic>. <italic>patelloides</italic> BGCs displayed low similarity with the databases. Thus, in order to evaluate if they could be strain-specific BGCs we run them through the BiG-SCAPE software (<xref ref-type="bibr" rid="B31">Navarro-Mu&#x00F1;oz et al., 2020</xref>). In this analysis, the 104 predicted BGCs from <italic>H</italic>. <italic>patelloides</italic> and the baeocyte-forming strains (<xref ref-type="supplementary-material" rid="TS1">Supplementary Table S1</xref>), as well as the ones from the Minimum Information about a Biosynthetic Gene cluster (MIBiG) repository (<xref ref-type="bibr" rid="B22">Kautsar et al., 2020</xref>), were compared. Using 0.3 and 0.5 similarity cut-offs no connections between <italic>H</italic>. <italic>patelloides</italic> BGCs and the other strains were obtained. However, the analysis with a 0.7 cut-off revealed four BGCs (2 PKS, 1 RiPP and 1 terpene) shared between <italic>H</italic>. <italic>patelloides</italic> and other strains (<xref ref-type="supplementary-material" rid="TS14">Supplementary Table S14</xref>). Of those clusters, the 2 PKS correspond to the ones mentioned above: the one that shares 13 genes in synteny with the orphan <italic>Pleurocapsa</italic> sp. PCC 7319 cluster (<xref ref-type="supplementary-material" rid="DS1">Supplementary Figure S1</xref>); the other PKS is the BGC involved in the production of a hydrocarbon (OLS pathway). This gene cluster family (GCF) comprises BGCs from <italic>Chroococcidiopsis</italic> sp. PCC 6712, <italic>Xenococcus</italic> sp. PCC 7305 and <italic>Myxosarcina</italic> sp. GI and it is connected to another one composed by BGCs from both <italic>Stanierias.</italic> According to the GCF phylogenetic tree, the <italic>H</italic>. <italic>patelloides</italic> BGC is closely related to the ones from <italic>Chroococcidiopsis</italic> sp. PCC 6712 and <italic>Xenococcus</italic> sp. PCC 7305 (<xref ref-type="supplementary-material" rid="DS2">Supplementary Figure S2</xref>). Although <italic>Pleurocapsa</italic> sp. PCC harbors this BGC, it appeared as a single node. The other two BGCs includes a bacteriocin/lanthipeptide gene cluster (RiPP) shared with <italic>Pleurocapsa</italic> sp. PCC 7319 and a terpene shared with <italic>Chroococcidiopsis</italic> sp. PCC 6712 (<xref ref-type="supplementary-material" rid="DS1">Supplementary Figure S1</xref>). These BGCs do not display similarities to known compounds. In addition, two GCFs composed by BGCs from <italic>H</italic>. <italic>patelloides</italic> related with different known ones from MIBiG (namely cyanopeptolin, micropeptin, cyanopeptin, and hapalosin biosynthetic gene clusters) were also observed (<xref ref-type="supplementary-material" rid="TS14">Supplementary Table S14</xref> and <xref ref-type="supplementary-material" rid="DS2">Supplementary Figure S2</xref>). The remaining <italic>H</italic>. <italic>patelloides</italic> BGCs appeared as unclustered individual nodes (singletons) representing each one a different family (<xref ref-type="supplementary-material" rid="TS14">Supplementary Table S14</xref>). Thus, <italic>H</italic>. <italic>patelloides</italic> displays 70% of strain-specific clusters that could be involved in the production of yet unknown compounds.</p>
<p>Additional genome mining analyses were performed in order to better characterize the <italic>H</italic>. <italic>patelloides</italic> BGCs. To target antibiotic producing BGCs, the cluster prioritization tool Antibiotic Resistance Target Seeker (ARTS) was used. ARTS identified 410 core/essential genes, 21 BGCs and 68 known resistance models. The core/essential genes identified with the ARTS criteria (duplication, BGC proximity, phylogeny and known resistance) are shown in <xref ref-type="supplementary-material" rid="TS15">Supplementary Table S15</xref>. Concerning the BGC proximity, this analysis revealed that 17 of the 21 BGCs harbored neighboring putative core (clusters 1, 4, 6, 7, 8, 9, 10, 14, 17, 18, and 19), known resistance (2, 13, 21) or both genes (clusters 5, 12, 15) (<xref ref-type="supplementary-material" rid="TS15">Supplementary Table S15</xref> and <xref ref-type="supplementary-material" rid="DS3">Supplementary Figure S3</xref>), emphasizing the ones with higher number of hits. For instance, in cluster 12, besides the core and two known resistant genes (ABC_efflux and pentapeptide repeat family), one core known resistance gene (Gp_dh_C) was also identified. In cluster 7, three of the core genes identified are marked as duplicated and with incongruent phylogeny. Therefore, some of the <italic>H</italic>. <italic>patelloides</italic> BGCs are associated with resistance mechanisms and further studies are needed in order to identity putative antibiotic-producing BGCs.</p>
</sec>
</sec>
<sec id="S3">
<title>Discussion</title>
<p>Despite the increasing number of available cyanobacterial genomes, there is an unbalanced distribution of genome sequences within the phylum. In the current study, the first draft genome from a member of <italic>Hyella</italic> genus is presented, contributing to overcome the total lack of information regarding this genus and to increase the data on baeocyte-forming cyanobacteria. The genome binning approaches ensured the quality of the <italic>H</italic>. <italic>patelloides</italic> draft genome and the CheckM analysis also supported it. Taken into account the <italic>H</italic>. <italic>patelloides</italic> position within the phylogenetic tree depicted in the present study, this strain constitutes a new representative of the main baeocytous subclade (within B2) previously described (<xref ref-type="bibr" rid="B38">Shih et al., 2013</xref>). Its large genome has the highest number of predicted ORFs compared to any of the baeocyte-forming cyanobacteria (<xref ref-type="supplementary-material" rid="TS1">Supplementary Table S1</xref>). Consequently, it was conceivable that <italic>H</italic>. <italic>patelloides</italic> genome could have unique features due to the expansion of its gene repertoire. When looking to the distribution of the genes among COG functional categories, the percentage of <italic>H</italic>. <italic>patelloides</italic> genes assigned to the different COG functional categories was within those found for the other strains analyzed (<xref ref-type="supplementary-material" rid="TS11">Supplementary Table S11</xref>). Thus, despite its large genome, the different gene functional categories seem to have been equally expanded. Nevertheless, it was still conceivable that <italic>H</italic>. <italic>patelloides</italic> shows some unique features, not only due to the observed high gene number, but also because it is highly divergent from the most closely related strains (<xref ref-type="fig" rid="F2">Figure 2</xref>). Furthermore, a large number of genes seem to have been horizontally transferred between <italic>H. patelloides</italic> and the baeocyte-forming strains, such as the marine <italic>Pleurocapsa</italic> sp. PCC 7319 (also isolated from a shell) and <italic>Myxosarcina</italic> sp. GI1 strains (<xref ref-type="fig" rid="F3">Figures 3</xref>, <xref ref-type="fig" rid="F5">5</xref>). Actually, there are more common genes between <italic>H</italic>. <italic>patelloides</italic> and <italic>Pleurocapsa</italic> sp. PCC 7319 than with its closest relative <italic>Chroococcidiopsis</italic> sp. PCC 6712, a freshwater strain (<xref ref-type="table" rid="T1">Table 1</xref> and <xref ref-type="fig" rid="F5">Figure 5</xref>). Since, <italic>H. patelloides, Pleurocapsa</italic> sp. PCC 7319 and <italic>Myxosarcina</italic> sp. GI1 are all originating from a marine environment, the occurrence of horizontal gene transfer (HGT) events between these strains is plausible. Actually, plasmid sequences were predicted in the <italic>H</italic>. <italic>patelloides</italic> genome (<xref ref-type="fig" rid="F6">Figure 6</xref> and <xref ref-type="supplementary-material" rid="TS9">Supplementary Tables S9</xref>, <xref ref-type="supplementary-material" rid="TS10">S10</xref>) and, although one cannot be sure that these sequences are correctly assigned, some genes present in <italic>H. patelloides</italic> and in another strain only, are predicted to be located in plasmids (<xref ref-type="table" rid="T2">Table 2</xref> and <xref ref-type="supplementary-material" rid="TS10">Supplementary Table S10</xref>). This might indicate that HGT events mainly involved plasmid transfer, between two distantly related species living in the same environment, supporting the hypothesis that <italic>H. patelloides</italic> has captured plasmids from distantly related species in the past. Still, <italic>H</italic>. <italic>patelloides</italic> possesses a considerable set of genes with no similarity to those present in the other six strains analyzed (794), but, although some of these genes are putative, there is a considerable percentage that are conserved or with a recognizable function (<xref ref-type="fig" rid="F4">Figure 4</xref> and <xref ref-type="supplementary-material" rid="TS7">Supplementary Tables S7</xref>, <xref ref-type="supplementary-material" rid="TS8">S8</xref>).</p>
<p>The genome-mining analysis reinforced the already suggested metabolic potential of <italic>H</italic>. <italic>patelloides</italic> (<xref ref-type="bibr" rid="B7">Brito et al., 2015</xref>). Remarkably, <italic>H. patelloides</italic> and <italic>Pleurocapsa</italic> sp. PCC 7319, both isolated from a marine organism shell, are the ones harboring the highest number of BGCs in their genomes (<xref ref-type="supplementary-material" rid="TS1">Supplementary Table S1</xref>). From the 21 identified in <italic>H</italic>. <italic>patelloides</italic>, more than half are PKS and NRPS clusters (3 PKS, 4 NRPS, and 5 PKS/NRPS hybrids). Since a high percentage correspond to strain-specific gene clusters, as observed by the BGCs similarity network analysis (<xref ref-type="supplementary-material" rid="TS14">Supplementary Table S14</xref>), we are probably facing new chemical scaffolds, corresponding to NPs with unique properties. In addition, according to ARTS prediction, some <italic>H</italic>. <italic>patelloides</italic> BGCs are colocalized with putative core/essential and/or resistance genes (<xref ref-type="supplementary-material" rid="TS15">Supplementary Table S15</xref> and <xref ref-type="supplementary-material" rid="DS3">Supplementary Figure S3</xref>), highlighting putative antibiotic producing BGCs. Moreover, in some studies the detection of duplicated housekeeping genes colocalized with BGCs led to the discovery of antibiotic producing gene clusters (<xref ref-type="bibr" rid="B43">Tang et al., 2015</xref>).</p>
<p>The <italic>H</italic>. <italic>patelloides</italic> BGC displaying high similarity to the orphan cluster previously identified in <italic>Pleurocapsa</italic> sp. PCC 7319 (<xref ref-type="fig" rid="F7">Figure 7</xref>) belongs to the CF-46 cluster family, only identified in five other cyanobacterial strains (<xref ref-type="bibr" rid="B10">Calteau et al., 2014</xref>). This cluster has similar gene content and organization with <italic>Pleurocapsa</italic> sp. PCC 7319 BGC, sharing 13 genes in synteny. This BGC is not linked to any known compound, as it happens for most of the baeocystous strain NPs. However, it is composed by one PKS followed by eleven tailoring genes, being more similar to the saxitoxin gene cluster than a long PKS gene cluster. Indeed, these genes might be working collectively to produce a family of compounds (as in saxitoxin and in some antibiotics) (<xref ref-type="bibr" rid="B17">Fischbach et al., 2009</xref>; <xref ref-type="bibr" rid="B28">Mihali et al., 2009</xref>). The gene clusters containing a single PKS, NRPS or hybrids thereof are common in cyanobacteria and, indeed, easily overlooked as considered as remnant gene clusters (<xref ref-type="bibr" rid="B10">Calteau et al., 2014</xref>). Therefore, it is highly interesting to investigate the common or related compounds produced by these two strains.</p>
<p>Furthermore, the BGC involved in the production of a terminal olefin, belongs to the CF-8 cluster family previously described (<xref ref-type="bibr" rid="B10">Calteau et al., 2014</xref>). The terminal olefin synthase (OLS pathway) is one of the two biosynthetic pathways that produce hydrocarbons from fatty acids identified in cyanobacteria (<xref ref-type="bibr" rid="B27">Mendez-Perez et al., 2011</xref>; <xref ref-type="bibr" rid="B12">Coates et al., 2014</xref>; <xref ref-type="bibr" rid="B48">Zhu et al., 2018</xref>). This pathway is composed by a large type I PKS with modular organization that includes the following domains organization: Fatty acyl-AMP ligase (FAAL), acyl carrier protein (ACP), ketosynthase (KS), acyltransferase (AT), ketoreductase (KR), ACP2, sulfotransferase (ST) and thioesterase (TE) (<xref ref-type="bibr" rid="B27">Mendez-Perez et al., 2011</xref>; <xref ref-type="bibr" rid="B12">Coates et al., 2014</xref>; <xref ref-type="bibr" rid="B48">Zhu et al., 2018</xref>). There is the indication that cyanobacterial strains harboring this pathway produced more hydrocarbons than those possessing the alternative one (<xref ref-type="bibr" rid="B12">Coates et al., 2014</xref>). Interestingly, the type of hydrocarbon produced by <italic>H. patelloides</italic> (C<sub>15</sub> chain length, <xref ref-type="fig" rid="F8">Figure 8A</xref>) was recently shown to be produced by <italic>Chroococcidiopsis</italic> sp. PCC 6712 and <italic>Xenococcus</italic> sp. PCC 7305 (<xref ref-type="bibr" rid="B48">Zhu et al., 2018</xref>). This data is in accordance with the phylogenetic analyses of the GCF, where the <italic>H</italic>. <italic>patelloides</italic> BGC is closely related with <italic>Chroococcidiopsis</italic> sp. PCC 6712 and <italic>Xenococcus</italic> sp. PCC 7305 BGCs. In addition, according to our phylogeny (<xref ref-type="fig" rid="F2">Figure 2</xref>), these two strains are the closest relatives to <italic>H</italic>. <italic>patelloides</italic>. This data is in accordance with <xref ref-type="bibr" rid="B48">Zhu et al. (2018)</xref> that suggested a correlation between the hydrocarbon profile and phylogeny based on the OLS pathway. <italic>H</italic>. <italic>patelloides</italic> and the other two C<sub>15</sub> terminal olefin-producing strains displayed the same fatty acid composition (<xref ref-type="fig" rid="F8">Figure 8B</xref>). Moreover, the amount of C<sub>14</sub> fatty acid produced by <italic>H</italic>. <italic>patelloides</italic> is higher and in line with the values observed for <italic>Cyanobacterium stanieri</italic> PCC 7202, <italic>Geminocystis herdmanii</italic> PCC 6308 and <italic>Chroococcidiopsis</italic> sp. PCC 6712 (<xref ref-type="bibr" rid="B48">Zhu et al., 2018</xref>). The amount of total fatty acid in <italic>H</italic>. <italic>patelloides</italic> (&#x003E;4% of DCW) is also higher, when compared to other terminal olefin-producing cyanobacterial strains (&#x223C;3% of DCW) (<xref ref-type="bibr" rid="B48">Zhu et al., 2018</xref>).</p>
<p>In summary, the first genome sequence of a cyanobacterium belonging to the baeocyte-forming genus <italic>Hyella</italic>, and the comparative analysis with its closest cyanobacterial counterparts, revealed its distinctiveness and, in particular, its unique biosynthetic potential (with more than twenty natural products BGCs identified). We found orphan and strain-specific BGCs that could be involved in the production of novel NPs and others could be studied as potential drug targets. Therefore, additional studies are needed to further characterize these clusters, as well as to identify and characterize the compounds produced and match them with the corresponding BGCs. For the BGC with a predicted product (terminal olefin) it was possible to quantify and characterize the hydrocarbon produced. Since terminal olefins have promising applications as advanced biofuels this molecular mechanism can be explored, for instance, using cell factories and a synthetic biology approach to increase the yield and become cost competitive.</p>
</sec>
<sec id="S4" sec-type="materials|methods">
<title>Materials and Methods</title>
<sec id="S4.SS1">
<title>Organism and Culture Conditions</title>
<p><italic>Hyella patelloides</italic> LEGE 07179 was previously isolated from a <italic>Patella</italic> sp. shell collected in the intertidal zone of the rocky beach in the North of Portugal (<xref ref-type="bibr" rid="B9">Brito et al., 2012</xref>, <xref ref-type="bibr" rid="B8">2017</xref>), and the unicyanobacterial culture is deposited at LEGE Culture Collection (LEGE CC) located at CIIMAR, Matosinhos, Portugal (<xref ref-type="bibr" rid="B34">Ramos et al., 2018</xref>). This strain is maintained in MN medium (<xref ref-type="bibr" rid="B35">Rippka, 1988</xref>) supplemented with 10 &#x03BC;g mL<sup>&#x2013;1</sup> of vitamin B<sub>12</sub> and kept at 25&#x00B0;C, under a 16 h light (15&#x2013;25 &#x03BC;mol photons m<sup>&#x2013;2</sup> s<sup>&#x2013;1</sup>)/8 h dark regimen.</p>
</sec>
<sec id="S4.SS2">
<title>Genome Sequencing and Assembly</title>
<p>Genomic DNA was extracted using the phenol/chloroform method described previously (<xref ref-type="bibr" rid="B41">Tamagnini et al., 1997</xref>), with the exception that the first aqueous phase was run through a Phase Lock Gel<sup>TM</sup> tube (5 Prime, Hilden, Germany) prior to chloroform extraction and DNA precipitation. Genomic DNA was sequenced at the Genomics Core Facility at the i3S institute (GENCORE), Porto. Libraries of 200 and 600-base-pair-reads were sequenced using an Ion Torrent S5<sup>TM</sup> XL Sequencer (Thermo Fisher Scientific). As the unicyanobacterial culture of <italic>Hyella</italic> was not axenic, its genomic data set was treated as a metagenome, and further binned to obtain cyanobacterial-specific contigs. The reads were initially assembled using the SPAdes v3.11.1 (<xref ref-type="bibr" rid="B4">Bankevich et al., 2012</xref>) and the ion-torrent specific option, the built-in read correction tool, and K-mer sizes 21, 33, 55, and 77. Contigs shorter than 1 kb were discarded resulting in a final set of 4818 contigs. A local BLASTn search using the 16S rRNA gene sequence of an uncultured marine cyanobacterium (JX477009) was performed and evidence for a minimum of nine different organisms was obtained. Besides the 16S rRNA gene sequence for <italic>H. patelloides</italic>, eight different bacteria were identified. Subject sequences were retrieved and a BLASTn analysis against the nt database at NCBI was carried out, revealing that the contaminants were from the Proteobacteria and Cytophaga-Flavobacterium-Bacteroides (CFB) groups. Thus, a local BLASTx was performed against a database containing the predicted proteomes of cyanobacteria (340 proteomes), proteobacteria (41109 proteomes) and CFB (1599 proteomes), and the first 20 hits were retrieved. 595 of the contigs displayed all hits with cyanobacterial proteins and they were considered as &#x201C;cyanobacterial contigs&#x201D; and retrieved. In addition, some contigs exhibited mostly hits with cyanobacterial proteins, and they were manually curated. 80 of them displayed a coverage compatible with their putative origin (higher than 70) and an identity much higher with cyanobacterial proteins than with CFB/proteobacteria and for this reason they were also retrieved. The remaining ones were considered as contaminants and were discarded.</p>
<p>To validate the previous approach and assure the accuracy and quality of the draft genome, the pegi3s Docker image<sup><xref ref-type="fn" rid="footnote1">1</xref></sup> of the automated binning tool MaxBin 2.0 (<xref ref-type="bibr" rid="B46">Wu et al., 2016</xref>) was also used and the majority of the cyanobacterial contigs (98%) are congruent, supporting our results (<xref ref-type="supplementary-material" rid="DS4">Supplementary Figure S4</xref>). However, there were 11 contigs that MaxBin 2.0 considered as non-cyanobacterial, although they display higher Blast identity and coverage with cyanobacteria than with other bacteria (<xref ref-type="supplementary-material" rid="DS5">Supplementary Figure S5A</xref>). There were also 27 contigs that MaxBin 2.0 considered as cyanobacterial ones but that show a high Blast identity and coverage both with cyanobacterial and non-cyanobacterial organisms (<xref ref-type="supplementary-material" rid="DS5">Supplementary Figure S5B</xref>). Thus, and in order to not wrongly infer horizontal gene transfer events (see section &#x201C;Results&#x201D;) these 27 contigs were not included.</p>
<p>The <italic>H</italic>. <italic>patelloides</italic> genome was submitted to the MicroScope platform v3.13.4 (<xref ref-type="bibr" rid="B45">Vallenet et al., 2019</xref>) for automatic annotation, and to evaluate genome completion and contamination using CheckM (<xref ref-type="bibr" rid="B33">Parks et al., 2015</xref>). The PlasFlow v1.1 was also used to predict plasmid sequences (<xref ref-type="bibr" rid="B24">Krawczyk et al., 2018</xref>; a Docker image is available at see text footnote 1).</p>
<p><italic>Hyella patelloides</italic> draft genome has been deposited to the European Nucleotide Archive (ENA) under the study accession number <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="PRJEB28569">PRJEB28569</ext-link>.</p>
</sec>
<sec id="S4.SS3">
<title>Homologous Genes and Phylogenomic Analysis</title>
<p>Based on the previous 16S rRNA gene phylogenetic study, <italic>H. patelloides</italic> is placed within the major baeocystous clade (<xref ref-type="bibr" rid="B8">Brito et al., 2017</xref>). Thus, <italic>H</italic>. <italic>patelloides</italic> genome was compared with each one of the following strains: <italic>Chroococcidiopsis</italic> sp. PCC 6712, <italic>Xenococcus</italic> sp. PCC 7305, <italic>Myxosarcina</italic> sp. GI1, <italic>Pleurocapsa</italic> sp. PCC 7319, <italic>Stanieria cyanosphaera</italic> PCC 7437 and <italic>Stanieria</italic> sp. NIES 3757 as well as <italic>Cyanothece</italic> sp. PCC 8802 and <italic>Moorea producens</italic> 3L as outgroups.</p>
<p>Orthologous and paralogous genes were identified using the traditional reverse BLAST, also known as reciprocal best hits (<xref ref-type="bibr" rid="B29">Moreno-Hagelsieb and Latimer, 2008</xref>). Briefly, CDSs from <italic>H</italic>. <italic>patelloides</italic> were used as query and CDSs from each one of the selected strains as subject in a local tblastx search (expect value of 0.05). The first CDS hit from each strain was retrieved and used as query and the CDSs of <italic>H</italic>. <italic>patelloides</italic> as subject, and again the first hit was retrieved. If the first tblastx search produces no hits, the query CDS from <italic>H</italic>. <italic>patelloides</italic> was labeled as &#x201C;no similarity.&#x201D; If the second tblastx search returns a sequence that is different from the original query sequence then the original query was labeled as &#x201C;paralog&#x201D;; if it returns the same sequence as the original one, it was labeled as &#x201C;ortholog.&#x201D;</p>
<p>Out of the 1670 orthologous genes identified, 1209 produced multiple of three alignments (<xref ref-type="supplementary-material" rid="TS2">Supplementary Table S2</xref>; see <xref ref-type="supplementary-material" rid="DS6">Supplementary File S1</xref> for the corresponding Fasta file showing the alignment of the concatenated sequences). Nucleotide alignments that were not multiple of three indicate that at least one sequence has a frameshift and thus those alignments were not used. For these analyses, Clustal Omega (<xref ref-type="bibr" rid="B39">Sievers et al., 2011</xref>) was used as the alignment algorithm and SEDA<sup><xref ref-type="fn" rid="footnote2">2</xref></sup> was used to select the multiple of three alignments. Therefore, the phylogenomic analysis was performed using the 1209 orthologous concatenated genes present in the selected strains and a Bayesian phylogenetic tree was obtained using MrBayes (<xref ref-type="bibr" rid="B36">Ronquist et al., 2012</xref>). The model of sequence evolution used was the General Time Reversible (GTR), allowing for among-site rate variation and a proportion of invariable sites. This was the selected model when using the Akaike information criterion (AIC), as implemented in jModelTest 2 (<xref ref-type="bibr" rid="B13">Darriba et al., 2012</xref>; a Docker image is available at see text footnote 1). Third codon positions were allowed to have a gamma distribution shape parameter different from that of first and second codon positions. Two independent runs of 1 000 000 Markov chain Monte Carlo generations with four chains each (one cold and three heated chains) were set up. The average standard deviation of split frequencies was below 0.000001. Moreover, the potential scale reduction factor for every parameter was about 1.00 showing that convergence has been achieved. Trees were sampled every 100th generation and the first 2500 samples were discarded (burn-in). The remaining trees were used to compute the Bayesian posterior probabilities of each clade of the consensus tree.</p>
<p>Genes identified in <italic>H</italic>. <italic>patelloides</italic> and another strain only, are listed in <xref ref-type="supplementary-material" rid="TS6">Supplementary Table S6</xref>. The corresponding Fasta file showing the alignment of the concatenated sequences can be found in <xref ref-type="supplementary-material" rid="DS7">Supplementary Files S2</xref>&#x2013;<xref ref-type="supplementary-material" rid="DS12">S7</xref>. The files have been used for the divergence analyses.</p>
<p>The number of non-synonymous changes per non-synonymous position (<italic>Ka</italic>) and the number of synonymous changes per synonymous position (<italic>Ks</italic>) values have been estimated using the DnaSP software (<xref ref-type="bibr" rid="B37">Rozas et al., 2017</xref>). The non-parametric Mann&#x2013;Whitney test was used in order to determine whether the two samples could have been obtained from the same <italic>Ks</italic> distribution. Tajima&#x2019;s relative rate tests were performed using the Mega 7 software (<xref ref-type="bibr" rid="B25">Kumar et al., 2016</xref>).</p>
</sec>
<sec id="S4.SS4">
<title>COG Categories Distribution</title>
<p>For each baeocyte-forming cyanobacterial strain, we examined how protein-coding genes are distributed in the various functional categories. Therefore, their distribution within the Clusters of Orthologous Groups (COG) was performed using the COGnitor tool (<xref ref-type="bibr" rid="B44">Tatusov et al., 1997</xref>), available at MicroScope platform (<xref ref-type="bibr" rid="B45">Vallenet et al., 2019</xref>).</p>
</sec>
<sec id="S4.SS5">
<title>Natural Product Biosynthetic Gene Clusters Analysis</title>
<p>Natural product biosynthetic gene clusters (BGCs) of <italic>H</italic>. <italic>patelloides</italic> were identified using the genome mining software antiSMASH v5.0 (<xref ref-type="bibr" rid="B6">Blin et al., 2019</xref>). Subsequently, the identified BGCs were analyzed and compared against available cyanobacterial genomes using the MicroScope platform (<xref ref-type="bibr" rid="B45">Vallenet et al., 2019</xref>).</p>
<p>BiG-SCAPE (Biosynthetic Gene Similarity Clustering and Prospecting Engine) software v1.0.0<sup><xref ref-type="fn" rid="footnote3">3</xref></sup> (<xref ref-type="bibr" rid="B31">Navarro-Mu&#x00F1;oz et al., 2020</xref>) was used to perform the biosynthetic gene cluster networking analyses. The dataset (input) included the 104 predicted BGCs (using antiSMASH v5.0) from <italic>H</italic>. <italic>patelloides</italic> and the baeocyte-forming strains studied here. The MIBiG parameter was set to include the MIBiG repository v1.4 (<xref ref-type="bibr" rid="B22">Kautsar et al., 2020</xref>). Analyses with different cut-offs of 0.3, 0.5, and 0.7 were performed. Phylogenetic trees inferring the evolutionary relationships of BGCs within each gene cluster family (GCF) provided by CORASON were generated within the BiG-SCAPE analysis.</p>
<p>The Antibiotic Resistance Target Seeker (ARTS) 2.0 tool (<xref ref-type="bibr" rid="B1">Alanjary et al., 2017</xref>) was used to target putative antibiotic producing BGCs (screening of putative resistance genes), using the default mode.</p>
</sec>
<sec id="S4.SS6">
<title>Hydrocarbon Extraction</title>
<p>Hydrocarbons were extracted as previously reported (<xref ref-type="bibr" rid="B42">Tan et al., 2011</xref>; <xref ref-type="bibr" rid="B48">Zhu et al., 2018</xref>) with some modifications. 30 mg of <italic>H. patelloides</italic> lyophilized biomass were resuspended in 10 mL of sterile deionized water, homogenized using a 20 mL tissue homogenizer, and lysed by sonication. The lysate was extracted using 10 mL of chloroform-methanol (v/v, 2:1) for 2 h at room temperature. Prior to extraction, 30 &#x03BC;g of eicosane (C<sub>20:0</sub>) was added to the cell lysate as an internal standard. The organic phase was separated by centrifugation (8 000 &#x00D7; <italic>g</italic>, 5 min), and the extract was dried under a nitrogen stream at 55&#x00B0;C. The residue containing the hydrocarbons was redissolved in 1 mL of n-hexane and analyzed as previously reported (<xref ref-type="bibr" rid="B48">Zhu et al., 2018</xref>).</p>
</sec>
<sec id="S4.SS7">
<title>Total Lipid Extraction and Methyl Esterification of Fatty Acids</title>
<p>Total lipids were extracted as previously reported (<xref ref-type="bibr" rid="B26">Lang et al., 2011</xref>; <xref ref-type="bibr" rid="B42">Tan et al., 2011</xref>; <xref ref-type="bibr" rid="B48">Zhu et al., 2018</xref>) with some modifications. 20 mg of <italic>H. patelloides</italic> lyophilized biomass were resuspended in 2 mL of sterile deionized water and homogenized using a 5 mL tissue homogenizer. Prior to extraction, 50 &#x03BC;g of nonadecanoic acid (C<sub>19:0</sub>) was added to the cell suspension as the internal standard. The samples were extracted using 4 mL of chloroform/methanol (v/v, 1:1), followed by homogenization using a vortex. The lower organic phase was separated by centrifugation (10 000 &#x00D7; <italic>g</italic>, 5 min), transferred into a 15 mL esterification tube, and dried under a nitrogen stream at 55&#x00B0;C. Then, 2 mL of 0.4 M KOH-methanol solution was added, and the mixture was incubated at 60&#x00B0;C for 1 h, allowing transesterification of lipid-bound fatty acids to the corresponding fatty acid methyl esters (FAMEs). Afterward, 4 mL of HCl/CH<sub>3</sub>OH (v/v, 1:9) were added to the mixture, and incubated at 60&#x00B0;C for 20 min. Finally, 2 mL of n-hexane and 3 mL of 5 M NaCl were added and gently mixed, and after keeping the mixture at room temperature for 20 min, the FAMEs (upper hexane phases) were transferred to sample vials and analyzed as previously reported (<xref ref-type="bibr" rid="B48">Zhu et al., 2018</xref>).</p>
</sec>
</sec>
<sec id="S5">
<title>Data Availability Statement</title>
<p>The datasets generated for this study can be found in the European Nucleotide Archive (ENA) under the study accession number <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="PRJEB28569">PRJEB28569</ext-link>.</p>
</sec>
<sec id="S6">
<title>Author Contributions</title>
<p>&#x00C2;B performed the experiments. &#x00C2;B, JV, CV, MG, PT, PL, and VR analyzed and interpreted the data. &#x00C2;B, JV, and CV assembled the genome and carried out the phylogenetic and genomic comparative analysis. TZ performed the hydrocarbon analysis. PT, MG, XL, and VV conceived and designed the study. All authors discussed, revised, and approved the final manuscript.</p>
</sec>
<sec id="conf1">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
</body>
<back>
<fn-group>
<fn fn-type="financial-disclosure">
<p><bold>Funding.</bold> This work was funded by the National Funds through FCT &#x2013; Funda&#x00E7;&#x00E3;o para a Ci&#x00EA;ncia e a Tecnologia, I.P., under the project UIDB/04293/2020, and by the project NORTE-01-0145-FEDER-000012, Structured Programme on Bioengineering Therapies for Infectious Diseases and Tissue Regeneration, supported by Norte Portugal Regional Operational Programme (NORTE 2020), under the PORTUGAL 2020 Partnership Agreement, through the European Regional Development Fund (ERDF). This work was also funded by the FCT grant SFRH/BPD/115571/2016 (&#x00C2;B) and by the FCT Projects UIDB/04423/2020 and UIDP/04423/2020. This work was also supported by the National Natural Science Foundation of China (Grant 31570068). The authors acknowledged the support and the use of resources of EMBRC-ERIC, specifically of the Portuguese infrastructure node of the European Marine Biological Resource Centre (EMBRC-PT) CIIMAR &#x2013; PINFRA/22121/2016 &#x2013; ALG-01-0145-FEDER-022121, financed by the European Regional Development Fund (ERDF) through COMPETE2020 &#x2013; Operational Programme for Competitiveness and Internationalization (POCI) and national funds through FCT/MCTES.</p>
</fn>
</fn-group>
<ack>
<p>We would like to thank the LABGeM (CEA/Genoscope and CNRS UMR 8030) and the France G&#x00E9;nomique National infrastructure (funded as part of Investissement d&#x2019;avenir program managed by Agence Nationale de la Recherche, contract ANR-10-INBS-09) for the support within the MicroScope annotation platform. We acknowledge Dr. Jorge Navarro Mu&#x00F1;oz and Dr. Marnix Medema for their helpful advice with BiG-SCAPE.</p>
</ack>
<sec id="S9" sec-type="supplementary material"><title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fmicb.2020.01527/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fmicb.2020.01527/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.PDF" id="DS1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_2.PDF" id="DS2" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_3.PDF" id="DS3" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_4.PDF" id="DS4" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_5.PDF" id="DS5" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_6.FASTA" id="DS6" mimetype="application/x-fasta" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_7.FASTA" id="DS7" mimetype="application/x-fasta" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_8.FASTA" id="DS8" mimetype="application/x-fasta" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_9.FASTA" id="DS9" mimetype="application/x-fasta" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_10.FASTA" id="DS10" mimetype="application/x-fasta" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_11.FASTA" id="DS11" mimetype="application/x-fasta" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_12.FASTA" id="DS12" mimetype="application/x-fasta" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_13.PDF" id="DS13" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_1.XLSX" id="TS1" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_2.XLSX" id="TS2" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_3.XLSX" id="TS3" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_4.DOCX" id="TS4" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_5.DOCX" id="TS5" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_6.XLSX" id="TS6" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_7.DOCX" id="TS7" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_8.XLSX" id="TS8" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_9.DOCX" id="TS9" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_10.XLSX" id="TS10" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_11.XLSX" id="TS11" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_12.XLSX" id="TS12" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_13.XLSX" id="TS13" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_14.XLSX" id="TS14" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_15.XLSX" id="TS15" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alanjary</surname> <given-names>M.</given-names></name> <name><surname>Kronmiller</surname> <given-names>B.</given-names></name> <name><surname>Adamek</surname> <given-names>M.</given-names></name> <name><surname>Blin</surname> <given-names>K.</given-names></name> <name><surname>Weber</surname> <given-names>T.</given-names></name> <name><surname>Huson</surname> <given-names>D.</given-names></name><etal/></person-group> (<year>2017</year>). <article-title>The Antibiotic Resistant Target Seeker (ARTS), an exploration engine for antibiotic cluster prioritization and novel drug target discovery.</article-title> <source><italic>Nucleic Acids Res</italic>.</source> <volume>45</volume> <fpage>W42</fpage>&#x2013;<lpage>W48</lpage>.</citation></ref>
<ref id="B2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Al-Thukair</surname> <given-names>A. A.</given-names></name></person-group> (<year>2011</year>). <article-title>Calculating boring rate of endolithic cyanobacteria <italic>Hyella immanis</italic> under laboratory conditions.</article-title> <source><italic>Int. Biodeter. Biodegr.</italic></source> <volume>65</volume> <fpage>664</fpage>&#x2013;<lpage>667</lpage>. <pub-id pub-id-type="doi">10.1016/j.ibiod.2011.03.009</pub-id></citation></ref>
<ref id="B3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Al&#x2212;Thukair</surname> <given-names>A. A.</given-names></name> <name><surname>Golubic</surname> <given-names>S.</given-names></name></person-group> (<year>1991</year>). <article-title>New endolithic cyanobacteria from the arabian gulf. I. <italic>Hyella immanis</italic> SP. NOV. 1.</article-title> <source><italic>J. Phycol.</italic></source> <volume>27</volume> <fpage>766</fpage>&#x2013;<lpage>780</lpage>. <pub-id pub-id-type="doi">10.1111/j.0022-3646.1991.00766.x</pub-id></citation></ref>
<ref id="B4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bankevich</surname> <given-names>A.</given-names></name> <name><surname>Nurk</surname> <given-names>S.</given-names></name> <name><surname>Antipov</surname> <given-names>D.</given-names></name> <name><surname>Gurevich</surname> <given-names>A. A.</given-names></name> <name><surname>Dvorkin</surname> <given-names>M.</given-names></name> <name><surname>Kulikov</surname> <given-names>A. S.</given-names></name><etal/></person-group> (<year>2012</year>). <article-title>SPAdes: a new genome assembly algorithm and its applications to single-cell sequencing.</article-title> <source><italic>J. Comput. Biol.</italic></source> <volume>19</volume> <fpage>455</fpage>&#x2013;<lpage>477</lpage>. <pub-id pub-id-type="doi">10.1089/cmb.2012.0021</pub-id> <pub-id pub-id-type="pmid">22506599</pub-id></citation></ref>
<ref id="B5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Benzerara</surname> <given-names>K.</given-names></name> <name><surname>Skouri-Panet</surname> <given-names>F.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>F&#x00E9;rard</surname> <given-names>C.</given-names></name> <name><surname>Gugger</surname> <given-names>M.</given-names></name> <name><surname>Laurent</surname> <given-names>T.</given-names></name><etal/></person-group> (<year>2014</year>). <article-title>Intracellular Ca-carbonate biomineralization is widespread in cyanobacteria.</article-title> <source><italic>P. Natl. Acad. Sci. U.S.A.</italic></source> <volume>111</volume> <fpage>10933</fpage>&#x2013;<lpage>10938</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1403510111</pub-id> <pub-id pub-id-type="pmid">25009182</pub-id></citation></ref>
<ref id="B6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Blin</surname> <given-names>K.</given-names></name> <name><surname>Shaw</surname> <given-names>S.</given-names></name> <name><surname>Steinke</surname> <given-names>K.</given-names></name> <name><surname>Villebro</surname> <given-names>R.</given-names></name> <name><surname>Ziemert</surname> <given-names>N.</given-names></name> <name><surname>Lee</surname> <given-names>S. Y.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>antiSMASH 5.0: updates to the secondary metabolite genome mining pipeline.</article-title> <source><italic>Nucleic Acids Res</italic>.</source> <volume>47</volume> <fpage>W81</fpage>&#x2013;<lpage>W87</lpage>.</citation></ref>
<ref id="B7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brito</surname> <given-names>&#x00C2;</given-names></name> <name><surname>Gaifem</surname> <given-names>J.</given-names></name> <name><surname>Ramos</surname> <given-names>V.</given-names></name> <name><surname>Glukhov</surname> <given-names>E.</given-names></name> <name><surname>Dorrestein</surname> <given-names>P. C.</given-names></name> <name><surname>Gerwick</surname> <given-names>W. H.</given-names></name><etal/></person-group> (<year>2015</year>). <article-title>Bioprospecting Portuguese Atlantic coast cyanobacteria for bioactive secondary metabolites reveals untapped chemodiversity.</article-title> <source><italic>Algal Res.</italic></source> <volume>9</volume> <fpage>218</fpage>&#x2013;<lpage>226</lpage>. <pub-id pub-id-type="doi">10.1016/j.algal.2015.03.016</pub-id></citation></ref>
<ref id="B8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brito</surname> <given-names>&#x00C2;</given-names></name> <name><surname>Ramos</surname> <given-names>V.</given-names></name> <name><surname>Mota</surname> <given-names>R.</given-names></name> <name><surname>Lima</surname> <given-names>S.</given-names></name> <name><surname>Santos</surname> <given-names>A.</given-names></name> <name><surname>Vieira</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2017</year>). <article-title>Description of new genera and species of marine cyanobacteria from the portuguese atlantic coast.</article-title> <source><italic>Mol. Phylogenet. Evol.</italic></source> <volume>111</volume> <fpage>18</fpage>&#x2013;<lpage>34</lpage>. <pub-id pub-id-type="doi">10.1016/j.ympev.2017.03.006</pub-id> <pub-id pub-id-type="pmid">28279808</pub-id></citation></ref>
<ref id="B9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brito</surname> <given-names>&#x00C2;</given-names></name> <name><surname>Ramos</surname> <given-names>V.</given-names></name> <name><surname>Seabra</surname> <given-names>R.</given-names></name> <name><surname>Santos</surname> <given-names>A.</given-names></name> <name><surname>Santos</surname> <given-names>C. L.</given-names></name> <name><surname>Lopo</surname> <given-names>M.</given-names></name><etal/></person-group> (<year>2012</year>). <article-title>Culture-dependent characterization of cyanobacterial diversity in the intertidal zones of the Portuguese coast: a polyphasic study.</article-title> <source><italic>Syst. Appl. Microbiol.</italic></source> <volume>35</volume> <fpage>110</fpage>&#x2013;<lpage>119</lpage>. <pub-id pub-id-type="doi">10.1016/j.syapm.2011.07.003</pub-id> <pub-id pub-id-type="pmid">22277323</pub-id></citation></ref>
<ref id="B10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Calteau</surname> <given-names>A.</given-names></name> <name><surname>Fewer</surname> <given-names>D. P.</given-names></name> <name><surname>Latifi</surname> <given-names>A.</given-names></name> <name><surname>Coursin</surname> <given-names>T.</given-names></name> <name><surname>Laurent</surname> <given-names>T.</given-names></name> <name><surname>Jokela</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2014</year>). <article-title>Phylum-wide comparative genomics unravel the diversity of secondary metabolism in cyanobacteria.</article-title> <source><italic>BMC Genomics</italic></source> <volume>15</volume>:<issue>977</issue>. <pub-id pub-id-type="doi">10.1186/1471-2164-15-977</pub-id> <pub-id pub-id-type="pmid">25404466</pub-id></citation></ref>
<ref id="B11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Castenholz</surname> <given-names>R. W.</given-names></name></person-group> (<year>2001</year>). &#x201C;<article-title>Phylum BX. cyanobacteria. oxygenic photosynthetic bacteria</article-title>,&#x201D; in <source><italic>Bergey&#x2019;s Manual of Systematic Bacteriology</italic></source>, <edition>2nd Edn</edition>, <role>eds</role> <person-group person-group-type="editor"><name><surname>Boone</surname> <given-names>D. R.</given-names></name> <name><surname>Castenholz</surname> <given-names>R. W.</given-names></name> <name><surname>Garrity</surname> <given-names>G.</given-names></name></person-group> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>473</fpage>&#x2013;<lpage>599</lpage>. <pub-id pub-id-type="doi">10.1007/978-0-387-21609-6_27</pub-id></citation></ref>
<ref id="B12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Coates</surname> <given-names>R. C.</given-names></name> <name><surname>Podell</surname> <given-names>S.</given-names></name> <name><surname>Korobeynikov</surname> <given-names>A.</given-names></name> <name><surname>Lapidus</surname> <given-names>A.</given-names></name> <name><surname>Pevzner</surname> <given-names>P.</given-names></name> <name><surname>Sherman</surname> <given-names>D. H.</given-names></name><etal/></person-group> (<year>2014</year>). <article-title>Characterization of cyanobacterial hydrocarbon composition and distribution of biosynthetic pathways.</article-title> <source><italic>PLoS One</italic></source> <volume>9</volume>:<issue>e85140</issue>. <pub-id pub-id-type="doi">10.1371/journal.pone.0085140</pub-id> <pub-id pub-id-type="pmid">24475038</pub-id></citation></ref>
<ref id="B13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Darriba</surname> <given-names>D.</given-names></name> <name><surname>Taboada</surname> <given-names>G. L.</given-names></name> <name><surname>Doallo</surname> <given-names>R.</given-names></name> <name><surname>Posada</surname> <given-names>D.</given-names></name></person-group> (<year>2012</year>). <article-title>jModelTest 2: more models, new heuristics and parallel computing.</article-title> <source><italic>Nat. Methods</italic></source> <volume>9</volume>:<issue>772</issue>. <pub-id pub-id-type="doi">10.1038/nmeth.2109</pub-id> <pub-id pub-id-type="pmid">22847109</pub-id></citation></ref>
<ref id="B14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Deruyter</surname> <given-names>Y. S.</given-names></name> <name><surname>Fromme</surname> <given-names>P.</given-names></name></person-group> (<year>2008</year>). &#x201C;<article-title>Molecular structure of the photosynthetic apparatus</article-title>,&#x201D; in <source><italic>The Cyanobacteria: Molecular biology, genomics and evolution</italic></source>, <role>eds</role> <person-group person-group-type="editor"><name><surname>Herrero</surname> <given-names>A.</given-names></name> <name><surname>Flores</surname> <given-names>E.</given-names></name></person-group> (<publisher-loc>Norfolk</publisher-loc>: <publisher-name>Caister Academic Press</publisher-name>), <fpage>217</fpage>&#x2013;<lpage>269</lpage>.</citation></ref>
<ref id="B15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>D&#x00ED;ez</surname> <given-names>B.</given-names></name> <name><surname>Nylander</surname> <given-names>J. A.</given-names></name> <name><surname>Ininbergs</surname> <given-names>K.</given-names></name> <name><surname>Dupont</surname> <given-names>C. L.</given-names></name> <name><surname>Allen</surname> <given-names>A. E.</given-names></name> <name><surname>Yooseph</surname> <given-names>S.</given-names></name><etal/></person-group> (<year>2016</year>). <article-title>Metagenomic analysis of the Indian ocean picocyanobacterial community: structure, potential function and evolution.</article-title> <source><italic>PLoS One</italic></source> <volume>11</volume>:<issue>e0155757</issue>. <pub-id pub-id-type="doi">10.1371/journal.pone.0155757</pub-id> <pub-id pub-id-type="pmid">27196065</pub-id></citation></ref>
<ref id="B16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dittmann</surname> <given-names>E.</given-names></name> <name><surname>Gugger</surname> <given-names>M.</given-names></name> <name><surname>Sivonen</surname> <given-names>K.</given-names></name> <name><surname>Fewer</surname> <given-names>D. P.</given-names></name></person-group> (<year>2015</year>). <article-title>Natural product biosynthetic diversity and comparative genomics of the cyanobacteria.</article-title> <source><italic>Trends Microbiol.</italic></source> <volume>23</volume> <fpage>642</fpage>&#x2013;<lpage>652</lpage>. <pub-id pub-id-type="doi">10.1016/j.tim.2015.07.008</pub-id> <pub-id pub-id-type="pmid">26433696</pub-id></citation></ref>
<ref id="B17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fischbach</surname> <given-names>M. A.</given-names></name> <name><surname>Walsh</surname> <given-names>C. T.</given-names></name> <name><surname>Clardy</surname> <given-names>J.</given-names></name></person-group> (<year>2009</year>). <article-title>The evolution of gene collectives: how natural selection drives chemical innovation.</article-title> <source><italic>Proc. Natl. Acad. Sci. U.S.A</italic>.</source> <volume>106</volume>:<issue>1679</issue>. <pub-id pub-id-type="doi">10.1073/pnas.0812594106</pub-id></citation></ref>
<ref id="B18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Flombaum</surname> <given-names>P.</given-names></name> <name><surname>Gallegos</surname> <given-names>J. L.</given-names></name> <name><surname>Gordillo</surname> <given-names>R. A.</given-names></name> <name><surname>Rinc&#x00F3;n</surname> <given-names>J.</given-names></name> <name><surname>Zabala</surname> <given-names>L. L.</given-names></name> <name><surname>Jiao</surname> <given-names>N.</given-names></name><etal/></person-group> (<year>2013</year>). <article-title>Present and future global distributions of the marine Cyanobacteria <italic>Prochlorococcus</italic> and <italic>Synechococcus</italic>.</article-title> <source><italic>Proc. Natl. Acad. Sci. U.S.A.</italic></source> <volume>110</volume> <fpage>9824</fpage>&#x2013;<lpage>9829</lpage>.</citation></ref>
<ref id="B19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Garcia-Pichel</surname> <given-names>F.</given-names></name> <name><surname>Belnap</surname> <given-names>J.</given-names></name> <name><surname>Neuer</surname> <given-names>S.</given-names></name> <name><surname>Schanz</surname> <given-names>F.</given-names></name></person-group> (<year>2003</year>). <article-title>Estimates of global cyanobacterial biomass and its distribution.</article-title> <source><italic>Algol. Stud.</italic></source> <volume>109</volume> <fpage>213</fpage>&#x2013;<lpage>227</lpage>. <pub-id pub-id-type="doi">10.1127/1864-1318/2003/0109-0213</pub-id></citation></ref>
<ref id="B20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Garcia-Pichel</surname> <given-names>F.</given-names></name> <name><surname>Ram&#x00ED;rez-Reinat</surname> <given-names>E.</given-names></name> <name><surname>Gao</surname> <given-names>Q.</given-names></name></person-group> (<year>2010</year>). <article-title>Microbial excavation of solid carbonates powered by P-type ATPase-mediated transcellular Ca<sup>2+</sup> transport.</article-title> <source><italic>Proc. Natl. Acad. Sci. U.S.A.</italic></source> <volume>107</volume> <fpage>21749</fpage>&#x2013;<lpage>21754</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1011884108</pub-id> <pub-id pub-id-type="pmid">21115827</pub-id></citation></ref>
<ref id="B21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Guida</surname> <given-names>B. S.</given-names></name> <name><surname>Garcia-Pichel</surname> <given-names>F.</given-names></name></person-group> (<year>2016</year>). <article-title>Extreme cellular adaptations and cell differentiation required by a cyanobacterium for carbonate excavation.</article-title> <source><italic>Proc. Natl. Acad. Sci. U.S.A.</italic></source> <volume>113</volume> <fpage>5712</fpage>&#x2013;<lpage>5717</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1524687113</pub-id> <pub-id pub-id-type="pmid">27140633</pub-id></citation></ref>
<ref id="B22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kautsar</surname> <given-names>S. A.</given-names></name> <name><surname>Blin</surname> <given-names>K.</given-names></name> <name><surname>Shaw</surname> <given-names>S.</given-names></name> <name><surname>Navarro-Mu&#x00F1;oz</surname> <given-names>J. C.</given-names></name> <name><surname>Terlouw</surname> <given-names>B. R.</given-names></name> <name><surname>van der Hooft</surname> <given-names>J. J.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>MIBiG 2.0: a repository for biosynthetic gene clusters of known function.</article-title> <source><italic>Nucleic Acids Res</italic>.</source> <volume>48</volume> <fpage>D454</fpage>&#x2013;<lpage>D458</lpage>.</citation></ref>
<ref id="B23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kleigrewe</surname> <given-names>K.</given-names></name> <name><surname>Almaliti</surname> <given-names>J.</given-names></name> <name><surname>Tian</surname> <given-names>I. Y.</given-names></name> <name><surname>Kinnel</surname> <given-names>R. B.</given-names></name> <name><surname>Korobeynikov</surname> <given-names>A.</given-names></name> <name><surname>Monroe</surname> <given-names>E. A.</given-names></name><etal/></person-group> (<year>2015</year>). <article-title>Combining mass spectrometric metabolic profiling with genomic analysis: a powerful approach for discovering natural products from cyanobacteria.</article-title> <source><italic>J. Nat. Prod.</italic></source> <volume>78</volume> <fpage>1671</fpage>&#x2013;<lpage>1682</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jnatprod.5b00301</pub-id> <pub-id pub-id-type="pmid">26149623</pub-id></citation></ref>
<ref id="B24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krawczyk</surname> <given-names>P. S.</given-names></name> <name><surname>Lipinski</surname> <given-names>L.</given-names></name> <name><surname>Dziembowski</surname> <given-names>A.</given-names></name></person-group> (<year>2018</year>). <article-title>PlasFlow: predicting plasmid sequences in metagenomic data using genome signatures.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>46</volume>:<issue>e35</issue>. <pub-id pub-id-type="doi">10.1093/nar/gkx1321</pub-id> <pub-id pub-id-type="pmid">29346586</pub-id></citation></ref>
<ref id="B25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kumar</surname> <given-names>S.</given-names></name> <name><surname>Stecher</surname> <given-names>G.</given-names></name> <name><surname>Tamura</surname> <given-names>K.</given-names></name></person-group> (<year>2016</year>). <article-title>MEGA7: molecular evolutionary genetics analysis version 7.0 for bigger datasets.</article-title> <source><italic>Mol. Biol. Evol.</italic></source> <volume>33</volume> <fpage>1870</fpage>&#x2013;<lpage>1874</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msw054</pub-id> <pub-id pub-id-type="pmid">27004904</pub-id></citation></ref>
<ref id="B26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lang</surname> <given-names>I.</given-names></name> <name><surname>Hodac</surname> <given-names>L.</given-names></name> <name><surname>Friedl</surname> <given-names>T.</given-names></name> <name><surname>Feussner</surname> <given-names>I.</given-names></name></person-group> (<year>2011</year>). <article-title>Fatty acid profiles and their distribution patterns in microalgae: a comprehensive analysis of more than 2000 strains from the SAG culture collection.</article-title> <source><italic>BMC Plant Biol.</italic></source> <volume>11</volume>:<issue>124</issue>. <pub-id pub-id-type="doi">10.1186/1471-2229-11-124</pub-id> <pub-id pub-id-type="pmid">21896160</pub-id></citation></ref>
<ref id="B27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mendez-Perez</surname> <given-names>D.</given-names></name> <name><surname>Begemann</surname> <given-names>M. B.</given-names></name> <name><surname>Pfleger</surname> <given-names>B. F.</given-names></name></person-group> (<year>2011</year>). <article-title>Modular synthase-encoding gene involved in &#x03B1;-olefin biosynthesis in <italic>Synechococcus</italic> sp. strain PCC 7002.</article-title> <source><italic>Appl. Environ. Microbiol.</italic></source> <volume>77</volume> <fpage>4264</fpage>&#x2013;<lpage>4267</lpage>. <pub-id pub-id-type="doi">10.1128/aem.00467-11</pub-id> <pub-id pub-id-type="pmid">21531827</pub-id></citation></ref>
<ref id="B28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mihali</surname> <given-names>T. K.</given-names></name> <name><surname>Kellmann</surname> <given-names>R.</given-names></name> <name><surname>Neilan</surname> <given-names>B. A.</given-names></name></person-group> (<year>2009</year>). <article-title>Characterisation of the paralytic shellfish toxin biosynthesis gene clusters in <italic>Anabaena circinalis</italic> AWQC131C and <italic>Aphanizomenon sp</italic>. NH-5.</article-title> <source><italic>BMC Biochem.</italic></source> <volume>10</volume>:<issue>8</issue>. <pub-id pub-id-type="doi">10.1186/1471-2091-10-8</pub-id> <pub-id pub-id-type="pmid">19331657</pub-id></citation></ref>
<ref id="B29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Moreno-Hagelsieb</surname> <given-names>G.</given-names></name> <name><surname>Latimer</surname> <given-names>K.</given-names></name></person-group> (<year>2008</year>). <article-title>Choosing BLAST options for better detection of orthologs as reciprocal best hits.</article-title> <source><italic>Bioinformatics</italic></source> <volume>24</volume> <fpage>319</fpage>&#x2013;<lpage>324</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btm585</pub-id> <pub-id pub-id-type="pmid">18042555</pub-id></citation></ref>
<ref id="B30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Moss</surname> <given-names>N. A.</given-names></name> <name><surname>Bertin</surname> <given-names>M. J.</given-names></name> <name><surname>Kleigrewe</surname> <given-names>K.</given-names></name> <name><surname>Le&#x00E3;o</surname> <given-names>T. F.</given-names></name> <name><surname>Gerwick</surname> <given-names>L.</given-names></name> <name><surname>Gerwick</surname> <given-names>W. H.</given-names></name></person-group> (<year>2016</year>). <article-title>Integrating mass spectrometry and genomics for cyanobacterial metabolite discovery.</article-title> <source><italic>J. Ind. Microbiol. Biotechnol.</italic></source> <volume>43</volume> <fpage>313</fpage>&#x2013;<lpage>324</lpage>. <pub-id pub-id-type="doi">10.1007/s10295-015-1705-7</pub-id> <pub-id pub-id-type="pmid">26578313</pub-id></citation></ref>
<ref id="B31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Navarro-Mu&#x00F1;oz</surname> <given-names>J. C.</given-names></name> <name><surname>Selem-Mojica</surname> <given-names>N.</given-names></name> <name><surname>Mullowney</surname> <given-names>M. W.</given-names></name> <name><surname>Kautsar</surname> <given-names>S.</given-names></name> <name><surname>Tryon</surname> <given-names>J. H.</given-names></name> <name><surname>Parkinson</surname> <given-names>E. I.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>A computational framework to explore large-scale biosynthetic diversity.</article-title> <source><italic>Nat. Chem. Biol.</italic></source> <volume>16</volume> <fpage>60</fpage>&#x2013;<lpage>68</lpage>.</citation></ref>
<ref id="B32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nunnery</surname> <given-names>J. K.</given-names></name> <name><surname>Mevers</surname> <given-names>E.</given-names></name> <name><surname>Gerwick</surname> <given-names>W. H.</given-names></name></person-group> (<year>2010</year>). <article-title>Biologically active secondary metabolites from marine cyanobacteria.</article-title> <source><italic>Curr. Opin. Biotechnol.</italic></source> <volume>21</volume> <fpage>787</fpage>&#x2013;<lpage>793</lpage>. <pub-id pub-id-type="doi">10.1016/j.copbio.2010.09.019</pub-id> <pub-id pub-id-type="pmid">21030245</pub-id></citation></ref>
<ref id="B33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Parks</surname> <given-names>D. H.</given-names></name> <name><surname>Imelfort</surname> <given-names>M.</given-names></name> <name><surname>Skennerton</surname> <given-names>C. T.</given-names></name> <name><surname>Hugenholtz</surname> <given-names>P.</given-names></name> <name><surname>Tyson</surname> <given-names>G. W.</given-names></name></person-group> (<year>2015</year>). <article-title>CheckM: assessing the quality of microbial genomes recovered from isolates, single cells, and metagenomes.</article-title> <source><italic>Genome Res.</italic></source> <volume>25</volume> <fpage>1043</fpage>&#x2013;<lpage>1055</lpage>. <pub-id pub-id-type="doi">10.1101/gr.186072.114</pub-id> <pub-id pub-id-type="pmid">25977477</pub-id></citation></ref>
<ref id="B34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ramos</surname> <given-names>V.</given-names></name> <name><surname>Morais</surname> <given-names>J.</given-names></name> <name><surname>Castelo-Branco</surname> <given-names>R.</given-names></name> <name><surname>Pinheiro</surname> <given-names>&#x00C2;</given-names></name> <name><surname>Martins</surname> <given-names>J.</given-names></name> <name><surname>Regueiras</surname> <given-names>A.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>Cyanobacterial diversity held in microbial biological resource centers as a biotechnological asset: the case study of the newly established LEGE culture collection.</article-title> <source><italic>J. Appl. Phycol.</italic></source> <volume>30</volume> <fpage>1437</fpage>&#x2013;<lpage>1451</lpage>. <pub-id pub-id-type="doi">10.1007/s10811-017-1369-y</pub-id> <pub-id pub-id-type="pmid">29899596</pub-id></citation></ref>
<ref id="B35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rippka</surname> <given-names>R.</given-names></name></person-group> (<year>1988</year>). &#x201C;<article-title>Isolation and purification of cyanobacteria</article-title>,&#x201D; in <source><italic>Method. Enzymol</italic></source>, <role>eds</role> <person-group person-group-type="editor"><name><surname>Packer</surname> <given-names>L.</given-names></name> <name><surname>Glazer</surname> <given-names>A. N.</given-names></name></person-group> (<publisher-loc>San Diego, CA</publisher-loc>: <publisher-name>Academic Press</publisher-name>), <fpage>3</fpage>&#x2013;<lpage>27</lpage>. <pub-id pub-id-type="doi">10.1016/0076-6879(88)67004-2</pub-id></citation></ref>
<ref id="B36"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ronquist</surname> <given-names>F.</given-names></name> <name><surname>Teslenko</surname> <given-names>M.</given-names></name> <name><surname>Van Der Mark</surname> <given-names>P.</given-names></name> <name><surname>Ayres</surname> <given-names>D. L.</given-names></name> <name><surname>Darling</surname> <given-names>A.</given-names></name> <name><surname>H&#x00F6;hna</surname> <given-names>S.</given-names></name><etal/></person-group> (<year>2012</year>). <article-title>MrBayes 3.2: efficient Bayesian phylogenetic inference and model choice across a large model space.</article-title> <source><italic>Syst. Biol.</italic></source> <volume>61</volume> <fpage>539</fpage>&#x2013;<lpage>542</lpage>. <pub-id pub-id-type="doi">10.1093/sysbio/sys029</pub-id> <pub-id pub-id-type="pmid">22357727</pub-id></citation></ref>
<ref id="B37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rozas</surname> <given-names>J.</given-names></name> <name><surname>Ferrer-Mata</surname> <given-names>A.</given-names></name> <name><surname>S&#x00E1;nchez-Delbarrio</surname> <given-names>J. C.</given-names></name> <name><surname>Guirao-Rico</surname> <given-names>S.</given-names></name> <name><surname>Librado</surname> <given-names>P.</given-names></name> <name><surname>Ramos-Onsins</surname> <given-names>S. E.</given-names></name><etal/></person-group> (<year>2017</year>). <article-title>DnaSP 6: DNA sequence polymorphism analysis of large data sets.</article-title> <source><italic>Mol. Biol. Evol.</italic></source> <volume>34</volume> <fpage>3299</fpage>&#x2013;<lpage>3302</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msx248</pub-id> <pub-id pub-id-type="pmid">29029172</pub-id></citation></ref>
<ref id="B38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shih</surname> <given-names>P. M.</given-names></name> <name><surname>Wu</surname> <given-names>D.</given-names></name> <name><surname>Latifi</surname> <given-names>A.</given-names></name> <name><surname>Axen</surname> <given-names>S. D.</given-names></name> <name><surname>Fewer</surname> <given-names>D. P.</given-names></name> <name><surname>Talla</surname> <given-names>E.</given-names></name><etal/></person-group> (<year>2013</year>). <article-title>Improving the coverage of the cyanobacterial phylum using diversity-driven genome sequencing.</article-title> <source><italic>Proc. Natl. Acad. Sci. U.S.A.</italic></source> <volume>110</volume> <fpage>1053</fpage>&#x2013;<lpage>1058</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1217107110</pub-id> <pub-id pub-id-type="pmid">23277585</pub-id></citation></ref>
<ref id="B39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sievers</surname> <given-names>F.</given-names></name> <name><surname>Wilm</surname> <given-names>A.</given-names></name> <name><surname>Dineen</surname> <given-names>D.</given-names></name> <name><surname>Gibson</surname> <given-names>T. J.</given-names></name> <name><surname>Karplus</surname> <given-names>K.</given-names></name> <name><surname>Li</surname> <given-names>W.</given-names></name><etal/></person-group> (<year>2011</year>). <article-title>Fast, scalable generation of high&#x2212;quality protein multiple sequence alignments using Clustal Omega.</article-title> <source><italic>Mol. Syst. Biol.</italic></source> <volume>7</volume>:<issue>539</issue>. <pub-id pub-id-type="doi">10.1038/msb.2011.75</pub-id> <pub-id pub-id-type="pmid">21988835</pub-id></citation></ref>
<ref id="B40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sivonen</surname> <given-names>K.</given-names></name> <name><surname>Leikoski</surname> <given-names>N.</given-names></name> <name><surname>Fewer</surname> <given-names>D. P.</given-names></name> <name><surname>Jokela</surname> <given-names>J.</given-names></name></person-group> (<year>2010</year>). <article-title>Cyanobactins&#x2013;ribosomal cyclic peptides produced by cyanobacteria.</article-title> <source><italic>Appl. Microbiol. Biotechnol.</italic></source> <volume>86</volume> <fpage>1213</fpage>&#x2013;<lpage>1225</lpage>. <pub-id pub-id-type="doi">10.1007/s00253-010-2482-x</pub-id> <pub-id pub-id-type="pmid">20195859</pub-id></citation></ref>
<ref id="B41"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tamagnini</surname> <given-names>P.</given-names></name> <name><surname>Troshina</surname> <given-names>O.</given-names></name> <name><surname>Oxelfelt</surname> <given-names>F.</given-names></name> <name><surname>Salema</surname> <given-names>R.</given-names></name> <name><surname>Lindblad</surname> <given-names>P.</given-names></name></person-group> (<year>1997</year>). <article-title>Hydrogenases in <italic>Nostoc</italic> sp. strain PCC 73102, a strain lacking a bidirectional enzyme.</article-title> <source><italic>Appl. Environ. Microbiol.</italic></source> <volume>63</volume> <fpage>1801</fpage>&#x2013;<lpage>1807</lpage>. <pub-id pub-id-type="doi">10.1128/aem.63.5.1801-1807.1997</pub-id></citation></ref>
<ref id="B42"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tan</surname> <given-names>X.</given-names></name> <name><surname>Yao</surname> <given-names>L.</given-names></name> <name><surname>Gao</surname> <given-names>Q.</given-names></name> <name><surname>Wang</surname> <given-names>W.</given-names></name> <name><surname>Qi</surname> <given-names>F.</given-names></name> <name><surname>Lu</surname> <given-names>X.</given-names></name></person-group> (<year>2011</year>). <article-title>Photosynthesis driven conversion of carbon dioxide to fatty alcohols and hydrocarbons in cyanobacteria.</article-title> <source><italic>Metab. Eng.</italic></source> <volume>13</volume> <fpage>169</fpage>&#x2013;<lpage>176</lpage>. <pub-id pub-id-type="doi">10.1016/j.ymben.2011.01.001</pub-id> <pub-id pub-id-type="pmid">21220042</pub-id></citation></ref>
<ref id="B43"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tang</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Mill&#x00E1;n-Agui&#x00F1;aga</surname> <given-names>N.</given-names></name> <name><surname>Zhang</surname> <given-names>J. J.</given-names></name> <name><surname>O&#x2019;Neill</surname> <given-names>E. C.</given-names></name> <name><surname>Ugalde</surname> <given-names>J. A.</given-names></name><etal/></person-group> (<year>2015</year>). <article-title>Identification of thiotetronic acid antibiotic biosynthetic pathways by target-directed genome mining.</article-title> <source><italic>ACS Chem. Biol.</italic></source> <volume>10</volume> <fpage>2841</fpage>&#x2013;<lpage>2849</lpage>. <pub-id pub-id-type="doi">10.1021/acschembio.5b00658</pub-id> <pub-id pub-id-type="pmid">26458099</pub-id></citation></ref>
<ref id="B44"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tatusov</surname> <given-names>R. L.</given-names></name> <name><surname>Koonin</surname> <given-names>E. V.</given-names></name> <name><surname>Lipman</surname> <given-names>D. J.</given-names></name></person-group> (<year>1997</year>). <article-title>A genomic perspective on protein families.</article-title> <source><italic>Science</italic></source> <volume>278</volume> <fpage>631</fpage>&#x2013;<lpage>637</lpage>. <pub-id pub-id-type="doi">10.1126/science.278.5338.631</pub-id> <pub-id pub-id-type="pmid">9381173</pub-id></citation></ref>
<ref id="B45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vallenet</surname> <given-names>D.</given-names></name> <name><surname>Calteau</surname> <given-names>A.</given-names></name> <name><surname>Dubois</surname> <given-names>M.</given-names></name> <name><surname>Amours</surname> <given-names>P.</given-names></name> <name><surname>Bazin</surname> <given-names>A.</given-names></name> <name><surname>Beuvin</surname> <given-names>M.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>MicroScope: an integrated platform for the annotation and exploration of microbial gene functions through genomic, pangenomic and metabolic comparative analysis.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>48</volume> <fpage>D579</fpage>&#x2013;<lpage>D589</lpage>.</citation></ref>
<ref id="B46"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>Y.-W.</given-names></name> <name><surname>Simmons</surname> <given-names>B. A.</given-names></name> <name><surname>Singer</surname> <given-names>S. W.</given-names></name></person-group> (<year>2016</year>). <article-title>MaxBin 2.0: an automated binning algorithm to recover genomes from multiple metagenomic datasets.</article-title> <source><italic>Bioinformatics</italic></source> <volume>32</volume> <fpage>605</fpage>&#x2013;<lpage>607</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btv638</pub-id> <pub-id pub-id-type="pmid">26515820</pub-id></citation></ref>
<ref id="B47"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zehr</surname> <given-names>J. P.</given-names></name></person-group> (<year>2011</year>). <article-title>Nitrogen fixation by marine cyanobacteria.</article-title> <source><italic>Trends Microbiol.</italic></source> <volume>19</volume> <fpage>162</fpage>&#x2013;<lpage>173</lpage>. <pub-id pub-id-type="doi">10.1016/j.tim.2010.12.004</pub-id> <pub-id pub-id-type="pmid">21227699</pub-id></citation></ref>
<ref id="B48"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>T.</given-names></name> <name><surname>Scalvenzi</surname> <given-names>T.</given-names></name> <name><surname>Sassoon</surname> <given-names>N.</given-names></name> <name><surname>Lu</surname> <given-names>X.</given-names></name> <name><surname>Gugger</surname> <given-names>M.</given-names></name></person-group> (<year>2018</year>). <article-title>Terminal olefin profiles and phylogenetic analyses of olefin synthases in diversified cyanobacterial species.</article-title> <source><italic>Appl. Environ. Microbiol.</italic></source> <volume>84</volume> <fpage>e425</fpage>&#x2013;<lpage>e418</lpage>.</citation></ref>
</ref-list>
<fn-group>
<fn id="footnote1">
<label>1</label>
<p><ext-link ext-link-type="uri" xlink:href="https://pegi3s.github.io/dockerfiles/">https://pegi3s.github.io/dockerfiles/</ext-link></p></fn>
<fn id="footnote2">
<label>2</label>
<p><ext-link ext-link-type="uri" xlink:href="https://www.sing-group.org/seda/">https://www.sing-group.org/seda/</ext-link></p></fn>
<fn id="footnote3">
<label>3</label>
<p><ext-link ext-link-type="uri" xlink:href="https://git.wageningenur.nl/medema-group/BiG-SCAPE">https://git.wageningenur.nl/medema-group/BiG-SCAPE</ext-link></p></fn>
</fn-group>
</back>
</article>
