<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="methods-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Microbiol.</journal-id>
<journal-title>Frontiers in Microbiology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Microbiol.</abbrev-journal-title>
<issn pub-type="epub">1664-302X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmicb.2021.764058</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Microbiology</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>CANT-HYD: A Curated Database of Phylogeny-Derived Hidden Markov Models for Annotation of Marker Genes Involved in Hydrocarbon Degradation</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Khot</surname> <given-names>Varada</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x2020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1435311/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Zorz</surname> <given-names>Jackie</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x2020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/462758/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Gittins</surname> <given-names>Daniel A.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1495306/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Chakraborty</surname> <given-names>Anirban</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x2021;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/278145/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Bell</surname> <given-names>Emma</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/566204/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Bautista</surname> <given-names>Mar&#x00ED;a A.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1522395/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Paquette</surname> <given-names>Alexandre J.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1495277/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Hawley</surname> <given-names>Alyse K.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x2021;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/383274/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Novotnik</surname> <given-names>Breda</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/656271/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Hubert</surname> <given-names>Casey R. J.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/30798/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Strous</surname> <given-names>Marc</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/18013/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Bhatnagar</surname> <given-names>Srijak</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x2021;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1452705/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Energy Bioengineering and Geomicrobiology Group, Department of Geoscience, University of Calgary</institution>, <addr-line>Calgary, AB</addr-line>, <country>Canada</country></aff>
<aff id="aff2"><sup>2</sup><institution>Energy Bioengineering and Geomicrobiology Group, Department of Biological Sciences, University of Calgary</institution>, <addr-line>Calgary, AB</addr-line>, <country>Canada</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Nicole Buan, University of Nebraska-Lincoln, United States</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Lauren M. Lui, Lawrence Berkeley National Laboratory, United States; Yanni Sun, City University of Hong Kong, Hong Kong SAR, China</p></fn>
<corresp id="c001">&#x002A;Correspondence: Srijak Bhatnagar, <email>sbhatnagar@athabascau.ca</email></corresp>
<fn fn-type="equal" id="fn001"><p><sup>&#x2020;</sup>These authors have contributed equally to this work and share first authorship</p></fn>
<fn fn-type="present-address" id="fn002"><p><sup>&#x2021;</sup>Present address: Anirban Chakraborty, Department of Biological Sciences, Idaho State University, Pocatello, ID, United States; Alyse K. Hawley, School of Engineering, University of British Columbia Okanagan, Kelowna, BC, Canada; Srijak Bhatnagar, Faculty of Science and Technology, Athabasca University, Athabasca, AB, Canada</p></fn>
<fn fn-type="other" id="fn004"><p>This article was submitted to Microbial Physiology and Metabolism, a section of the journal Frontiers in Microbiology</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>07</day>
<month>01</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2021</year>
</pub-date>
<volume>12</volume>
<elocation-id>764058</elocation-id>
<history>
<date date-type="received">
<day>24</day>
<month>08</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>08</day>
<month>12</month>
<year>2021</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2022 Khot, Zorz, Gittins, Chakraborty, Bell, Bautista, Paquette, Hawley, Novotnik, Hubert, Strous and Bhatnagar.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Khot, Zorz, Gittins, Chakraborty, Bell, Bautista, Paquette, Hawley, Novotnik, Hubert, Strous and Bhatnagar</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Many pathways for hydrocarbon degradation have been discovered, yet there are no dedicated tools to identify and predict the hydrocarbon degradation potential of microbial genomes and metagenomes. Here we present the Calgary approach to ANnoTating HYDrocarbon degradation genes (CANT-HYD), a database of 37 HMMs of marker genes involved in anaerobic and aerobic degradation pathways of aliphatic and aromatic hydrocarbons. Using this database, we identify understudied or overlooked hydrocarbon degradation potential in many phyla. We also demonstrate its application in analyzing high-throughput sequence data by predicting hydrocarbon utilization in large metagenomic datasets from diverse environments. CANT-HYD is available at <ext-link ext-link-type="uri" xlink:href="https://github.com/dgittins/CANT-HYD-HydrocarbonBiodegradation">https://github.com/dgittins/CANT-HYD-HydrocarbonBiodegradation</ext-link>.</p>
</abstract>
<kwd-group>
<kwd>hydrocarbon degradation</kwd>
<kwd>Marker genes</kwd>
<kwd>Hidden Markov Models</kwd>
<kwd>gene annotation</kwd>
<kwd>hydrocarbon cycling</kwd>
</kwd-group>
<contract-sponsor id="cn001">Genome Canada<named-content content-type="fundref-id">10.13039/100008762</named-content></contract-sponsor>
<counts>
<fig-count count="7"/>
<table-count count="1"/>
<equation-count count="0"/>
<ref-count count="72"/>
<page-count count="15"/>
<word-count count="8248"/>
</counts>
</article-meta>
</front>
<body>
<sec id="S1" sec-type="intro">
<title>Introduction</title>
<p>Hydrocarbons are diverse compounds consisting of carbon and hydrogen atoms that differ in size, structure, and reactivity. They can be the product of geological processes as well as produced biogenically by organisms in all domains of life (<xref ref-type="bibr" rid="B63">Tornabene et al., 1969</xref>; <xref ref-type="bibr" rid="B38">Lerdau et al., 1997</xref>; <xref ref-type="bibr" rid="B37">Lea-Smith et al., 2015</xref>). Assessing hydrocarbon use by microorganisms, as a source of carbon and/or energy, is important for evaluating the consequences of hydrocarbon presence or contamination (<xref ref-type="bibr" rid="B5">Atlas and Hazen, 2011</xref>), understanding the global carbon cycle (<xref ref-type="bibr" rid="B24">Gonz&#x00E1;lez-Gaya et al., 2019</xref>), and for industrial applications, such as the synthesis of biocatalysts (<xref ref-type="bibr" rid="B54">Prier and Kosjek, 2019</xref>). Degradation of hydrocarbon molecules is kinetically challenging due to the chemical inertness of the organic C&#x2013;H bond, and when present, the stability of aromatic ring structures (<xref ref-type="bibr" rid="B56">Rabus et al., 2016</xref>). Microorganisms employ a range of enzymes to use hydrocarbons (<xref ref-type="bibr" rid="B56">Rabus et al., 2016</xref>; <xref ref-type="bibr" rid="B69">Xu et al., 2018</xref>) in oxic and anoxic conditions. Catabolism of these hydrocarbons is coupled with reduction of terminal electron acceptors such as oxygen, nitrate, sulfate, and iron or <italic>via</italic> syntrophy with methanogens (<xref ref-type="bibr" rid="B71">Zhang et al., 2019</xref>).</p>
<p>The discovery of hydrocarbon degrading microorganisms has traditionally relied on cultivation in the laboratory using hydrocarbon substrates (<xref ref-type="bibr" rid="B58">Rueter et al., 1994</xref>; <xref ref-type="bibr" rid="B36">Kniemeyer et al., 2007</xref>). Successful cultivation preceded the identification of genes involved in hydrocarbon metabolism with techniques such as gene knockouts, protein expression analyses, and gene sequencing (<xref ref-type="bibr" rid="B59">Schneiker et al., 2006</xref>; <xref ref-type="bibr" rid="B65">Wang and Shao, 2014</xref>; <xref ref-type="bibr" rid="B25">Gregson et al., 2018</xref>; <xref ref-type="bibr" rid="B66">Wang et al., 2018</xref>; <xref ref-type="bibr" rid="B43">Liu et al., 2019</xref>; <xref ref-type="bibr" rid="B41">Li et al., 2020</xref>). These studies are crucial for providing fundamental knowledge on the ever-growing diversity of hydrocarbon degrading microorganisms as well as uncovering new degradation pathways. The recent exponential rise in sequence data and the consequential increase in known microbial diversity have provided new opportunities to explore hydrocarbon degradation potential in diverse environments and uncultured microorganisms. One approach for exploring sequence data is to annotate genes using Hidden Markov Model (HMM). HMMs are trained on the multiple sequence alignments of amino acid sequences and produce position-specific scores and penalties when searching query sequences. HMMs have better sensitivity and recall for identifying homologs of conserved protein domains, compared to conventional pairwise alignment tools such as blastp (<xref ref-type="bibr" rid="B45">McGinnis and Madden, 2004</xref>), which use a position-independent scoring matrix (<xref ref-type="bibr" rid="B18">Eddy, 2004</xref>). Detection of metabolic potential in whole genomes or metagenomic datasets is generally accomplished using functional annotation tools aided by HMM databases such as KEGG (<xref ref-type="bibr" rid="B32">Kanehisa and Goto, 2000</xref>) and Pfam (<xref ref-type="bibr" rid="B23">Finn et al., 2014</xref>). While these large databases can confidently identify central metabolic and other well studied pathways, specific HMMs and tools for accurate annotation of catalytic genes in hydrocarbon degradation pathways are currently lacking. Genes involved in hydrocarbon degradation can share sequence similarity to genes from other metabolic pathways and consequently, are often misannotated (<xref ref-type="bibr" rid="B11">Callaghan et al., 2008</xref>; <xref ref-type="bibr" rid="B35">Khelifi et al., 2014</xref>). Hence, there is a need for a purpose-built tool for the accurate detection of hydrocarbon degradation pathways in sequence data.</p>
<p>Here we present the Calgary approach to ANnoTating HYDrocarbon degradation genes (CANT-HYD), a database of 37 HMMs designed for the identification and annotation of marker genes that are critical for the aerobic and anaerobic degradation of alkane and aromatic hydrocarbons. CANT-HYD is tested and validated against 72 genomes of known hydrocarbon degrading bacteria, representing a broad spectrum of hydrocarbon metabolism. Using these validated HMMs, over 30,000 representative genomes covering the entire bacterial and archaeal tree of life are analyzed to identify hydrocarbon degrading microorganisms. Forty-one publicly available metagenomes from diverse environments are also analyzed using CANT-HYD to explore hydrocarbon degradation potential in diverse environments. Lastly, we compare the performance of CANT-HYD HMMs to their counterparts from eggNOG, Pfam, and KEGG Orthology (KO) databases.</p>
</sec>
<sec id="S2" sec-type="methods">
<title>Methods</title>
<sec id="S2.SS1">
<title>Selection and Clustering of Archetype Reference Sequences</title>
<p>Enzymes involved in the activation of hydrocarbon substrates in aerobic and anaerobic hydrocarbon degradation pathways of aliphatic and aromatic compounds were identified through a literature search (<xref ref-type="fig" rid="F1">Figure 1</xref>). Amino acid sequences encoding the catalytic subunits of these enzymes were obtained from Genbank and were classified as either &#x201C;experimentally verified&#x201D; or &#x201C;putative.&#x201D; The &#x201C;experimentally verified&#x201D; sequences refer to amino acid sequences from published studies with experimental proof of the intended function. Experimental proof consisted of gene cloning or protein purification and corresponding enzyme assays, or gene knockout studies. Gene sequences labeled &#x201C;putative&#x201D; refer to sequences with strong evidence of function but lacking these definitive prerequisite analyses. Putative sequences often originated from isolates or enrichment cultures where there is evidence of hydrocarbon degradation or genomic and/or proteomic evidence for the enzyme responsible. The resulting curated 105 amino acid sequences from 53 different species (<xref ref-type="supplementary-material" rid="TS1">Supplementary Table 1</xref>) are referred henceforth as &#x201C;archetype&#x201D; as they represent the gene sequence pattern for a verifiable hydrocarbon degradation function. Amino acid sequences were clustered into homologous groups based on &#x2265;20% amino acid identity as determined by blastp v2.9.0 (<xref ref-type="bibr" rid="B45">McGinnis and Madden, 2004</xref>) to place related archetype sequences on the same phylogenetic tree and reduce the downstream computational resource requirements. A loose grouping as carried out here is unlikely to affect the final HMM, as manual curation and pruning of phylogenetic trees in downstream processing took this clustering into account. Archetype sequences that did not cluster at 20% were either manually added into homologous groups of similar function or left as singletons.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption><p>Hydrocarbon degradation reactions covered by CANT-HYD. Reactions for the degradation of alkanes through aerobic <bold>(A)</bold> and anaerobic pathways <bold>(B)</bold> and degradation of aromatic hydrocarbons through aerobic <bold>(C)</bold> and anaerobic <bold>(D)</bold> pathways.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-12-764058-g001.tif"/>
</fig>
</sec>
<sec id="S2.SS2">
<title>Homology Search to Obtain Sequences Similar to Archetypes</title>
<p>Amino acid sequences sharing sequence homology to the 105 archetype genes were recruited from the NCBI non-redundant (nr) protein database using a Diamond homology search (<xref ref-type="bibr" rid="B10">Buchfink et al., 2014</xref>). All hits with query coverage &#x2265;70% and e-value &#x2264;10<sup>&#x2013;4</sup> were retained as putatively phylogenetically related sequences and added to the query&#x2019;s homologous group. The sequences of homologous groups were dereplicated, followed by clustering at 98% amino acid identity using the USEARCH v9.0.2 <italic>derep_fulllength</italic> and <italic>cluster_fast</italic> commands (<xref ref-type="bibr" rid="B21">Edgar, 2010</xref>). The resulting sequences from the Diamond homology search are referred to as the &#x201C;DIAMOND sequences&#x201D; database and were used in downstream steps of HMM construction, and to generate cutoff scores for the HMMs.</p>
</sec>
<sec id="S2.SS3">
<title>Grouping Genes With Similar Functions Using Phylogenetic Analysis</title>
<p>A multiple sequence alignment was generated for each clustered (98%) homologous group using MUSCLE v3.8.31 (<xref ref-type="bibr" rid="B20">Edgar, 2004</xref>). The alignments were used to create maximum-likelihood trees using FastTreeMP v2.1 (<xref ref-type="bibr" rid="B53">Price et al., 2010</xref>) with the parameters &#x2013;<italic>pseudo</italic> and <italic>-spr 4</italic>. Trees were manually inspected using iTOL v5.6.3 (<xref ref-type="bibr" rid="B39">Letunic and Bork, 2019</xref>) or Dendroscope v3.6.3 (<xref ref-type="bibr" rid="B30">Huson et al., 2007</xref>) to identify monophyletic clades of genes containing experimentally verified archetype sequences (<xref ref-type="supplementary-material" rid="DS2">Supplementary Figure 1</xref>). These monophyletic clades were easily identifiable as archetype clustering was carried out at a low threshold (on &#x2265;20% amino acid identity) and thus a substrate-specific set of seed sequences was extracted from each clade. In some instances, where a clear monophyletic distinction was lacking, a broader function-specific set of seed sequences were extracted (e.g., MAH_alpha group includes TcbA, IpbA, BnzA, and BphA).</p>
</sec>
<sec id="S2.SS4">
<title>Processing Homologous Groups With &#x003E;5,000 Sequences</title>
<p>As some group sizes were in the order of 10<sup>5</sup> sequences and the computational requirement for aligning sequences grows exponentially with every added sequence, a nested clustering and phylogenetic pruning approach was implemented to overcome computational challenges for groups &#x003E;5,000 sequences. The homologous group was first clustered at a lower identity (e.g., 50%) to reduce the size of the group, followed by alignment, phylogenetic reconstruction, and phylogenetic neighborhood pruning as described above. This process reduced the sequence search space around the archetype sequences by pruning the phylogenetic trees at a higher identity threshold. Because each sequence in a clustered group represents a group of sequences, for the selected pruned neighborhood, the clustered sequences were placed back in. Then the process of pruning the phylogenetic neighborhood of the reference sequence(s) was iterated using a higher clustering identity (e.g., 70%, 90%, etc.) until the prune group was &#x2264;5,000 sequences or the clustering identity was raised to 98% (<xref ref-type="fig" rid="F2">Figure 2</xref>), at which point the seed sequences for HMM creation were selected as described above.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption><p>Workflow of the process underlying the creation of the CANT-HYD HMM Database.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-12-764058-g002.tif"/>
</fig>
</sec>
<sec id="S2.SS5">
<title>HMM Creation and Determination of Cutoff Scores</title>
<p>Because HMMs are sensitive to the alignment and position of each amino acid, the seed sequences that are now orders of magnitude smaller than the homologous groups they were derived from were realigned using a sensitive aligner, Clustal-Omega v1.2.4 (<xref ref-type="bibr" rid="B60">Sievers et al., 2011</xref>) followed by manual inspection of the alignment using Jalview v2.11.1.4 (<xref ref-type="bibr" rid="B67">Waterhouse et al., 2009</xref>), and generation of HMMs using the <italic>hmmbuild</italic> command of HMMER v3.2.1 (<xref ref-type="bibr" rid="B19">Eddy, 2011</xref>). The HMMs were used to search the archetype reference sequences and &#x2018;DIAMOND Sequences&#x2019; database (<xref ref-type="fig" rid="F2">Figure 2</xref>) using <italic>hmmsearch</italic> of HMMER v3.2.1 (<xref ref-type="bibr" rid="B19">Eddy, 2011</xref>). The domain scores of the hits to each HMM were plotted to visualize the frequency distribution pattern of the scores (<xref ref-type="supplementary-material" rid="DS2">Supplementary Figure 2</xref>). A &#x201C;trusted&#x201D; and a &#x201C;noise&#x201D; cutoff was chosen for each HMM using these score distributions. The trusted cutoff is the domain score above which a sequence can be confidently annotated for the function, as all experimentally verified genes used for the HMM scored above this cutoff. The noise cutoff was chosen to exclude genes that were predicted to have a different function. Thus, any hits scoring below the noise cutoffs are expected to <italic>not</italic> carry out the function represented by the HMM. <xref ref-type="supplementary-material" rid="TS2">Supplementary Table 2</xref> includes information on genes that are the closest phylogenetic relatives to the archetype sequences of CANT-HYD HMMs.</p>
</sec>
<sec id="S2.SS6">
<title>Validation of CANT-HYD HMMs Using Genomes of Known Hydrocarbon Degraders</title>
<p>Seventy-two genomes of microorganisms with published experimental evidence of an ability to degrade hydrocarbons were downloaded from GenBank and RefSeq (<xref ref-type="bibr" rid="B48">O&#x2019;Leary et al., 2016</xref>) and categorized by the type of substrate and respiration (<xref ref-type="supplementary-material" rid="TS3">Supplementary Table 3</xref>). If the exact strain was not available, its closest relative from the Genome Taxonomy Database (GTDB) was chosen. For example, <italic>Aromatoleum aromaticum</italic> EbN1 anaerobically degrades aromatic compounds (<xref ref-type="bibr" rid="B68">W&#x00F6;hlbrand et al., 2007</xref>). The genomes were searched using CANT-HYD HMMs and the resulting gene annotations, scoring above the trusted cutoff, were compared to the established degradation capability of the organism. If a gene hit multiple HMMs above the confidence threshold, it was assigned to the highest scoring HMM.</p>
</sec>
<sec id="S2.SS7">
<title>Analysis of GTDB Genomes to Identify Potentially Novel Hydrocarbon Degrading Bacteria</title>
<p>The GTDB database (05-RS95 17th July 2020) (<xref ref-type="bibr" rid="B51">Parks et al., 2020</xref>) of representative bacterial and archaeal genomes was downloaded and searched using the CANT-HYD HMMs. For further investigation, gene sequences from cyanobacterial genomes with hits to LadA beta (above the noise cutoff) were combined with archetype reference sequences of long-chain alkane monooxygenases (LadA-alpha, LadA-beta, and LadB). The combined sequences were then clustered at 70% amino acid identity using USEARCH v9.0.2132_i86linux64 (<xref ref-type="bibr" rid="B21">Edgar, 2010</xref>) <italic>cluster_fast.</italic> Representative sequences of each cluster were then aligned using Muscle v3.8.31 (<xref ref-type="bibr" rid="B20">Edgar, 2004</xref>), followed by a maximum-likelihood phylogenetic reconstruction using FastTreeMP v2.1 (<xref ref-type="bibr" rid="B53">Price et al., 2010</xref>) (<xref ref-type="supplementary-material" rid="DS1">Supplementary Data Sheet 1</xref>).</p>
</sec>
<sec id="S2.SS8">
<title>Analysis of Diverse Metagenomes Using CANT-HYD</title>
<p>Metagenomes representing diverse environments such as petroleum reservoirs (<xref ref-type="bibr" rid="B28">Hu et al., 2016</xref>; <xref ref-type="bibr" rid="B47">Nie et al., 2016</xref>; <xref ref-type="bibr" rid="B42">Liu et al., 2018</xref>; <xref ref-type="bibr" rid="B13">Christman et al., 2020</xref>), oil spill experimental microcosms (<xref ref-type="bibr" rid="B62">Tan et al., 2015</xref>; <xref ref-type="bibr" rid="B15">Dombrowski et al., 2016</xref>), marine systems (<xref ref-type="bibr" rid="B49">Orellana et al., 2017</xref>; <xref ref-type="bibr" rid="B64">Tully et al., 2018</xref>; <xref ref-type="bibr" rid="B16">Dong et al., 2019</xref>, <xref ref-type="bibr" rid="B17">2020</xref>), host-associated microbiomes (<xref ref-type="bibr" rid="B22">Feigelman et al., 2017</xref>; <xref ref-type="bibr" rid="B27">Herman et al., 2020</xref>; <xref ref-type="bibr" rid="B6">Avila-Maga&#x00F1;a et al., 2021</xref>), and other environments (<xref ref-type="bibr" rid="B70">Yao et al., 2017</xref>; <xref ref-type="bibr" rid="B72">Zorz et al., 2019</xref>), were downloaded either as unassembled raw data from the NCBI SRA or as predicted gene sequences from the JGI Genome Portal (<xref ref-type="supplementary-material" rid="TS4">Supplementary Table 4</xref>). Raw reads from unassembled metagenomes were filtered using BBDuk<sup><xref ref-type="fn" rid="footnote1">1</xref></sup> for a minimum quality of 15 and a minimum read length of 150 bp. Reads passing quality control were assembled using MEGAHIT (<xref ref-type="bibr" rid="B40">Li et al., 2015</xref>) with default parameters, followed by gene calling by Prodigal v2.6.3 (<xref ref-type="bibr" rid="B31">Hyatt et al., 2010</xref>) with the metagenomic option (<italic>-p meta</italic>). The amino acid sequences of predicted genes were searched against the CANT-HYD database using the <italic>hmmsearch</italic> command of HMMER v3.2.1 (<xref ref-type="bibr" rid="B19">Eddy, 2011</xref>) and only hits scoring above the noise cutoff for each HMM were visualized. Hit count for each metagenome was normalized by the total number of predicted genes.</p>
</sec>
<sec id="S2.SS9">
<title>Comparison of CANT-HYD HMMs to Existing HMM Databases</title>
<p>The CANT-HYD HMMs were compared to equivalent HMMs from Pfam (<xref ref-type="bibr" rid="B7">Bateman et al., 2000</xref>), eggNOG (<xref ref-type="bibr" rid="B29">Huerta-Cepas et al., 2019</xref>) and KO (<xref ref-type="bibr" rid="B33">Kanehisa et al., 2016</xref>). Archetype sequences used to build CANT-HYD HMMs were annotated using eggNOG mapper (<xref ref-type="bibr" rid="B12">Cantalapiedra et al., 2021</xref>) with default parameters to identify and retrieve the closest eggNOG, Pfam and Kofam HMMs (<xref ref-type="supplementary-material" rid="TS5">Supplementary Table 5</xref>). These equivalent HMMs were used to annotate the isolate genomes, and hydrocarbon enrichment and host-associated metagenomes using <italic>hmmsearch</italic> (<xref ref-type="bibr" rid="B19">Eddy, 2011</xref>). Because suggested cutoffs were not included with eggNOG, Pfam, or KO database, an e-value cutoff of 10<sup>&#x2013;50</sup> was used to filter results.</p>
</sec>
</sec>
<sec id="S3" sec-type="results|discussion">
<title>Results and Discussion</title>
<sec id="S3.SS1">
<title>Validation of CANT-HYD HMMs</title>
<p>Genomes of 72 microorganisms with experimental evidence of hydrocarbon degradation were analyzed with the CANT-HYD HMMs for validation. For 62 out of 72 organisms, gene predictions using CANT-HYD were consistent with experimental data (<xref ref-type="fig" rid="F3">Figure 3</xref>). Of the remaining 10 genomes, two genomes had hits with a score between the noise and trusted cutoffs, and eight genomes lacked hits above the noise cutoff. In a few instances, genomes were not available for the exact strain and a GTDB representative genome was used in their place. Although GTDB representatives share 95% average nucleotide identity with the cluster they represent, hydrocarbon degradation genes may be missing or different. Genomes of three organisms isolated on phenanthrene, chlorophenol and benzoate did not yield hits to any CANT-HYD HMMs. Although these substrates can be degraded by dioxygenases which share homology with mono- and polyaromatic ring hydroxylating dioxygenases, the lack of hits, even below the noise cutoff, indicates that the three organisms potentially use alternative metabolic pathways which were not covered by CANT-HYD. CANT-HYD predicted additional or unreported hydrocarbon substrate degradation capabilities for 16 genomes. For example, genes for toluene-2-monooxygenase (Tom) and toluene-4-monooxygenase (Tmo), and monoaromatic dioxygenase (MAH_alpha and MAH_beta) were found in the genome of <italic>Pseudoxanthomonas spadix</italic> BD-a59, a well-known benzene, toluene, ethylbenzene, and xylene (BTEX) degrader (<xref ref-type="bibr" rid="B14">Chun et al., 2010</xref>). Anaerobic hydrocarbon degradation genes were only detected in the genomes of anaerobes, further showing the prediction accuracy of CANT-HYD HMMs. Every HMM had at least one hit, except for the bacterial benzene carboxylase (AbcA_1) and toluene-benzene monooxygenase beta subunit (TmoB_BmoB). Anaerobic benzene degradation <italic>via</italic> benzene carboxylase (AbcA_1) has been identified in a single uncultured organism belonging to <italic>Clostridia</italic>, for which a genome is currently unavailable (<xref ref-type="bibr" rid="B2">Abu Laban et al., 2010</xref>). Toluene-benzene monooxygenase beta subunit (TmoB_BmoB) was found adjacent to TmoA_BmoA gene on the <italic>Pseudoxanthomonas spadix</italic> BD-a59 genome with a score above the noise cutoff, which indicates that it likely is a TmoB_BmoB gene divergent from the seed sequences that were used to make the HMM. Overall, these results show that CANT-HYD reliably identifies hydrocarbon degradation marker genes and can thus be used to predict the hydrocarbon degradation potential of genomes and in metagenomes.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption><p>Concordance of CANT-HYD annotations of genomes and their experimentally verified hydrocarbon degradation activity. Each bubble represents hits above the confidence threshold from experientially verified hydrocarbon degrading genomes (x-axis) plotted against the HMMs of the CANT-HYD database (y-axis). The size of the bubble represents number of unique hits in the genome. The genomes (x-axis) and the CANT-HYD HMMs (y-axis) are organized by the hydrocarbon substrate and respiration. A complete list of isolate genomes, their Genbank accession, and their published hydrocarbon degradation capability can be found in <xref ref-type="supplementary-material" rid="TS3">Supplementary Table 3</xref>.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-12-764058-g003.tif"/>
</fig>
</sec>
<sec id="S3.SS2">
<title>Diversity of Hydrocarbon Degrading Bacteria and Archaea</title>
<p>A large number of bacterial (30,238) and archaeal (1,672) genomes from the Genome Taxonomy Database (GTDB) were searched against the CANT-HYD HMMs (<xref ref-type="bibr" rid="B51">Parks et al., 2020</xref>). In total, 4,601 representative genomes from 18 bacterial phyla, had at least one hit to an HMM that scored higher than the trusted cutoff (<xref ref-type="supplementary-material" rid="TS2">Supplementary Table 2</xref>), and in total, 5,845 genomes from 24 bacterial phyla had hits to at least one CANT-HYD HMM above the noise cutoff (<xref ref-type="fig" rid="F4">Figure 4</xref>). HMM hits from diverse bacterial phyla demonstrate the widespread potential for hydrocarbon degradation across bacteria (<xref ref-type="fig" rid="F4">Figures 4A,B</xref>).</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption><p>Diversity of hydrocarbon-degrading bacteria and archaea. GTDB representative genomes with CANT-HYD HMM hits shown on panel <bold>(A)</bold> the GTDB phylogenomic tree collapsed at phylum level. Phyla with genomes containing genes with a CANT-HYD HMM score greater than the noise cutoff are shown in blue. The corresponding bar and number indicates the percentage of the representative genomes in that phylum containing at least one hit to a CANT-HYD HMM. For example, Abyssubacteria contained two genome representatives in GTDB, and one of the genomes contained a high confidence match to a CANT-HYD HMM; thus, the bar shows 50%. <bold>(B)</bold> Distribution and number of hits for each phylum in GTDB across the CANT-HYD HMMs.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-12-764058-g004.tif"/>
</fig>
<p>Many of these phyla contain no cultured representatives, and therefore annotation tools like CANT-HYD become important for offering clues about their metabolic potential. For instance, the potential for hydrocarbon degradation was found in genomes of poorly represented phyla including Abyssubacteria, Tectomicrobia, and Eremiobacterota (<xref ref-type="fig" rid="F4">Figure 4B</xref> and <xref ref-type="supplementary-material" rid="DS2">Supplementary Figure 4</xref>). The phylum Abyssubacteria, often found in association with subsurface and hydrocarbon contaminated environments (<xref ref-type="bibr" rid="B46">Momper et al., 2018</xref>), had hits to anaerobic alkane degradation (AhyA). Eremiobacterota (formerly WPS-2), previously found in hydrocarbon enrichment cultures (<xref ref-type="bibr" rid="B57">Ramadass et al., 2018</xref>), had three members with aerobic aromatic hydrocarbon degradation potential (NdoB, MAH_alpha, MAH_beta). Two genomes from Tectomicrobia had high HMM scores to enzymes responsible for the aerobic degradation of monoaromatics (MAH_Beta), polyaromatics (NdoB), and long-chain alkanes (LadA_alpha and CYP153). There is currently no literature associating Tectomicrobia with hydrocarbon containing environments, however, high confidence matches to CANT-HYD HMMs suggest that they may have a previously unidentified role in the aerobic metabolism of a range of hydrocarbons.</p>
<sec id="S3.SS2.SSS1">
<title>Hydrocarbon Degradation in Archaea</title>
<p>Archaea contained fewer hydrocarbon degradation genes compared to bacteria. Only four genomes, all from the phylum Halobacteriota, had HMM hits above the trusted cutoff. Another 102 genomes, also from Halobacteriota, had at least one HMM hit above the noise cutoff (<xref ref-type="supplementary-material" rid="DS2">Supplementary Figure 3</xref>). The phylum Halobacteriota (formerly a member of phylum Euryarchaeota) is known to contain halophilic hydrocarbon degrading species (<xref ref-type="bibr" rid="B4">Al-Mailem et al., 2010</xref>; <xref ref-type="bibr" rid="B50">Oren, 2019</xref>). The identification of only a few archaeal hydrocarbon degraders may be due to either a lower representation of sequenced archaeal genomes, or an increased phylogenetic distance of archaeal hydrocarbon degradation genes to the primarily bacterial sequences that have been experimentally validated. Additionally, methanotrophy, the most well studied archaeal hydrocarbon degradation, is not covered by CANT-HYD. As more experimental evidence of archaeal genes emerge, the annotation of archaeal hydrocarbon degradation will improve.</p>
</sec>
<sec id="S3.SS2.SSS2">
<title>Cyanobacteria as Alkane Degraders</title>
<p>Thirty-six genes from 29 cyanobacterial GTDB representative genomes, mostly from the family Nostocaceae and the genera <italic>Nostoc</italic> and <italic>Aulosira</italic>, were predicted to contain LadA beta, a long-chain alkane monooxygenase (<xref ref-type="fig" rid="F5">Figure 5</xref>). LadA beta is one of the three LadA-type long-chain alkane monooxygenase enzymes with experimental evidence of long-chain alkane degradation (<xref ref-type="bibr" rid="B9">Boonmak et al., 2014</xref>). Phylogenetically, these cyanobacterial genes were related to the experimentally verified LadA beta sequence from <italic>Geobacillus thermoleovorans</italic> (BAM76372.1), suggesting that the genes perform a similar role in their photosynthetic hosts (<xref ref-type="fig" rid="F5">Figure 5A</xref>). Many cyanobacterial species produce long-chain alkanes, potentially at globally relevant levels (<xref ref-type="bibr" rid="B37">Lea-Smith et al., 2015</xref>; <xref ref-type="bibr" rid="B44">Love et al., 2021</xref>), and alkane degradation has been observed in microbial communities with abundant cyanobacteria (<xref ref-type="bibr" rid="B1">Abed, 2010</xref>). Thus far however, it has been inconclusive whether the alkane degradation is performed by cyanobacteria or other heterotrophic community members, and if the cyanobacteria are responsible, which degradation pathways they utilize (<xref ref-type="bibr" rid="B3">Al-Hasan et al., 1998</xref>; <xref ref-type="bibr" rid="B55">Qiao et al., 2020</xref>). Strong hits to LadA beta suggest that some cyanobacterial species have the metabolic potential for long-chain alkane degradation <italic>via</italic> LadA, although experiments are needed to confirm if this genetic potential is realized.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption><p>Detection of long-chain alkane monooxygenase (Lad) family of genes in Cyanobacteria. <bold>(A)</bold> Phylogenetic tree of Lad sequences including a cluster of 102 sequences detected in Cyanobacteria genomes by CANT-HYD (green). <bold>(B)</bold> Distribution of scores for LadA beta HMM hits in Cyanobacteria genomes. Original tree file for is available as <xref ref-type="supplementary-material" rid="DS1">Supplementary Data Sheet 1</xref>.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-12-764058-g005.tif"/>
</fig>
</sec>
</sec>
<sec id="S3.SS3">
<title>Hydrocarbon Degradation in Diverse Environments</title>
<p>The CANT-HYD HMMs were used to search for hydrocarbon degradation potential in 41 metagenomes, representing diverse environments including hydrocarbon degrading enrichment cultures, petroleum reservoirs, oceans, host-associated microbiomes, alkaline lakes, and hot springs. Hydrocarbon degradation genes were detected in all these environments, except for the host-associated microbiomes, which are presumed to have a limited presence of hydrocarbons (<xref ref-type="fig" rid="F6">Figure 6</xref>).</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption><p>Metagenomes from diverse environments analyzed with the CANT-HYD HMMs. CANT-HYD HMM hits (scores &#x2265; noise cutoff) (y-axis) to each metagenome (x-axis) normalized to per million protein coding genes. The size of the bubble depicts proportion of hydrocarbon degrading genes to total genic content of the metagenome. The metagenomes (x-axis) are grouped under Enrichments (hydrocarbon-degrading enrichment cultures), Petroleum Reservoirs (produced well water from petroleum reservoirs from Alaska, Gulf of Mexico, Jiangsu and Qinghai, China, and Medicine Hat, Alberta), Oceans (cold seeps, TARA surface seawater), Host-Assoc (host-associated microbiomes), Other: (other environments). The CAN-HYD HMMs are grouped by hydrocarbon substrate (alkane and aromatic) and respiration (aerobic and anaerobic). See <xref ref-type="supplementary-material" rid="TS4">Supplementary Table 4</xref> for details regarding these metagenomes.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-12-764058-g006.tif"/>
</fig>
<p>When normalized for total gene content, the highest proportion of hydrocarbon degradation genes were detected in the metagenomes of hydrocarbon-degrading enrichments. Genes for aerobic alkane degradation, such as AlkB, CYP153, and AlmA, were the most widely detected in this dataset. Many marker genes for aerobic hydrocarbon degradation were found ubiquitously in metagenomes from ocean surface waters, while genes for anaerobic hydrocarbon degradation were largely detected in metagenomes sequenced from anoxic habitats such as petroleum reservoirs, subseafloor sediments, and anoxic hydrocarbon degrading microcosms. Further, some degradation enzymes covered by the CANT-HYD HMMs yielded no hits, namely butane monooxygenases (pBmoA, pBmoB, and pBmoC, and sBmoX, sBmoY, and sBmoZ), and benzene and naphthalene carboxylases (AbcA_1 and K27540). These HMMs were made with less than five seed sequences as they had only a few close relatives (&#x2265;50% sequence identity) in the public nucleotide database at the time of this work.</p>
<sec id="S3.SS3.SSS1">
<title>Enrichment Cultures of Hydrocarbon Degrading Microorganisms</title>
<p>Genes from the glycyl radical enzyme family (Ass/Bss/Nms) for anaerobic hydrocarbon degradation were identified in two out of the three anoxic cultures, namely the toluene and short-chain-alkane enrichments (<xref ref-type="bibr" rid="B62">Tan et al., 2015</xref>; <xref ref-type="fig" rid="F6">Figure 6</xref>). The number of genes identified in this study were half of those reported in the original, which were also annotated using custom HMMs (<xref ref-type="table" rid="T1">Table 1</xref>). Observed discrepancies include the detection of partial gene sequences in the metagenomic assemblies (<xref ref-type="table" rid="T1">Table 1</xref>: denoted with an &#x002A;) (<xref ref-type="bibr" rid="B62">Tan et al., 2015</xref>). Partial gene sequence matches to the CANT-HYD Ass/Bss/Nms HMMs scored below the noise cutoffs and did not pass the threshold. Partial hits to HMMs scoring below the noise cutoffs cannot be reliably annotated as they may be a partial sequence of a related gene that encodes an enzyme with a different function. In this case, the partial hits for genes encoding for glycyl radical enzymes (Ass/Bss/Nms) can be functionally similar (<xref ref-type="supplementary-material" rid="DS2">Supplementary Figure 1</xref>) to and share high sequence homology with pyruvate formate lyase. While CANT-HYD will not perform optimally with unassembled, partially or poorly assembled data, the noise and trusted cutoffs are recommended for a low false-positive rate, by filtering out partial genes that risk being misannotated.</p>
<table-wrap position="float" id="T1">
<label>TABLE 1</label>
<caption><p>Number of AssA, BssA, and NmsA genes detected in this analysis compared to the original study (<xref ref-type="bibr" rid="B62">Tan et al., 2015</xref>).</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<thead>
<tr>
<td valign="top" align="left">Substrate</td>
<td valign="top" align="center" colspan="2">Short-chain alkanes</td>
<td valign="top" align="center" colspan="2">Toluene</td>
<td valign="top" align="center" colspan="2">Naphtha</td>
</tr>
<tr>
<td/>
<td valign="top" colspan="6"><hr/></td>
</tr>
<tr>
<td valign="top" align="left">Genes</td>
<td valign="top" align="center">Original study</td>
<td valign="top" align="center">CANT-HYD</td>
<td valign="top" align="center">Original study</td>
<td valign="top" align="center">CANT-HYD</td>
<td valign="top" align="center">Original study</td>
<td valign="top" align="center">CANT-HYD</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><bold>AssA</bold></td>
<td valign="top" align="center">4 + 1<xref ref-type="table-fn" rid="t1fns1">&#x002A;</xref></td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">1<xref ref-type="table-fn" rid="t1fns1">&#x002A;</xref></td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">1<xref ref-type="table-fn" rid="t1fns1">&#x002A;</xref></td>
<td valign="top" align="center">&#x2013;</td>
</tr>
<tr>
<td valign="top" align="left"><bold>BssA</bold></td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1 + 2<xref ref-type="table-fn" rid="t1fns1">&#x002A;</xref></td>
<td valign="top" align="center">&#x2013;</td>
</tr>
<tr>
<td valign="top" align="left"><bold>NmsA</bold></td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">1<xref ref-type="table-fn" rid="t1fns1">&#x002A;</xref></td>
<td valign="top" align="center">&#x2013;</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="t1fns1"><p><italic>Asterisk (&#x002A;) indicates partial gene homologs.</italic></p></fn>
</table-wrap-foot>
</table-wrap>
<p>Several genes for aerobic hydrocarbon degradation were identified in oxic cultures inoculated with the Deepwater Horizon oil plume and enriched on hexadecane, naphthalene, and phenanthrene as hydrocarbon substrates (<xref ref-type="fig" rid="F6">Figure 6</xref>). Genes for the aerobic degradation of medium and long-chain alkanes (CYP153, AlkB, and AlmA_GroupI) were detected in the hexadecane and naphthalene enrichments, while a variety of genes for aromatic hydrocarbon degradation were detected in the oxic naphthalene and phenanthrene enrichment cultures.</p>
</sec>
<sec id="S3.SS3.SSS2">
<title>Petroleum Reservoirs and Hydrocarbon Biodegradation</title>
<p>Multiple marker genes were identified in all metagenomes from petroleum reservoirs, except for four reservoirs that were either at a high temperature (&#x003E;80&#x00B0;C) or were deep subsurface (<xref ref-type="fig" rid="F6">Figure 6</xref>). Produced water metagenomes from petroleum reservoirs in Alaska (wells SB1, SB2, K1, K2, I1, and I2) were found to contain only anaerobic hydrocarbon degradation genes in agreement with the associated study (<xref ref-type="bibr" rid="B28">Hu et al., 2016</xref>). CANT-HYD further detected the presence of a putative benzene carboxylase gene (AbcA_2) in the metagenomes from two of the oil wells (K2 and I2), which was not reported in the original study. These oil reservoirs were reported to contain complete and partial genomes of a sulfide-producing archaeon, <italic>Archaeoglobus</italic>, and our results indicate that it can potentially metabolize benzene anaerobically.</p>
<p>Similar observation of an <italic>Archaeoglobus</italic> metagenome-assembled-genome (MAG) and its association with a benzene carboxylase (AbcA_2) gene comes from the metagenome of well W2 from the Jiangsu Oil Reservoir, China (<xref ref-type="bibr" rid="B42">Liu et al., 2018</xref>). Although the original study found alkyl succinate synthase genes (Ass) in an <italic>Archaeoglobus</italic> MAG, in this study, genes for the glycyl radical family of enzymes from these metagenomes and the <italic>Archaeoblogus</italic> genome scored below the noise cutoff. Any hits below the noise cutoff cannot be reliably annotated automatically and therefore require manual curation, such as using gene phylogeny to differentiate between glycyl radical enzyme and pyruvate formate lyase. A study by <xref ref-type="bibr" rid="B35">Khelifi et al. (2014)</xref> also shows evidence for anaerobic long-chain alkane degradation by <italic>Archaeoglobus</italic>, however, the genes identified as responsible for this metabolism share low sequence homology with bacterial alkyl succinate synthase alpha subunit and will not be annotated by the CANT-HYD AssA HMM. Therefore, a separate HMM for archaeal alkyl succinate synthase alpha subunit would be required when strong experimental evidence for these genes becomes available. Several genes for aerobic alkane and monoaromatic hydrocarbon degradation (<xref ref-type="fig" rid="F6">Figure 6</xref>) were also identified in the Well W2 metagenome, which were not originally reported. These findings highlight the utility of CANT-HYD, which can search for a comprehensive suite of hydrocarbon degradation markers, independent of <italic>a priori</italic> knowledge of the system.</p>
</sec>
<sec id="S3.SS3.SSS3">
<title>Widespread Hydrocarbon Degradation Potential in Global Surface Seawaters</title>
<p>Widespread potential for aerobic hydrocarbon degradation was detected in the surface seawater metagenomes collected by the TARA Oceans survey (<xref ref-type="bibr" rid="B34">Karsenti et al., 2011</xref>; <xref ref-type="bibr" rid="B64">Tully et al., 2018</xref>) and other studies (<xref ref-type="bibr" rid="B49">Orellana et al., 2017</xref>). Predicted hydrocarbon metabolism was largely driven by medium- and long-chain alkane hydroxylases, and ring-hydroxylating dioxygenases. This pervasive metabolic capability in the global ocean surface could be the result of biogenic alkanes synthesized by cyanobacteria (<xref ref-type="bibr" rid="B37">Lea-Smith et al., 2015</xref>) or the accumulation of polyaromatic hydrocarbons by other ocean phytoplankton, resulting in a &#x201C;cryptic hydrocarbon cycle&#x201D; (<xref ref-type="bibr" rid="B8">Binark et al., 2000</xref>; <xref ref-type="bibr" rid="B44">Love et al., 2021</xref>). An ocean metagenome from the Ross Sea, Antarctica had an exceptionally high abundance of marker genes involved in polyaromatic hydrocarbon degradation. The Ross Sea is well-known for its seasonal algal blooms and rapid carbon turnover (<xref ref-type="bibr" rid="B61">Smith et al., 2000</xref>; <xref ref-type="bibr" rid="B52">Peloquin and Smith, 2007</xref>), which have been associated with biogenic alkanes, polyaromatic hydrocarbons, and PAH degraders (<xref ref-type="bibr" rid="B26">Gutierrez et al., 2011</xref>; <xref ref-type="bibr" rid="B37">Lea-Smith et al., 2015</xref>; <xref ref-type="bibr" rid="B44">Love et al., 2021</xref>). Overall, our findings support the recent experimental evidence of a marine hydrocarbon cycle (<xref ref-type="bibr" rid="B44">Love et al., 2021</xref>).</p>
</sec>
<sec id="S3.SS3.SSS4">
<title>Comparison of CANT-HYD HMMs to Existing HMM Databases</title>
<p>The annotation performance of CANT-HYD HMMs was compared to HMMs from Pfam, KO, and eggNOG databases. As the purpose-built CANT-HYD database is curated for hydrocarbon degradation genes, it assigned more specific annotations than Pfam, eggNOG and KO databases. Pfam and eggNOG often annotated genes as the broad protein family to which the hydrocarbon degradation genes belong. Examples of these generic descriptions included &#x201C;monooxygenase&#x201D; for long-chain alkane monooxygenase (LadA), and &#x201C;pyruvate-formate-lyase like protein&#x201D; for the AssA, BssA, and NmsA genes (<xref ref-type="supplementary-material" rid="TS5">Supplementary Table 5</xref>). While these descriptions are broadly accurate, they do not give sufficient information about the gene for strong functional inference. The KO database assigned more specific annotations of the enzyme function compared to eggNOG and Pfam, but not necessarily to the same level of substrate specificity as the CANT-HYD HMMs. The KO database also missed annotations to more recently discovered hydrocarbon degradation genes such as naphthalene carboxylase (K27540), and long-chain alkane monooxygenase beta (LadA beta) (<xref ref-type="supplementary-material" rid="TS5">Supplementary Table 5</xref>).</p>
<p>The equivalent HMMs from the three databases were compared with CANT-HYD HMMs using genomes of 62 experimentally verified hydrocarbon degrading isolates and 10 metagenomes from hydrocarbon enrichments and host associated microbiomes. This search resulted in the 1000s of hits to from the Pfam, eggNOG and KO database and hence to remove spurious matches and increase confidence in annotations, only hits with an e-value below 10<sup>&#x2013;50</sup> were retained for analysis (<xref ref-type="fig" rid="F7">Figure 7</xref> and <xref ref-type="supplementary-material" rid="TS5">Supplementary Table 5</xref>). After filtering, Pfam, eggNOG and KO still reported &#x003E;1,000 hits for 62 genomes, while the total number of genes identified by CANT-HYD above the trusted cutoff was only 190. Furthermore, as previously mentioned that not all HMMs of expected specific metabolisms were found in these public databases, a large proportion of hits were to HMMs describing broad protein families such as &#x201C;cytochrome p450,&#x201D; &#x201C;Pyr_redox_2,&#x201D; and &#x201C;monooxygenase.&#x201D; A search of the metagenomes resulted in a total of 925, 525, and 463 genes identified by Pfam, KO, and eggNOG HMMs, respectively. In contrast, only 50 genes were identified in the same dataset using the noise cutoff of the CANT-HYD HMMs. These discrepancies in identified genes in genomes and metagenomes originates from the differences in the HMM targets of the databases. The lack of specificity of Pfam, eggNOG, and some KO is likely to detect genes that are related to hydrocarbon degradation genes, but not involved in hydrocarbon degradation. For the same reasons, while all other databases produced hits in host-associated metagenomes, CANT-HYD did not (<xref ref-type="fig" rid="F7">Figure 7</xref>). Together, these comparisons highlight the usefulness and accuracy of CANT-HYD to identify and annotate specific hydrocarbon metabolic potential by using curated cutoffs for HMMs designed specifically for hydrocarbon degradation marker genes.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption><p>Comparison of hydrocarbon metabolism prediction in metagenomes. Hits to aromatic hydrocarbon <bold>(A)</bold> and alkane <bold>(B)</bold> metabolism HMMs (y-axis) from CANT-HYD (red), eggNOG (green), KO (teal), and Pfam (purple) to each metagenome (x-axis) normalized to per million coding genes. The metagenomes (x-axis) are grouped under Enrichment Cultures or Host associated microbiome. The enrichment culture metagenomes representing microbial community grown on under aerobic conditions on hexadecane (Hex), naphthalene (Nap), or phenanthrene (Phen) or under anaerobic conditions on short-chain alkanes (SCA), Toluene (Tol), or Naphtha (Nha). The host-associated microbiomes are from child gut (CG), Coral (Cor), or Koumiss (Kou). The size of the bubble depicts proportion of hydrocarbon degrading genes to total genic content of the metagenome. See <xref ref-type="supplementary-material" rid="TS5">Supplementary Table 5</xref> for details regarding the HMMs and the underlying data.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-12-764058-g007.tif"/>
</fig>
</sec>
</sec>
</sec>
<sec id="S4" sec-type="conclusion">
<title>Conclusion</title>
<p>Here, we describe CANT-HYD, an HMM database of marker genes for hydrocarbon degradation. These phylogenetically informed HMMs accurately identify over 37 genes relevant to aerobic and anaerobic metabolisms of aliphatic and aromatic hydrocarbons in genomes and metagenomes. Each CANT-HYD HMM includes a manually curated trusted and noise cutoff score for automated reliable detection of these hydrocarbon degradation marker genes. To the best of our knowledge, CANT-HYD is the first dedicated tool for annotation of hydrocarbon degradation genes in genomes and metagenomes. We demonstrate the use of CANT-HYD as an exploratory tool by surveying all genomes in GTDB (30,238 bacterial and 1,672 archaeal), as well as several large metagenomic datasets. We uncovered the potential for long-chain alkane degradation in some cyanobacterial genomes and identified widespread potential for aerobic hydrocarbon degradation in global ocean surface waters, supporting a recently discovered marine hydrocarbon cycle. The comparison to other publicly available HMMs highlights the need for a curated HMM database for specific and precise annotation of hydrocarbon degradation genes and large-scale detection of hydrocarbon degrading capabilities in genomes and metagenomes.</p>
</sec>
<sec id="S5" sec-type="data-availability">
<title>Data Availability Statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="DS1">Supplementary Material</xref>, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="S6">
<title>Author Contributions</title>
<p>VK, JZ, DAG, AC, EB, MAB, AJP, AKH, BN, and SB carried out the literature review and sequence data searching. VK, JZ, DAG, AC, EB, MAB, AJP, and SB made and validated the HMMs. VK, JZ, DAG, AC, EB, and SB wrote the manuscript. VK, JZ, DAG, AC, MAB, and SB performed the genomic and metagenomic data analyses. VK, JZ, AC, and SB made the figures. BN and MS conceived the study. DAG created and maintained the GitHub archive. All authors contributed toward methodology development and editing and reviewing the manuscript.</p>
</sec>
<sec id="conf1" sec-type="COI-statement">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="pudiscl1" sec-type="disclaimer">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<sec id="S7" sec-type="funding-information">
<title>Funding</title>
<p>This work was supported by funds from Genome Canada to <italic>GENICE&#x2014;The microbial genomics for oil spill preparedness in the Canadian Arctic</italic>, to CRJH and MS. We acknowledge support from the Canada First Research Excellence Fund to BN and AJP and from the Government of Alberta to VK, JZ, and MS. Additional support for JZ, AJP, and AKH was provided by Natural Sciences and Engineering Research Council of Canada (NSERC).</p>
</sec>
<ack>
<p>We would like to thank Lisa Gieg, Gerrit Voordouw, and Muhe Diao for providing valuable insights into hydrocarbon metabolism. We would also like to thank Dongshan An, Gerrit Voordouw, Daniel Colman, Eric Boyd, Monica Orellana, Viridiana Avila-Maga&#x00F1;a, and Monica Medina for some of the metagenomic data analyzed here. We would like to extend our thanks to Xiaoli Dong for help with access and use of the computational servers. We thank the members of Energy Bioengineering and Geomicrobiology group and countless other people for their moral support during this &#x201C;hackathon&#x201D; that lasted more than a year.</p>
</ack>
<sec id="S9" sec-type="supplementary-material">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fmicb.2021.764058/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fmicb.2021.764058/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.zip" id="DS1" mimetype="application/zip" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_2.docx" id="DS2" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_3.docx" id="DS3" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_1.xlsx" id="TS1" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_2.xlsx" id="TS2" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_3.xlsx" id="TS3" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_4.xlsx" id="TS4" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_5.xlsx" id="TS5" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Abed</surname> <given-names>R. M. M.</given-names></name></person-group> (<year>2010</year>). <article-title>Interaction between cyanobacteria and aerobic heterotrophic bacteria in the degradation of hydrocarbons.</article-title> <source><italic>Int. Biodeterior. Biodegradation</italic></source> <volume>64</volume> <fpage>58</fpage>&#x2013;<lpage>64</lpage>. <pub-id pub-id-type="doi">10.1016/j.ibiod.2009.10.008</pub-id></citation></ref>
<ref id="B2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Abu Laban</surname> <given-names>N.</given-names></name> <name><surname>Selesi</surname> <given-names>D.</given-names></name> <name><surname>Rattei</surname> <given-names>T.</given-names></name> <name><surname>Tischler</surname> <given-names>P.</given-names></name> <name><surname>Meckenstock</surname> <given-names>R. U.</given-names></name></person-group> (<year>2010</year>). <article-title>Identification of enzymes involved in anaerobic benzene degradation by a strictly anaerobic iron-reducing enrichment culture.</article-title> <source><italic>Environ. Microbiol.</italic></source> <volume>12</volume> <fpage>2783</fpage>&#x2013;<lpage>2796</lpage>. <pub-id pub-id-type="doi">10.1111/j.1462-2920.2010.02248.x</pub-id> <pub-id pub-id-type="pmid">20545743</pub-id></citation></ref>
<ref id="B3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Al-Hasan</surname> <given-names>R. H.</given-names></name> <name><surname>Al-Bader</surname> <given-names>D. A.</given-names></name> <name><surname>Sorkhoh</surname> <given-names>N. A.</given-names></name> <name><surname>Radwan</surname> <given-names>S. S.</given-names></name></person-group> (<year>1998</year>). <article-title>Evidence for n-alkane consumption and oxidation by filamentous cyanobacteria from oil-contaminated coasts of the Arabian Gulf.</article-title> <source><italic>Mar. Biol.</italic></source> <volume>130</volume> <fpage>521</fpage>&#x2013;<lpage>527</lpage>. <pub-id pub-id-type="doi">10.1007/s002270050272</pub-id></citation></ref>
<ref id="B4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Al-Mailem</surname> <given-names>D. M.</given-names></name> <name><surname>Sorkhoh</surname> <given-names>N. A.</given-names></name> <name><surname>Al-Awadhi</surname> <given-names>H.</given-names></name> <name><surname>Eliyas</surname> <given-names>M.</given-names></name> <name><surname>Radwan</surname> <given-names>S. S.</given-names></name></person-group> (<year>2010</year>). <article-title>Biodegradation of crude oil and pure hydrocarbons by extreme halophilic archaea from hypersaline coasts of the Arabian Gulf.</article-title> <source><italic>Extremophiles</italic></source> <volume>14</volume> <fpage>321</fpage>&#x2013;<lpage>328</lpage>. <pub-id pub-id-type="doi">10.1007/s00792-010-0312-9</pub-id> <pub-id pub-id-type="pmid">20364355</pub-id></citation></ref>
<ref id="B5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Atlas</surname> <given-names>R. M.</given-names></name> <name><surname>Hazen</surname> <given-names>T. C.</given-names></name></person-group> (<year>2011</year>). <article-title>Oil biodegradation and bioremediation: a tale of the two worst spills in U.S. history.</article-title> <source><italic>Environ. Sci. Technol.</italic></source> <volume>45</volume> <fpage>6709</fpage>&#x2013;<lpage>6715</lpage>. <pub-id pub-id-type="doi">10.1021/es2013227</pub-id> <pub-id pub-id-type="pmid">21699212</pub-id></citation></ref>
<ref id="B6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Avila-Maga&#x00F1;a</surname> <given-names>V.</given-names></name> <name><surname>Kamel</surname> <given-names>B.</given-names></name> <name><surname>DeSalvo</surname> <given-names>M.</given-names></name> <name><surname>G&#x00F3;mez-Campo</surname> <given-names>K.</given-names></name> <name><surname>Enr&#x00ED;quez</surname> <given-names>S.</given-names></name> <name><surname>Kitano</surname> <given-names>H.</given-names></name><etal/></person-group> (<year>2021</year>). <article-title>Elucidating gene expression adaptation of phylogenetically divergent coral holobionts under heat stress.</article-title> <source><italic>Nat. Commun.</italic></source> <volume>12</volume>:<issue>5731</issue>. <pub-id pub-id-type="doi">10.1038/S41467-021-25950-4</pub-id> <pub-id pub-id-type="pmid">34593802</pub-id></citation></ref>
<ref id="B7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bateman</surname> <given-names>A.</given-names></name> <name><surname>Birney</surname> <given-names>E.</given-names></name> <name><surname>Durbin</surname> <given-names>R.</given-names></name> <name><surname>Eddy</surname> <given-names>S.</given-names></name> <name><surname>Howe</surname> <given-names>K.</given-names></name> <name><surname>Sonnhammer</surname> <given-names>E.</given-names></name></person-group> (<year>2000</year>). <article-title>The Pfam protein families database.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>28</volume> <fpage>263</fpage>&#x2013;<lpage>266</lpage>. <pub-id pub-id-type="doi">10.1093/nar/28.1.263</pub-id> <pub-id pub-id-type="pmid">10592242</pub-id></citation></ref>
<ref id="B8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Binark</surname> <given-names>N.</given-names></name> <name><surname>G&#x00FC;ven</surname> <given-names>K. C.</given-names></name> <name><surname>Gezgin</surname> <given-names>T.</given-names></name> <name><surname>&#x00DC;nl&#x00FC;</surname> <given-names>S.</given-names></name></person-group> (<year>2000</year>). <article-title>Oil pollution of marine algae.</article-title> <source><italic>Bull. Environ. Contam. Toxicol.</italic></source> <volume>64</volume> <fpage>866</fpage>&#x2013;<lpage>872</lpage>. <pub-id pub-id-type="doi">10.1007/s0012800083</pub-id> <pub-id pub-id-type="pmid">10856345</pub-id></citation></ref>
<ref id="B9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Boonmak</surname> <given-names>C.</given-names></name> <name><surname>Takahashi</surname> <given-names>Y.</given-names></name> <name><surname>Morikawa</surname> <given-names>M.</given-names></name></person-group> (<year>2014</year>). <article-title>Cloning and expression of three ladA-type alkane monooxygenase genes from an extremely thermophilic alkane-degrading bacterium Geobacillus thermoleovorans B23.</article-title> <source><italic>Extremophiles</italic></source> <volume>18</volume> <fpage>515</fpage>&#x2013;<lpage>523</lpage>. <pub-id pub-id-type="doi">10.1007/s00792-014-0636-y</pub-id> <pub-id pub-id-type="pmid">24682607</pub-id></citation></ref>
<ref id="B10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Buchfink</surname> <given-names>B.</given-names></name> <name><surname>Xie</surname> <given-names>C.</given-names></name> <name><surname>Huson</surname> <given-names>D. H.</given-names></name></person-group> (<year>2014</year>). <article-title>Fast and sensitive protein alignment using DIAMOND.</article-title> <source><italic>Nat. Methods</italic></source> <volume>12</volume> <fpage>59</fpage>&#x2013;<lpage>60</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.3176</pub-id> <pub-id pub-id-type="pmid">25402007</pub-id></citation></ref>
<ref id="B11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Callaghan</surname> <given-names>A. V.</given-names></name> <name><surname>Wawrik</surname> <given-names>B.</given-names></name> <name><surname>N&#x00ED; Chadhain</surname> <given-names>S. M.</given-names></name> <name><surname>Young</surname> <given-names>L. Y.</given-names></name> <name><surname>Zylstra</surname> <given-names>G. J.</given-names></name></person-group> (<year>2008</year>). <article-title>Anaerobic alkane-degrading strain AK-01 contains two alkylsuccinate synthase genes.</article-title> <source><italic>Biochem. Biophys. Res. Commun.</italic></source> <volume>366</volume> <fpage>142</fpage>&#x2013;<lpage>148</lpage>. <pub-id pub-id-type="doi">10.1016/j.bbrc.2007.11.094</pub-id> <pub-id pub-id-type="pmid">18053803</pub-id></citation></ref>
<ref id="B12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cantalapiedra</surname> <given-names>C. P.</given-names></name> <name><surname>Hern Andez-Plaza</surname> <given-names>A.</given-names></name> <name><surname>Letunic</surname> <given-names>I.</given-names></name> <name><surname>Bork</surname> <given-names>P.</given-names></name> <name><surname>Huerta-Cepas</surname> <given-names>J.</given-names></name></person-group> (<year>2021</year>). <article-title>eggNOG-mapper v2: functional annotation, orthology assignments, and domain prediction at the metagenomic scale.</article-title> <source><italic>Mol. Biol. Evol.</italic></source> <volume>1</volume>:<issue>msab293</issue>. <pub-id pub-id-type="doi">10.1093/MOLBEV/MSAB293</pub-id> <pub-id pub-id-type="pmid">34597405</pub-id></citation></ref>
<ref id="B13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Christman</surname> <given-names>G. D.</given-names></name> <name><surname>Le&#x00F3;n-Zayas</surname> <given-names>R. I.</given-names></name> <name><surname>Zhao</surname> <given-names>R.</given-names></name> <name><surname>Summers</surname> <given-names>Z. M.</given-names></name> <name><surname>Biddle</surname> <given-names>J. F.</given-names></name></person-group> (<year>2020</year>). <article-title>Novel clostridial lineages recovered from metagenomes of a hot oil reservoir.</article-title> <source><italic>Sci. Rep.</italic></source> <volume>10</volume>:<issue>8048</issue>. <pub-id pub-id-type="doi">10.1038/s41598-020-64904-6</pub-id> <pub-id pub-id-type="pmid">32415178</pub-id></citation></ref>
<ref id="B14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chun</surname> <given-names>J.</given-names></name> <name><surname>Kim</surname> <given-names>K.</given-names></name> <name><surname>Lee</surname> <given-names>J. H.</given-names></name> <name><surname>Choi</surname> <given-names>Y.</given-names></name></person-group> (<year>2010</year>). <article-title>The analysis of oral microbial communities of wild-type and toll-like receptor 2-deficient mice using a 454 GS FLX Titanium pyrosequencer.</article-title> <source><italic>BMC Microbiol.</italic></source> <volume>10</volume>:<issue>101</issue>. <pub-id pub-id-type="doi">10.1186/1471-2180-10-101</pub-id> <pub-id pub-id-type="pmid">20370919</pub-id></citation></ref>
<ref id="B15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dombrowski</surname> <given-names>N.</given-names></name> <name><surname>Donaho</surname> <given-names>J. A.</given-names></name> <name><surname>Gutierrez</surname> <given-names>T.</given-names></name> <name><surname>Seitz</surname> <given-names>K. W.</given-names></name> <name><surname>Teske</surname> <given-names>A. P.</given-names></name> <name><surname>Baker</surname> <given-names>B. J.</given-names></name></person-group> (<year>2016</year>). <article-title>Reconstructing metabolic pathways of hydrocarbon-degrading bacteria from the Deepwater Horizon oil spill.</article-title> <source><italic>Nat. Microbiol.</italic></source> <volume>1</volume> <fpage>1</fpage>&#x2013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1038/nmicrobiol.2016.57</pub-id> <pub-id pub-id-type="pmid">27572965</pub-id></citation></ref>
<ref id="B16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dong</surname> <given-names>X.</given-names></name> <name><surname>Greening</surname> <given-names>C.</given-names></name> <name><surname>Rattray</surname> <given-names>J. E.</given-names></name> <name><surname>Chakraborty</surname> <given-names>A.</given-names></name> <name><surname>Chuvochina</surname> <given-names>M.</given-names></name> <name><surname>Mayumi</surname> <given-names>D.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Metabolic potential of uncultured bacteria and archaea associated with petroleum seepage in deep-sea sediments.</article-title> <source><italic>Nat. Commun.</italic></source> <volume>10</volume> <fpage>1</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1038/s41467-019-09747-0</pub-id> <pub-id pub-id-type="pmid">31000700</pub-id></citation></ref>
<ref id="B17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dong</surname> <given-names>X.</given-names></name> <name><surname>Rattray</surname> <given-names>J. E.</given-names></name> <name><surname>Campbell</surname> <given-names>D. C.</given-names></name> <name><surname>Webb</surname> <given-names>J.</given-names></name> <name><surname>Chakraborty</surname> <given-names>A.</given-names></name> <name><surname>Adebayo</surname> <given-names>O.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>Thermogenic hydrocarbon biodegradation by diverse depth-stratified microbial populations at a Scotian Basin cold seep.</article-title> <source><italic>Nat. Commun.</italic></source> <volume>11</volume>:<issue>5825</issue>. <pub-id pub-id-type="doi">10.1038/s41467-020-19648-2</pub-id> <pub-id pub-id-type="pmid">33203858</pub-id></citation></ref>
<ref id="B18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Eddy</surname> <given-names>S. R.</given-names></name></person-group> (<year>2004</year>). <article-title>What is a hidden markov model?</article-title> <source><italic>Nat. Biotechnol.</italic></source> <volume>22</volume> <fpage>1315</fpage>&#x2013;<lpage>1316</lpage>. <pub-id pub-id-type="doi">10.1038/nbt1004-1315</pub-id> <pub-id pub-id-type="pmid">15470472</pub-id></citation></ref>
<ref id="B19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Eddy</surname> <given-names>S. R.</given-names></name></person-group> (<year>2011</year>). <article-title>Accelerated profile HMM searches.</article-title> <source><italic>PLoS Comput. Biol.</italic></source> <volume>7</volume>:<issue>1002195</issue>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1002195</pub-id> <pub-id pub-id-type="pmid">22039361</pub-id></citation></ref>
<ref id="B20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Edgar</surname> <given-names>R. C.</given-names></name></person-group> (<year>2004</year>). <article-title>MUSCLE: multiple sequence alignment with high accuracy and high throughput.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>32</volume> <fpage>1792</fpage>&#x2013;<lpage>1797</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkh340</pub-id> <pub-id pub-id-type="pmid">15034147</pub-id></citation></ref>
<ref id="B21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Edgar</surname> <given-names>R. C.</given-names></name></person-group> (<year>2010</year>). <article-title>Search and clustering orders of magnitude faster than BLAST.</article-title> <source><italic>Bioinformatics</italic></source> <volume>26</volume> <fpage>2460</fpage>&#x2013;<lpage>2461</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btq461</pub-id> <pub-id pub-id-type="pmid">20709691</pub-id></citation></ref>
<ref id="B22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Feigelman</surname> <given-names>R.</given-names></name> <name><surname>Kahlert</surname> <given-names>C. R.</given-names></name> <name><surname>Baty</surname> <given-names>F.</given-names></name> <name><surname>Rassouli</surname> <given-names>F.</given-names></name> <name><surname>Kleiner</surname> <given-names>R. L.</given-names></name> <name><surname>Kohler</surname> <given-names>P.</given-names></name><etal/></person-group> (<year>2017</year>). <article-title>Sputum DNA sequencing in cystic fibrosis: non-invasive access to the lung microbiome and to pathogen details.</article-title> <source><italic>Microbiome</italic></source> <volume>5</volume>:<issue>20</issue>. <pub-id pub-id-type="doi">10.1186/s40168-017-0234-1</pub-id> <pub-id pub-id-type="pmid">28187782</pub-id></citation></ref>
<ref id="B23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Finn</surname> <given-names>R. D.</given-names></name> <name><surname>Bateman</surname> <given-names>A.</given-names></name> <name><surname>Clements</surname> <given-names>J.</given-names></name> <name><surname>Coggill</surname> <given-names>P.</given-names></name> <name><surname>Eberhardt</surname> <given-names>R. Y.</given-names></name> <name><surname>Eddy</surname> <given-names>S. R.</given-names></name><etal/></person-group> (<year>2014</year>). <article-title>Pfam: the protein families database.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>42</volume> <fpage>D222</fpage>&#x2013;<lpage>D230</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkt1223</pub-id> <pub-id pub-id-type="pmid">24288371</pub-id></citation></ref>
<ref id="B24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gonz&#x00E1;lez-Gaya</surname> <given-names>B.</given-names></name> <name><surname>Mart&#x00ED;nez-Varela</surname> <given-names>A.</given-names></name> <name><surname>Vila-Costa</surname> <given-names>M.</given-names></name> <name><surname>Casal</surname> <given-names>P.</given-names></name> <name><surname>Cerro-G&#x00E1;lvez</surname> <given-names>E.</given-names></name> <name><surname>Berrojalbiz</surname> <given-names>N.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Biodegradation as an important sink of aromatic hydrocarbons in the oceans.</article-title> <source><italic>Nat. Geosci.</italic></source> <volume>12</volume> <fpage>119</fpage>&#x2013;<lpage>125</lpage>. <pub-id pub-id-type="doi">10.1038/s41561-018-0285-3</pub-id></citation></ref>
<ref id="B25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gregson</surname> <given-names>B. H.</given-names></name> <name><surname>Metodieva</surname> <given-names>G.</given-names></name> <name><surname>Metodiev</surname> <given-names>M. V.</given-names></name> <name><surname>Golyshin</surname> <given-names>P. N.</given-names></name> <name><surname>McKew</surname> <given-names>B. A.</given-names></name></person-group> (<year>2018</year>). <article-title>Differential protein expression during growth on medium versus long-chain alkanes in the obligate marine hydrocarbon-degrading bacterium <italic>Thalassolituus oleivorans</italic> MIL-1.</article-title> <source><italic>Front. Microbiol.</italic></source> <volume>9</volume>:<issue>3130</issue>. <pub-id pub-id-type="doi">10.3389/fmicb.2018.03130</pub-id> <pub-id pub-id-type="pmid">30619200</pub-id></citation></ref>
<ref id="B26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gutierrez</surname> <given-names>T.</given-names></name> <name><surname>Singleton</surname> <given-names>D. R.</given-names></name> <name><surname>Aitken</surname> <given-names>M. D.</given-names></name> <name><surname>Semple</surname> <given-names>K. T.</given-names></name></person-group> (<year>2011</year>). <article-title>Stable isotope probing of an algal bloom to identify uncultivated members of the <italic>Rhodobacteraceae</italic> associated with low-molecular-weight polycyclic aromatic hydrocarbon degradation.</article-title> <source><italic>Appl. Environ. Microbiol.</italic></source> <volume>77</volume> <fpage>7856</fpage>&#x2013;<lpage>7860</lpage>. <pub-id pub-id-type="doi">10.1128/AEM.06200-11</pub-id> <pub-id pub-id-type="pmid">21926219</pub-id></citation></ref>
<ref id="B27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Herman</surname> <given-names>D. R.</given-names></name> <name><surname>Rhoades</surname> <given-names>N.</given-names></name> <name><surname>Mercado</surname> <given-names>J.</given-names></name> <name><surname>Argueta</surname> <given-names>P.</given-names></name> <name><surname>Lopez</surname> <given-names>U.</given-names></name> <name><surname>Flores</surname> <given-names>G. E.</given-names></name></person-group> (<year>2020</year>). <article-title>Dietary habits of 2- to 9-year-old american children are associated with gut microbiome composition.</article-title> <source><italic>J. Acad. Nutr. Diet.</italic></source> <volume>120</volume> <fpage>517</fpage>&#x2013;<lpage>534</lpage>. <pub-id pub-id-type="doi">10.1016/j.jand.2019.07.024</pub-id> <pub-id pub-id-type="pmid">31668602</pub-id></citation></ref>
<ref id="B28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>P.</given-names></name> <name><surname>Tom</surname> <given-names>L.</given-names></name> <name><surname>Singh</surname> <given-names>A.</given-names></name> <name><surname>Thomas</surname> <given-names>B. C.</given-names></name> <name><surname>Baker</surname> <given-names>B. J.</given-names></name> <name><surname>Piceno</surname> <given-names>Y. M.</given-names></name><etal/></person-group> (<year>2016</year>). <article-title>Genome-resolved metagenomic analysis reveals roles for candidate phyla and other microbial community members in biogeochemical transformations in oil reservoirs.</article-title> <source><italic>mBio</italic></source> <volume>7</volume>:<issue>e01669-15</issue>. <pub-id pub-id-type="doi">10.1128/mBio.01669-15</pub-id> <pub-id pub-id-type="pmid">26787827</pub-id></citation></ref>
<ref id="B29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huerta-Cepas</surname> <given-names>J.</given-names></name> <name><surname>Szklarczyk</surname> <given-names>D.</given-names></name> <name><surname>Heller</surname> <given-names>D.</given-names></name> <name><surname>Hern&#x00E1;ndez-Plaza</surname> <given-names>A.</given-names></name> <name><surname>Forslund</surname> <given-names>S. K.</given-names></name> <name><surname>Cook</surname> <given-names>H.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>eggNOG 5.0: a hierarchical, functionally and phylogenetically annotated orthology resource based on 5090 organisms and 2502 viruses.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>47</volume> <fpage>D309</fpage>&#x2013;<lpage>D314</lpage>. <pub-id pub-id-type="doi">10.1093/NAR/GKY1085</pub-id> <pub-id pub-id-type="pmid">30418610</pub-id></citation></ref>
<ref id="B30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huson</surname> <given-names>D. H.</given-names></name> <name><surname>Richter</surname> <given-names>D. C.</given-names></name> <name><surname>Rausch</surname> <given-names>C.</given-names></name> <name><surname>Dezulian</surname> <given-names>T.</given-names></name> <name><surname>Franz</surname> <given-names>M.</given-names></name> <name><surname>Rupp</surname> <given-names>R.</given-names></name></person-group> (<year>2007</year>). <article-title>Dendroscope: an interactive viewer for large phylogenetic trees.</article-title> <source><italic>BMC Bioinform.</italic></source> <volume>8</volume>:<issue>460</issue>. <pub-id pub-id-type="doi">10.1186/1471-2105-8-460</pub-id> <pub-id pub-id-type="pmid">18034891</pub-id></citation></ref>
<ref id="B31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hyatt</surname> <given-names>D.</given-names></name> <name><surname>Chen</surname> <given-names>G. L.</given-names></name> <name><surname>LoCascio</surname> <given-names>P. F.</given-names></name> <name><surname>Land</surname> <given-names>M. L.</given-names></name> <name><surname>Larimer</surname> <given-names>F. W.</given-names></name> <name><surname>Hauser</surname> <given-names>L. J.</given-names></name></person-group> (<year>2010</year>). <article-title>Prodigal: prokaryotic gene recognition and translation initiation site identification.</article-title> <source><italic>BMC Bioinform.</italic></source> <volume>11</volume>:<issue>119</issue>. <pub-id pub-id-type="doi">10.1186/1471-2105-11-119</pub-id> <pub-id pub-id-type="pmid">20211023</pub-id></citation></ref>
<ref id="B32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kanehisa</surname> <given-names>M.</given-names></name> <name><surname>Goto</surname> <given-names>S.</given-names></name></person-group> (<year>2000</year>). <article-title>KEGG: kyoto encyclopedia of genes and genomes.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>28</volume> <fpage>27</fpage>&#x2013;<lpage>30</lpage>. <pub-id pub-id-type="doi">10.1093/nar/28.1.27</pub-id> <pub-id pub-id-type="pmid">10592173</pub-id></citation></ref>
<ref id="B33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kanehisa</surname> <given-names>M.</given-names></name> <name><surname>Sato</surname> <given-names>Y.</given-names></name> <name><surname>Kawashima</surname> <given-names>M.</given-names></name> <name><surname>Furumichi</surname> <given-names>M.</given-names></name> <name><surname>Tanabe</surname> <given-names>M.</given-names></name></person-group> (<year>2016</year>). <article-title>KEGG as a reference resource for gene and protein annotation.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>44</volume> <fpage>D457</fpage>&#x2013;<lpage>D462</lpage>. <pub-id pub-id-type="doi">10.1093/NAR/GKV1070</pub-id> <pub-id pub-id-type="pmid">26476454</pub-id></citation></ref>
<ref id="B34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Karsenti</surname> <given-names>E.</given-names></name> <name><surname>Acinas</surname> <given-names>S. G.</given-names></name> <name><surname>Bork</surname> <given-names>P.</given-names></name> <name><surname>Bowler</surname> <given-names>C.</given-names></name> <name><surname>De Vargas</surname> <given-names>C.</given-names></name> <name><surname>Raes</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2011</year>). <article-title>A holistic approach to marine eco-systems biology.</article-title> <source><italic>PLoS Biol.</italic></source> <volume>9</volume>:<issue>e1001177</issue>. <pub-id pub-id-type="doi">10.1371/journal.pbio.1001177</pub-id> <pub-id pub-id-type="pmid">22028628</pub-id></citation></ref>
<ref id="B35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Khelifi</surname> <given-names>N.</given-names></name> <name><surname>Amin Ali</surname> <given-names>O.</given-names></name> <name><surname>Roche</surname> <given-names>P.</given-names></name> <name><surname>Grossi</surname> <given-names>V.</given-names></name> <name><surname>Brochier-Armanet</surname> <given-names>C.</given-names></name> <name><surname>Valette</surname> <given-names>O.</given-names></name><etal/></person-group> (<year>2014</year>). <article-title>Anaerobic oxidation of long-chain n-alkanes by the hyperthermophilic sulfate-reducing archaeon, <italic>Archaeoglobus fulgidus</italic>.</article-title> <source><italic>ISME J.</italic></source> <volume>8</volume> <fpage>2153</fpage>&#x2013;<lpage>2166</lpage>. <pub-id pub-id-type="doi">10.1038/ismej.2014.58</pub-id> <pub-id pub-id-type="pmid">24763368</pub-id></citation></ref>
<ref id="B36"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kniemeyer</surname> <given-names>O.</given-names></name> <name><surname>Musat</surname> <given-names>F.</given-names></name> <name><surname>Sievert</surname> <given-names>S. M.</given-names></name> <name><surname>Knittel</surname> <given-names>K.</given-names></name> <name><surname>Wilkes</surname> <given-names>H.</given-names></name> <name><surname>Blumenberg</surname> <given-names>M.</given-names></name><etal/></person-group> (<year>2007</year>). <article-title>Anaerobic oxidation of short-chain hydrocarbons by marine sulphate-reducing bacteria.</article-title> <source><italic>Nature</italic></source> <volume>449</volume> <fpage>898</fpage>&#x2013;<lpage>901</lpage>. <pub-id pub-id-type="doi">10.1038/nature06200</pub-id> <pub-id pub-id-type="pmid">17882164</pub-id></citation></ref>
<ref id="B37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lea-Smith</surname> <given-names>D. J.</given-names></name> <name><surname>Biller</surname> <given-names>S. J.</given-names></name> <name><surname>Davey</surname> <given-names>M. P.</given-names></name> <name><surname>Cotton</surname> <given-names>C. A. R.</given-names></name> <name><surname>Sepulveda</surname> <given-names>B. M. P.</given-names></name> <name><surname>Turchyn</surname> <given-names>A. V.</given-names></name><etal/></person-group> (<year>2015</year>). <article-title>Contribution of cyanobacterial alkane production to the ocean hydrocarbon cycle.</article-title> <source><italic>Proc. Natl. Acad. Sci. U.S.A.</italic></source> <volume>112</volume> <fpage>13591</fpage>&#x2013;<lpage>13596</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1507274112</pub-id> <pub-id pub-id-type="pmid">26438854</pub-id></citation></ref>
<ref id="B38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lerdau</surname> <given-names>M.</given-names></name> <name><surname>Guenther</surname> <given-names>A.</given-names></name> <name><surname>Monson</surname> <given-names>R.</given-names></name></person-group> (<year>1997</year>). <article-title>Plant production and emission of volatile organic compounds: plant-produced hydrocarbons influence not only the plant itself but the atmosphere a well.</article-title> <source><italic>BioScience</italic></source> <volume>47</volume> <fpage>373</fpage>&#x2013;<lpage>383</lpage>. <pub-id pub-id-type="doi">10.2307/1313152</pub-id></citation></ref>
<ref id="B39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Letunic</surname> <given-names>I.</given-names></name> <name><surname>Bork</surname> <given-names>P.</given-names></name></person-group> (<year>2019</year>). <article-title>Interactive tree of life (iTOL) v4: recent updates and new developments.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>47</volume> <fpage>W256</fpage>&#x2013;<lpage>W259</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkz239</pub-id> <pub-id pub-id-type="pmid">30931475</pub-id></citation></ref>
<ref id="B40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>D.</given-names></name> <name><surname>Liu</surname> <given-names>C. M.</given-names></name> <name><surname>Luo</surname> <given-names>R.</given-names></name> <name><surname>Sadakane</surname> <given-names>K.</given-names></name> <name><surname>Lam</surname> <given-names>T. W.</given-names></name></person-group> (<year>2015</year>). <article-title>MEGAHIT: an ultra-fast single-node solution for large and complex metagenomics assembly <italic>via</italic> succinct de Bruijn graph.</article-title> <source><italic>Bioinformatics</italic></source> <volume>31</volume> <fpage>1674</fpage>&#x2013;<lpage>1676</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btv033</pub-id> <pub-id pub-id-type="pmid">25609793</pub-id></citation></ref>
<ref id="B41"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Y. P.</given-names></name> <name><surname>Pan</surname> <given-names>J. C.</given-names></name> <name><surname>Ma</surname> <given-names>Y. L.</given-names></name></person-group> (<year>2020</year>). <article-title>Elucidation of multiple alkane hydroxylase systems in biodegradation of crude oil n-alkane pollution by <italic>Pseudomonas aeruginosa</italic> DN1.</article-title> <source><italic>J. Appl. Microbiol.</italic></source> <volume>128</volume> <fpage>151</fpage>&#x2013;<lpage>160</lpage>. <pub-id pub-id-type="doi">10.1111/jam.14470</pub-id> <pub-id pub-id-type="pmid">31566849</pub-id></citation></ref>
<ref id="B42"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Y. F.</given-names></name> <name><surname>Galzerani</surname> <given-names>D. D.</given-names></name> <name><surname>Mbadinga</surname> <given-names>S. M.</given-names></name> <name><surname>Zaramela</surname> <given-names>L. S.</given-names></name> <name><surname>Gu</surname> <given-names>J. D.</given-names></name> <name><surname>Mu</surname> <given-names>B. Z.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>Metabolic capability and in situ activity of microorganisms in an oil reservoir.</article-title> <source><italic>Microbiome</italic></source> <volume>6</volume>:<issue>5</issue>. <pub-id pub-id-type="doi">10.1186/s40168-017-0392-1</pub-id> <pub-id pub-id-type="pmid">29304850</pub-id></citation></ref>
<ref id="B43"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Y. F.</given-names></name> <name><surname>Qi</surname> <given-names>Z. Z.</given-names></name> <name><surname>Shou</surname> <given-names>L. B.</given-names></name> <name><surname>Liu</surname> <given-names>J. F.</given-names></name> <name><surname>Yang</surname> <given-names>S. Z.</given-names></name> <name><surname>Gu</surname> <given-names>J. D.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Anaerobic hydrocarbon degradation in candidate phylum &#x2018;Atribacteria&#x2019; (JS1) inferred from genomics.</article-title> <source><italic>ISME J.</italic></source> <volume>13</volume> <fpage>2377</fpage>&#x2013;<lpage>2390</lpage>. <pub-id pub-id-type="doi">10.1038/s41396-019-0448-2</pub-id> <pub-id pub-id-type="pmid">31171858</pub-id></citation></ref>
<ref id="B44"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Love</surname> <given-names>C. R.</given-names></name> <name><surname>Arrington</surname> <given-names>E. C.</given-names></name> <name><surname>Gosselin</surname> <given-names>K. M.</given-names></name> <name><surname>Reddy</surname> <given-names>C. M.</given-names></name> <name><surname>Van Mooy</surname> <given-names>B. A. S.</given-names></name> <name><surname>Nelson</surname> <given-names>R. K.</given-names></name><etal/></person-group> (<year>2021</year>). <article-title>Microbial production and consumption of hydrocarbons in the global ocean.</article-title> <source><italic>Nat. Microbiol.</italic></source> <volume>6</volume> <fpage>489</fpage>&#x2013;<lpage>498</lpage>. <pub-id pub-id-type="doi">10.1038/s41564-020-00859-8</pub-id> <pub-id pub-id-type="pmid">33526885</pub-id></citation></ref>
<ref id="B45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>McGinnis</surname> <given-names>S.</given-names></name> <name><surname>Madden</surname> <given-names>T. L.</given-names></name></person-group> (<year>2004</year>). <article-title>BLAST: at the core of a powerful and diverse set of sequence analysis tools.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>32</volume> <fpage>20</fpage>&#x2013;<lpage>25</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkh435</pub-id> <pub-id pub-id-type="pmid">15215342</pub-id></citation></ref>
<ref id="B46"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Momper</surname> <given-names>L.</given-names></name> <name><surname>Aronson</surname> <given-names>H. S.</given-names></name> <name><surname>Amend</surname> <given-names>J. P.</given-names></name></person-group> (<year>2018</year>). <article-title>Genomic description of &#x2018;candidatus abyssubacteria,&#x2019; a novel subsurface lineage within the candidate phylum hydrogenedentes.</article-title> <source><italic>Front. Microbiol.</italic></source> <volume>9</volume>:<issue>1993</issue>. <pub-id pub-id-type="doi">10.3389/fmicb.2018.01993</pub-id> <pub-id pub-id-type="pmid">30210471</pub-id></citation></ref>
<ref id="B47"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nie</surname> <given-names>Y.</given-names></name> <name><surname>Zhao</surname> <given-names>J.-Y.</given-names></name> <name><surname>Tang</surname> <given-names>Y.-Q.</given-names></name> <name><surname>Guo</surname> <given-names>P.</given-names></name> <name><surname>Yang</surname> <given-names>Y.</given-names></name> <name><surname>Wu</surname> <given-names>X.-L.</given-names></name><etal/></person-group> (<year>2016</year>). <article-title>Species divergence vs. functional convergence characterizes crude oil microbial community assembly.</article-title> <source><italic>Front. Microbiol.</italic></source> <volume>7</volume>:<issue>1254</issue>. <pub-id pub-id-type="doi">10.3389/fmicb.2016.01254</pub-id> <pub-id pub-id-type="pmid">27570522</pub-id></citation></ref>
<ref id="B48"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>O&#x2019;Leary</surname> <given-names>N. A.</given-names></name> <name><surname>Wright</surname> <given-names>M. W.</given-names></name> <name><surname>Brister</surname> <given-names>J. R.</given-names></name> <name><surname>Ciufo</surname> <given-names>S.</given-names></name> <name><surname>Haddad</surname> <given-names>D.</given-names></name> <name><surname>McVeigh</surname> <given-names>R.</given-names></name><etal/></person-group> (<year>2016</year>). <article-title>Reference sequence (RefSeq) database at NCBI: current status, taxonomic expansion, and functional annotation.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>44</volume> <fpage>D733</fpage>&#x2013;<lpage>D745</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkv1189</pub-id> <pub-id pub-id-type="pmid">26553804</pub-id></citation></ref>
<ref id="B49"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Orellana</surname> <given-names>M. V.</given-names></name> <name><surname>L&#x00F3;pez-Garc&#x00ED;a de Lomana</surname> <given-names>A.</given-names></name> <name><surname>Jennings</surname> <given-names>M. K.</given-names></name> <name><surname>Lee</surname> <given-names>A.</given-names></name> <name><surname>Hansman</surname> <given-names>R. L.</given-names></name> <name><surname>Thompson</surname> <given-names>A. W.</given-names></name><etal/></person-group> (<year>2017</year>). <source><italic>On the Influence of Diatom Programmed Cell Death on Carbon Export in the Ross Sea.</italic></source> <publisher-loc>Honolulu</publisher-loc>: <publisher-name>American Fisheries Society</publisher-name>.</citation></ref>
<ref id="B50"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Oren</surname> <given-names>A.</given-names></name></person-group> (<year>2019</year>). &#x201C;<article-title>Aerobic hydrocarbon-degrading archaea</article-title>,&#x201D; in <source><italic>Taxonomy, Genomics and Ecophysiology of Hydrocarbon-Degrading Microbes</italic></source>, <role>ed.</role> <person-group person-group-type="editor"><name><surname>McGenity</surname> <given-names>T. J.</given-names></name></person-group> (<publisher-loc>Berlin</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>), <fpage>41</fpage>&#x2013;<lpage>51</lpage>.</citation></ref>
<ref id="B51"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Parks</surname> <given-names>D. H.</given-names></name> <name><surname>Chuvochina</surname> <given-names>M.</given-names></name> <name><surname>Chaumeil</surname> <given-names>P. A.</given-names></name> <name><surname>Rinke</surname> <given-names>C.</given-names></name> <name><surname>Mussig</surname> <given-names>A. J.</given-names></name> <name><surname>Hugenholtz</surname> <given-names>P.</given-names></name></person-group> (<year>2020</year>). <article-title>A complete domain-to-species taxonomy for Bacteria and Archaea.</article-title> <source><italic>Nat. Biotechnol.</italic></source> <volume>38</volume> <fpage>1079</fpage>&#x2013;<lpage>1086</lpage>. <pub-id pub-id-type="doi">10.1038/s41587-020-0501-8</pub-id> <pub-id pub-id-type="pmid">32341564</pub-id></citation></ref>
<ref id="B52"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Peloquin</surname> <given-names>J. A.</given-names></name> <name><surname>Smith</surname> <given-names>W. O.</given-names></name></person-group> (<year>2007</year>). <article-title>Phytoplankton blooms in the Ross Sea, Antarctica: interannual variability in magnitude, temporal patterns, and composition.</article-title> <source><italic>J. Geophys. Res. Oceans</italic></source> <volume>112</volume>:<issue>C08013</issue>. <pub-id pub-id-type="doi">10.1029/2006JC003816</pub-id></citation></ref>
<ref id="B53"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Price</surname> <given-names>M. N.</given-names></name> <name><surname>Dehal</surname> <given-names>P. S.</given-names></name> <name><surname>Arkin</surname> <given-names>A. P.</given-names></name></person-group> (<year>2010</year>). <article-title>FastTree 2 &#x2013; approximately maximum-likelihood trees for large alignments.</article-title> <source><italic>PLoS One</italic></source> <volume>5</volume>:<issue>e0009490</issue>. <pub-id pub-id-type="doi">10.1371/journal.pone.0009490</pub-id> <pub-id pub-id-type="pmid">20224823</pub-id></citation></ref>
<ref id="B54"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Prier</surname> <given-names>C. K.</given-names></name> <name><surname>Kosjek</surname> <given-names>B.</given-names></name></person-group> (<year>2019</year>). <article-title>Recent preparative applications of redox enzymes.</article-title> <source><italic>Curr. Opin. Chem. Biol.</italic></source> <volume>49</volume> <fpage>105</fpage>&#x2013;<lpage>112</lpage>. <pub-id pub-id-type="doi">10.1016/j.cbpa.2018.11.011</pub-id> <pub-id pub-id-type="pmid">30554005</pub-id></citation></ref>
<ref id="B55"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qiao</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>W.</given-names></name> <name><surname>Lu</surname> <given-names>X.</given-names></name></person-group> (<year>2020</year>). <article-title>High light induced alka(e)ne biodegradation for lipid and redox homeostasis in cyanobacteria.</article-title> <source><italic>Front. Microbiol.</italic></source> <volume>11</volume>:<issue>1659</issue>. <pub-id pub-id-type="doi">10.3389/fmicb.2020.01659</pub-id> <pub-id pub-id-type="pmid">32765469</pub-id></citation></ref>
<ref id="B56"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rabus</surname> <given-names>R.</given-names></name> <name><surname>Boll</surname> <given-names>M.</given-names></name> <name><surname>Heider</surname> <given-names>J.</given-names></name> <name><surname>Meckenstock</surname> <given-names>R. U.</given-names></name> <name><surname>Buckel</surname> <given-names>W.</given-names></name> <name><surname>Einsle</surname> <given-names>O.</given-names></name><etal/></person-group> (<year>2016</year>). <article-title>Anaerobic microbial degradation of hydrocarbons: from enzymatic reactions to the environment.</article-title> <source><italic>J. Mol. Microbiol. Biotechnol.</italic></source> <volume>26</volume> <fpage>5</fpage>&#x2013;<lpage>28</lpage>. <pub-id pub-id-type="doi">10.1159/000443997</pub-id> <pub-id pub-id-type="pmid">26960061</pub-id></citation></ref>
<ref id="B57"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ramadass</surname> <given-names>K.</given-names></name> <name><surname>Megharaj</surname> <given-names>M.</given-names></name> <name><surname>Venkateswarlu</surname> <given-names>K.</given-names></name> <name><surname>Naidu</surname> <given-names>R.</given-names></name></person-group> (<year>2018</year>). <article-title>Bioavailability of weathered hydrocarbons in engine oil-contaminated soil: impact of bioaugmentation mediated by <italic>Pseudomonas</italic> spp. on bioremediation.</article-title> <source><italic>Sci. Total Environ.</italic></source> <volume>636</volume> <fpage>968</fpage>&#x2013;<lpage>974</lpage>. <pub-id pub-id-type="doi">10.1016/j.scitotenv.2018.04.379</pub-id> <pub-id pub-id-type="pmid">29913620</pub-id></citation></ref>
<ref id="B58"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rueter</surname> <given-names>P.</given-names></name> <name><surname>Rabus</surname> <given-names>R.</given-names></name> <name><surname>Wilkest</surname> <given-names>H.</given-names></name> <name><surname>Aeckersberg</surname> <given-names>F.</given-names></name> <name><surname>Rainey</surname> <given-names>F. A.</given-names></name> <name><surname>Jannasch</surname> <given-names>H. W.</given-names></name><etal/></person-group> (<year>1994</year>). <article-title>Anaerobic oxidation of hydrocarbons in crude oil by new types of sulphate-reducing bacteria.</article-title> <source><italic>Nature</italic></source> <volume>372</volume> <fpage>455</fpage>&#x2013;<lpage>458</lpage>. <pub-id pub-id-type="doi">10.1038/372455a0</pub-id> <pub-id pub-id-type="pmid">7984238</pub-id></citation></ref>
<ref id="B59"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schneiker</surname> <given-names>S.</given-names></name> <name><surname>Dos Santos</surname> <given-names>V. A. P. M.</given-names></name> <name><surname>Bartels</surname> <given-names>D.</given-names></name> <name><surname>Bekel</surname> <given-names>T.</given-names></name> <name><surname>Brecht</surname> <given-names>M.</given-names></name> <name><surname>Buhrmester</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2006</year>). <article-title>Genome sequence of the ubiquitous hydrocarbon-degrading marine bacterium <italic>Alcanivorax borkumensis</italic>.</article-title> <source><italic>Nat. Biotechnol.</italic></source> <volume>24</volume> <fpage>997</fpage>&#x2013;<lpage>1004</lpage>. <pub-id pub-id-type="doi">10.1038/nbt1232</pub-id> <pub-id pub-id-type="pmid">16878126</pub-id></citation></ref>
<ref id="B60"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sievers</surname> <given-names>F.</given-names></name> <name><surname>Wilm</surname> <given-names>A.</given-names></name> <name><surname>Dineen</surname> <given-names>D.</given-names></name> <name><surname>Gibson</surname> <given-names>T. J.</given-names></name> <name><surname>Karplus</surname> <given-names>K.</given-names></name> <name><surname>Li</surname> <given-names>W.</given-names></name><etal/></person-group> (<year>2011</year>). <article-title>Fast, scalable generation of high-quality protein multiple sequence alignments using Clustal Omega.</article-title> <source><italic>Mol. Syst. Biol.</italic></source> <volume>7</volume>:<issue>539</issue>. <pub-id pub-id-type="doi">10.1038/msb.2011.75</pub-id> <pub-id pub-id-type="pmid">21988835</pub-id></citation></ref>
<ref id="B61"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Smith</surname> <given-names>W. O.</given-names></name> <name><surname>Marra</surname> <given-names>J.</given-names></name> <name><surname>Hiscock</surname> <given-names>M. R.</given-names></name> <name><surname>Barber</surname> <given-names>R. T.</given-names></name></person-group> (<year>2000</year>). <article-title>The seasonal cycle of phytoplankton biomass and primary productivity in the Ross Sea, Antarctica.</article-title> <source><italic>Deep Sea Res. II Top. Stud. Oceanogr.</italic></source> <volume>47</volume> <fpage>3119</fpage>&#x2013;<lpage>3140</lpage>. <pub-id pub-id-type="doi">10.1016/S0967-0645(00)00061-8</pub-id></citation></ref>
<ref id="B62"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tan</surname> <given-names>B.</given-names></name> <name><surname>Jane Fowler</surname> <given-names>S.</given-names></name> <name><surname>Laban</surname> <given-names>N. A.</given-names></name> <name><surname>Dong</surname> <given-names>X.</given-names></name> <name><surname>Sensen</surname> <given-names>C. W.</given-names></name> <name><surname>Foght</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2015</year>). <article-title>Comparative analysis of metagenomes from three methanogenic hydrocarbon-degrading enrichment cultures with 41 environmental samples.</article-title> <source><italic>ISME J.</italic></source> <volume>9</volume> <fpage>2028</fpage>&#x2013;<lpage>2045</lpage>. <pub-id pub-id-type="doi">10.1038/ismej.2015.22</pub-id> <pub-id pub-id-type="pmid">25734684</pub-id></citation></ref>
<ref id="B63"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tornabene</surname> <given-names>T. G.</given-names></name> <name><surname>Kates</surname> <given-names>M.</given-names></name> <name><surname>Gelpi</surname> <given-names>E.</given-names></name> <name><surname>Oro</surname> <given-names>J.</given-names></name></person-group> (<year>1969</year>). <article-title>Occurrence of squalene, di- and tetrahydrosqualenes, and vitamin MK8 in an extremely halophilic bacterium, <italic>Halobacterium cutirubrun</italic>.</article-title> <source><italic>J. Lipid Res.</italic></source> <volume>10</volume> <fpage>294</fpage>&#x2013;<lpage>303</lpage>. <pub-id pub-id-type="doi">10.1016/s0022-2275(20)43087-1</pub-id></citation></ref>
<ref id="B64"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tully</surname> <given-names>B. J.</given-names></name> <name><surname>Graham</surname> <given-names>E. D.</given-names></name> <name><surname>Heidelberg</surname> <given-names>J. F.</given-names></name></person-group> (<year>2018</year>). <article-title>The reconstruction of 2,631 draft metagenome-assembled genomes from the global oceans.</article-title> <source><italic>Sci. Data</italic></source> <volume>5</volume>:<issue>203</issue>. <pub-id pub-id-type="doi">10.1038/sdata.2017.203</pub-id> <pub-id pub-id-type="pmid">29337314</pub-id></citation></ref>
<ref id="B65"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>W.</given-names></name> <name><surname>Shao</surname> <given-names>Z.</given-names></name></person-group> (<year>2014</year>). <article-title>The long-chain alkane metabolism network of <italic>Alcanivorax dieselolei</italic>.</article-title> <source><italic>Nat. Commun.</italic></source> <volume>5</volume> <fpage>1</fpage>&#x2013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1038/ncomms6755</pub-id> <pub-id pub-id-type="pmid">25502912</pub-id></citation></ref>
<ref id="B66"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>W.</given-names></name> <name><surname>Wang</surname> <given-names>L.</given-names></name> <name><surname>Shao</surname> <given-names>Z.</given-names></name></person-group> (<year>2018</year>). <article-title>Polycyclic aromatic hydrocarbon (PAH) degradation pathways of the obligate marine PAH degrader <italic>Cycloclasticus</italic> sp. strain P1.</article-title> <source><italic>Appl. Environ. Microbiol.</italic></source> <volume>84</volume>:<issue>e01261-18</issue>. <pub-id pub-id-type="doi">10.1128/AEM.01261-18</pub-id> <pub-id pub-id-type="pmid">30171002</pub-id></citation></ref>
<ref id="B67"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Waterhouse</surname> <given-names>A. M.</given-names></name> <name><surname>Procter</surname> <given-names>J. B.</given-names></name> <name><surname>Martin</surname> <given-names>D. M. A.</given-names></name> <name><surname>Clamp</surname> <given-names>M.</given-names></name> <name><surname>Barton</surname> <given-names>G. J.</given-names></name></person-group> (<year>2009</year>). <article-title>Jalview Version 2-A multiple sequence alignment editor and analysis workbench.</article-title> <source><italic>Bioinformatics</italic></source> <volume>25</volume> <fpage>1189</fpage>&#x2013;<lpage>1191</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp033</pub-id> <pub-id pub-id-type="pmid">19151095</pub-id></citation></ref>
<ref id="B68"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>W&#x00F6;hlbrand</surname> <given-names>L.</given-names></name> <name><surname>Kallerhoff</surname> <given-names>B.</given-names></name> <name><surname>Lange</surname> <given-names>D.</given-names></name> <name><surname>Hufnagel</surname> <given-names>P.</given-names></name> <name><surname>Thiermann</surname> <given-names>J.</given-names></name> <name><surname>Reinhardt</surname> <given-names>R.</given-names></name><etal/></person-group> (<year>2007</year>). <article-title>Functional proteomic view of metabolic regulation in &#x201C;<italic>Aromatoleum aromaticum</italic>&#x201D; strain EbN1.</article-title> <source><italic>Proteomics</italic></source> <volume>7</volume> <fpage>2222</fpage>&#x2013;<lpage>2239</lpage>. <pub-id pub-id-type="doi">10.1002/pmic.200600987</pub-id> <pub-id pub-id-type="pmid">17549795</pub-id></citation></ref>
<ref id="B69"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>X.</given-names></name> <name><surname>Liu</surname> <given-names>W.</given-names></name> <name><surname>Tian</surname> <given-names>S.</given-names></name> <name><surname>Wang</surname> <given-names>W.</given-names></name> <name><surname>Qi</surname> <given-names>Q.</given-names></name> <name><surname>Jiang</surname> <given-names>P.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>Petroleum hydrocarbon-degrading bacteria for the remediation of oil pollution under aerobic conditions: a perspective analysis.</article-title> <source><italic>Front. Microbiol.</italic></source> <volume>9</volume>:<issue>2885</issue>. <pub-id pub-id-type="doi">10.3389/fmicb.2018.02885</pub-id> <pub-id pub-id-type="pmid">30559725</pub-id></citation></ref>
<ref id="B70"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yao</surname> <given-names>G.</given-names></name> <name><surname>Yu</surname> <given-names>J.</given-names></name> <name><surname>Hou</surname> <given-names>Q.</given-names></name> <name><surname>Hui</surname> <given-names>W.</given-names></name> <name><surname>Liu</surname> <given-names>W.</given-names></name> <name><surname>Kwok</surname> <given-names>L. Y.</given-names></name><etal/></person-group> (<year>2017</year>). <article-title>A perspective study of koumiss microbiome by metagenomics analysis based on single-cell amplification technique.</article-title> <source><italic>Front. Microbiol.</italic></source> <volume>8</volume>:<issue>165</issue>. <pub-id pub-id-type="doi">10.3389/fmicb.2017.00165</pub-id> <pub-id pub-id-type="pmid">28223973</pub-id></citation></ref>
<ref id="B71"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>K.</given-names></name> <name><surname>Hu</surname> <given-names>Z.</given-names></name> <name><surname>Zeng</surname> <given-names>F.</given-names></name> <name><surname>Yang</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Jing</surname> <given-names>R.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Biodegradation of petroleum hydrocarbons and changes in microbial community structure in sediment under nitrate-, ferric-, sulfate-reducing and methanogenic conditions.</article-title> <source><italic>J. Environ. Manage.</italic></source> <volume>249</volume>:<issue>109425</issue>. <pub-id pub-id-type="doi">10.1016/j.jenvman.2019.109425</pub-id> <pub-id pub-id-type="pmid">31446121</pub-id></citation></ref>
<ref id="B72"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zorz</surname> <given-names>J. K.</given-names></name> <name><surname>Sharp</surname> <given-names>C.</given-names></name> <name><surname>Kleiner</surname> <given-names>M.</given-names></name> <name><surname>Gordon</surname> <given-names>P. M. K.</given-names></name> <name><surname>Pon</surname> <given-names>R. T.</given-names></name> <name><surname>Dong</surname> <given-names>X.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>A shared core microbiome in soda lakes separated by large distances.</article-title> <source><italic>Nat. Commun.</italic></source> <volume>10</volume>:<issue>4230</issue>. <pub-id pub-id-type="doi">10.1038/s41467-019-12195-5</pub-id> <pub-id pub-id-type="pmid">31530813</pub-id></citation></ref>
</ref-list>
<glossary>
<title>Abbreviations</title>
<def-list id="DL1">
<def-item><term>CANT-HYD</term><def><p>Calgary approach to ANnoTating HYDrocarbon degradation genes</p></def></def-item>
<def-item><term>HMM</term><def><p>Hidden Markov Models</p></def></def-item>
<def-item><term>GTDB</term><def><p>Genome Taxonomy Database</p></def></def-item>
<def-item><term>BTEX</term><def><p>Benzene, Toluene, Ethyl Benzene, and Xylene.</p></def></def-item>
</def-list>
</glossary>
<fn-group>
<fn id="footnote1">
<label>1</label>
<p>Bushnell, B. <italic>BBTools</italic>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://jgi.doe.gov/data-andtools/bbtools/">https://jgi.doe.gov/data-andtools/bbtools/</ext-link></p></fn>
</fn-group>
</back>
</article>
