<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Mol. Biosci.</journal-id>
<journal-title>Frontiers in Molecular Biosciences</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Mol. Biosci.</abbrev-journal-title>
<issn pub-type="epub">2296-889X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1395450</article-id>
<article-id pub-id-type="doi">10.3389/fmolb.2024.1395450</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Molecular Biosciences</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A pangenome analysis of ESKAPE bacteriophages: the underrepresentation may impact machine learning models</article-title>
<alt-title alt-title-type="left-running-head">Lee et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fmolb.2024.1395450">10.3389/fmolb.2024.1395450</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Lee</surname>
<given-names>Jeesu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Hunter</surname>
<given-names>Branden</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Shim</surname>
<given-names>Hyunjin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1295265/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Center for Biosystems and Biotech Data Science</institution>, <institution>Ghent University Global Campus</institution>, <addr-line>Incheon</addr-line>, <country>Republic of Korea</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Biology</institution>, <institution>California State University</institution>, <addr-line>Fresno</addr-line>, <addr-line>CA</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1869423/overview">Kun Qu</ext-link>, National University of Singapore, Singapore</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2221439/overview">Srirupa Chakraborty</ext-link>, Northeastern University, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2560103/overview">Chang Liu</ext-link>, Biogen Idec, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Hyunjin Shim, <email>shim@csufresno.edu</email>
</corresp>
<fn fn-type="other" id="fn1">
<label>
<sup>&#x2020;</sup>
</label>
<p>ORCID: Jeesu Lee, <ext-link ext-link-type="uri" xlink:href="http://orcid.org/0000-0003-1673-241X">orcid.org/0000-0003-1673-241X</ext-link>; Branden Hunter, <ext-link ext-link-type="uri" xlink:href="http://orcid.org/0009-0001-4490-8307">orcid.org/0009-0001-4490-8307</ext-link>; Hyunjin Shim, <ext-link ext-link-type="uri" xlink:href="http://orcid.org/0000-0002-7052-0971">orcid.org/0000-0002-7052-0971</ext-link>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>21</day>
<month>06</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>11</volume>
<elocation-id>1395450</elocation-id>
<history>
<date date-type="received">
<day>03</day>
<month>03</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>31</day>
<month>05</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Lee, Hunter and Shim.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Lee, Hunter and Shim</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Bacteriophages are the most prevalent biological entities in the biosphere. However, limitations in both medical relevance and sequencing technologies have led to a systematic underestimation of the genetic diversity within phages. This underrepresentation not only creates a significant gap in our understanding of phage roles across diverse biosystems but also introduces biases in computational models reliant on these data for training and testing. In this study, we focused on publicly available genomes of bacteriophages infecting high-priority ESKAPE pathogens to show the extent and impact of this underrepresentation. First, we demonstrate a stark underrepresentation of ESKAPE phage genomes within the public genome and protein databases. Next, a pangenome analysis of these ESKAPE phages reveals extensive sharing of core genes among phages infecting the same host. Furthermore, genome analyses and clustering highlight close nucleotide-level relationships among the ESKAPE phages, raising concerns about the limited diversity within current public databases. Lastly, we uncover a scarcity of unique lytic phages and phage proteins with antimicrobial activities against ESKAPE pathogens. This comprehensive analysis of the ESKAPE phages underscores the severity of underrepresentation and its potential implications. This lack of diversity in phage genomes may restrict the resurgence of phage therapy and cause biased outcomes in data-driven computational models due to incomplete and unbalanced biological datasets.</p>
</abstract>
<kwd-group>
<kwd>ESKAPE pathogens</kwd>
<kwd>pangenome</kwd>
<kwd>protein prediction</kwd>
<kwd>unbalanced datasets</kwd>
<kwd>bacteriophages</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Structural Biology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Background</title>
<p>Bacteriophages (phages), comprising the most abundant biological entities in the biosphere, play a crucial role in various ecological and microbial systems (<xref ref-type="bibr" rid="B54">Suttle, 2005</xref>; <xref ref-type="bibr" rid="B9">Clokie et al., 2011</xref>). They contribute significantly to the dynamics of microbial ecosystems, influencing bacterial populations and diversity. The interconnectedness of bacteriophages with bacterial communities underscores their importance in shaping microbial dynamics, with potential consequences for human health, agriculture, and environmental processes. Despite their ubiquity, the comprehensive understanding of their genetic repertoire has been relatively limited compared to that of other organisms (<xref ref-type="bibr" rid="B9">Clokie et al., 2011</xref>). This systematic understudy has resulted in a significant gap in our biological knowledge, particularly regarding the multifaceted roles phages play in diverse biosystems (<xref ref-type="bibr" rid="B2">Al- et al., 2020</xref>; <xref ref-type="bibr" rid="B50">Shim et al., 2021</xref>).</p>
<p>While historically perceived to have limited medical relevance compared to bacteria, the significance of phages in the medical domain is experiencing a resurgence. This renewed interest is closely linked to the escalating threat of antimicrobial resistance (<xref ref-type="bibr" rid="B3">Antimicrobial Resistance Collaborators, 2022</xref>), which has prompted a critical reassessment of alternative therapeutic strategies, notably the effectiveness of phage therapy. In the face of increasing bacterial resistance to traditional antibiotics, phages - viruses that infect and replicate within bacteria - have emerged as promising candidates for combating bacterial infections (<xref ref-type="bibr" rid="B56">Young and Gill, 2015</xref>; <xref ref-type="bibr" rid="B21">Gordillo et al., 2019</xref>). The specificity of phages in targeting particular bacterial strains, coupled with their ability to co-evolve with bacteria, presents a dynamic and potentially effective approach to counteract the challenges posed by antimicrobial resistance (<xref ref-type="bibr" rid="B49">Shim, 2023</xref>). This shift in perspective underscores the evolving landscape of medical research, emphasizing the importance of harnessing the unique attributes of phages in addressing the pressing global concern of antimicrobial resistance.</p>
<p>The underrepresentation of phages across various datasets not only impedes our capacity to unravel the complex dynamics of microbial interactions in nature but also introduces biases in diverse biological models. A majority of these models, reliant on current genomic data, may inadvertently incorporate a skewed perspective, thereby limiting their accuracy and applicability. This biased outcome is analogous to the recent issues in computer vision due to dataset imbalance and bias (<xref ref-type="bibr" rid="B11">Deviyani, 2022</xref>; <xref ref-type="bibr" rid="B25">Jones et al., 2024</xref>). For example, deep learning models have made significant strides in solving the long-standing protein folding problem in biology (<xref ref-type="bibr" rid="B12">Dill et al., 2008</xref>). However, these models rely on experimental protein structures for training and testing, which are used in combination with genomic data for multiple sequence alignment (<xref ref-type="bibr" rid="B26">Jumper et al., 2021</xref>). If genomic databases and protein structure repositories exhibit biases toward certain organisms, computational models derived from these datasets may fail to accurately represent the biological landscape. Bacteriophages, renowned for their rich diversity of small proteins (<xref ref-type="bibr" rid="B50">Shim et al., 2021</xref>), represent a facet of the protein landscape that remains relatively unexplored within the current biological context (<xref ref-type="fig" rid="F1">Figure 1A</xref>). Therefore, addressing this knowledge gap becomes imperative not only for advancing our understanding of phage biology but also for refining computational models essential for numerous scientific applications.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Underrepresentation of bacteriophages in the public databases. <bold>(A)</bold> Three-dimensional protein structure landscape represented by the variations in size and complexity. <bold>(B)</bold> Experimental protein structure entries by natural source organisms in the Protein Data Bank (PDB) classified at the Kingdom level. <bold>(C)</bold> Complete genomes of ESKAPE phages in the NCBI Virus database.</p>
</caption>
<graphic xlink:href="fmolb-11-1395450-g001.tif"/>
</fig>
<p>In this study, we examine the extent of the underrepresentation of phages in the genome and protein databases, focusing on the phages that are infecting the high-priority pathogens defined by the WHO (<xref ref-type="bibr" rid="B49">Shim, 2023</xref>). These phages may become medically relevant as an alternative antimicrobial therapy against ESKAPE pathogens, which is an acronym for the high-priority pathogens of <italic>Enterococcus</italic> spp., <italic>Staphylococcus aureus</italic>, <italic>Klebsiella pneumoniae</italic>, <italic>Acinetobacter baumannii</italic>, <italic>Pseudomonas aeruginosa</italic>, and <italic>Enterobacter</italic> spp. (<xref ref-type="bibr" rid="B44">Santajit and Indrawattana, 2016</xref>). In this study, we included 11 bacterial species from the WHO priority list and studied the phages of these ESKAPE pathogens using genome and pangenome analyses. We first analyzed the extent of biased datasets against bacteriophages in the Protein Data Bank (<xref ref-type="bibr" rid="B5">Berman et al., 2000</xref>) (<xref ref-type="fig" rid="F1">Figure 1B</xref>), before downloading all the complete genomes of the ESKAPE phages in the NCBI Virus database for the downstream analysis (<xref ref-type="fig" rid="F1">Figure 1C</xref>).</p>
<p>From this study, we aim to provide a quantitative analysis in understanding the extent of underrepresented datasets of phages and engage the scientific community to focus more attention on phage-related data collection given the wide implications on diverse fields from phage therapy to machine learning models. As research advances, the multifaceted roles of bacteriophages in modulating bacterial behavior, participating in microbial community dynamics, and offering therapeutic solutions continue to unfold. This evolving knowledge challenges the historical notion of limited medical relevance and positions bacteriophages as integral components of the complex microbial world with far-reaching implications for diverse fields, including medicine, ecology, and biotechnology.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>Materials and methods</title>
<sec id="s2-1">
<title>Curation of ESKAPE phage datasets</title>
<p>We first analyzed the underrepresentation of bacteriophages in the genome and protein databases. For the protein database, we downloaded the data distribution provided by the Protein Data Bank (PDB) (<xref ref-type="bibr" rid="B5">Berman et al., 2000</xref>) on the entries of experimental protein structures by natural source organisms (downloaded 2024/01/24). The definition of natural source organisms for these structures is from a natural and non-modified source. To categorize each entry to a natural source organism at the kingdom level (<xref ref-type="fig" rid="F1">Figure 1B</xref>), we used a combination of human expertise and a generative AI based on the large language model (<xref ref-type="bibr" rid="B7">ChatGPT, 2023</xref>).</p>
<p>Next, we curated the ESKAPE phage genome dataset by downloading the reference genomes of bacteriophages from the NCBI Virus database. Here, we define the ESKAPE phages as bacteriophages that infect &#x200b;&#x200b;the WHO priority pathogens (<italic>Helicobacter pylori</italic>, <italic>Campylobacter jejuni</italic>, <italic>Salmonella enterica</italic>, <italic>Streptococcus pneumoniae</italic>, <italic>Haemophilus influenzae</italic>, <italic>Shigella flexneri</italic>) encompassing the ESKAPE (<italic>Enterococcus</italic> spp., <italic>Staphylococcus aureus</italic>, <italic>Klebsiella pneumoniae</italic>, <italic>Acinetobacter baumannii</italic>, <italic>Pseudomonas aeruginosa</italic>, and <italic>Enterobacter</italic> spp.) (<xref ref-type="bibr" rid="B43">Prioritization of pathogens to guide, 2019</xref>). Some pathogens, such as <italic>Klebsiella pneumoniae</italic> and <italic>Neisseria gonorrhoeae</italic>, were omitted from the list as they do not have associated bacteriophages in the NCBI virus database. All the RefSeq genomes with the specified host were downloaded for each pathogen species as whole genome sequences and protein sequences (downloaded 2022/10/18).</p>
<p>The genomes of each ESKAPE phage were analyzed using several biological features and statistical measures, which are important for understanding the variability in these phage datasets. The biological features included phage ID, phage type, and DNA type, and the statistical measures included GC content, AT content, GC/AT ratio, number of proteins, and sequence length (<xref ref-type="sec" rid="s10">Supplementary Table S1</xref>). These measures were summarized into the most common phage type, the most common DNA type, GC content, and the number of open-reading frames (ORF) for each host ESKAPE pathogen (<xref ref-type="table" rid="T1">Table 1</xref>).</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Summary statistics of phage genomes associated with the ESKAPE pathogens.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Host pathogen</th>
<th align="left">Most common phage type</th>
<th align="left">Most common DNA type</th>
<th align="left">GC content (mean &#xb1; s.d.)</th>
<th align="left">Number of ORF (mean &#xb1; s.d.)</th>
<th align="left">Genome length (mean &#xb1; s.d.)</th>
<th align="left">Most common lifestyle</th>
<th align="left">Probability of most common lifestyle (mean &#xb1; s.d.)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<italic>Acinetobacter baumannii</italic>
</td>
<td align="left">Obolenskvirus</td>
<td align="left">dsDNA</td>
<td align="left">39.12 &#xb1; 2.55</td>
<td align="left">84.76 &#xb1; 83.49</td>
<td align="left">59,042 &#xb1; 53,609</td>
<td align="left">Lytic</td>
<td align="left">0.52 &#xb1; 0.03</td>
</tr>
<tr>
<td align="left">
<italic>Campylobacter jejuni</italic>
</td>
<td align="left">Fletchervirus</td>
<td align="left">dsDNA</td>
<td align="left">35.07 &#xb1; 8.19</td>
<td align="left">54.92 &#xb1; 75.23</td>
<td align="left">82,425 &#xb1; 57,083</td>
<td align="left">Temperate</td>
<td align="left">0.51 &#xb1; 0.01</td>
</tr>
<tr>
<td align="left">
<italic>Escherichia coli</italic>
</td>
<td align="left">Gequatrovirus</td>
<td align="left">dsDNA</td>
<td align="left">43.90 &#xb1; 6.50</td>
<td align="left">75.70 &#xb1; 98.48</td>
<td align="left">52,765 &#xb1; 61,362</td>
<td align="left">Temperate</td>
<td align="left">0.56 &#xb1; 0.08</td>
</tr>
<tr>
<td align="left">
<italic>Enterococcus faecium</italic>
</td>
<td align="left">Siphoviridae</td>
<td align="left">dsDNA</td>
<td align="left">36.83 &#xb1; 4.41</td>
<td align="left">8.84 &#xb1; 35.86</td>
<td align="left">8,185 &#xb1; 29,593</td>
<td align="left">Temperate</td>
<td align="left">0.51 &#xb1; 0.01</td>
</tr>
<tr>
<td align="left">
<italic>Haemophilus influenzae</italic>
</td>
<td align="left">Hpunavirus</td>
<td align="left">dsDNA</td>
<td align="left">39.97 &#xb1; 0.05</td>
<td align="left">39.50 &#xb1; 3.54</td>
<td align="left">31,932 &#xb1; 599</td>
<td align="left">Temperate</td>
<td align="left">0.63 &#xb1; 0.04</td>
</tr>
<tr>
<td align="left">
<italic>Helicobacter pylori</italic>
</td>
<td align="left">Schmidvirus</td>
<td align="left">dsDNA</td>
<td align="left">37.44 &#xb1; 1.71</td>
<td align="left">17.09 &#xb1; 15.73</td>
<td align="left">14,224 &#xb1; 13,342</td>
<td align="left">Temperate</td>
<td align="left">0.52 &#xb1; 0.01</td>
</tr>
<tr>
<td align="left">
<italic>Pseudomonas aeruginosa</italic>
</td>
<td align="left">Pbunavirus</td>
<td align="left">dsDNA</td>
<td align="left">56.53 &#xb1; 8.00</td>
<td align="left">53.45 &#xb1; 67.50</td>
<td align="left">42,401 &#xb1; 50,165</td>
<td align="left">Temperate</td>
<td align="left">0.52 &#xb1; 0.03</td>
</tr>
<tr>
<td align="left">
<italic>Staphylococcus aureus</italic>
</td>
<td align="left">Kayvirus</td>
<td align="left">dsDNA</td>
<td align="left">32.67 &#xb1; 2.80</td>
<td align="left">85.50 &#xb1; 78.11</td>
<td align="left">63,602 &#xb1; 54,105</td>
<td align="left">Temperate</td>
<td align="left">0.60 &#xb1; 0.07</td>
</tr>
<tr>
<td align="left">
<italic>Streptococcus pneumoniae</italic>
</td>
<td align="left">Siphoviridae</td>
<td align="left">dsDNA</td>
<td align="left">39.09 &#xb1; 2.05</td>
<td align="left">42.51 &#xb1; 19.23</td>
<td align="left">29,383 &#xb1; 13,721</td>
<td align="left">Temperate</td>
<td align="left">0.53 &#xb1; 0.02</td>
</tr>
<tr>
<td align="left">
<italic>Salmonella enterica</italic>
</td>
<td align="left">Jerseyvirus</td>
<td align="left">dsDNA</td>
<td align="left">46.64 &#xb1; 6.12</td>
<td align="left">71.06 &#xb1; 75.33</td>
<td align="left">53,119 &#xb1; 53,774</td>
<td align="left">Lytic</td>
<td align="left">0.53 &#xb1; 0.04</td>
</tr>
<tr>
<td align="left">
<italic>Shigella flexneri</italic>
</td>
<td align="left">Tequatrovirus</td>
<td align="left">dsDNA</td>
<td align="left">43.14 &#xb1; 5.87</td>
<td align="left">117.44 &#xb1; 88.79</td>
<td align="left">78,388 &#xb1; 54,418</td>
<td align="left">Lytic</td>
<td align="left">0.59 &#xb1; 0.11</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>To visualize the statistical information, we created boxplots for two variables: phage genome lengths and the number of open reading frames (<xref ref-type="fig" rid="F2">Figures 2A, B</xref>). Additionally, we created scatter plots to illustrate the relationship between these two variables (<xref ref-type="fig" rid="F2">Figures 2C, D</xref>). Each data point on the plot represents every &#x2018;phage ID&#x2019;, with the phage genome lengths plotted along the <italic>x</italic>-axis and the number of open reading frames plotted along the <italic>y</italic>-axis. The scatter plots included data from all 11 species, where each species was displayed in a different color. The statistics from the <italic>E. coli</italic> phages were plotted separately due to the high number of data points (<xref ref-type="fig" rid="F2">Figure 2D</xref>). Then, we used a &#x2018;LinearRegression&#x2019; class to fit a model that represented the linear relationship. By using this model, we calculated the regression equation and the <italic>R</italic>
<sup>2</sup> value. The scatter plot with an overlayed regression line helped to identify whether the variables have a positive, negative, or no apparent relationship.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Summary statistics of the ESKAPE phage genome dataset. <bold>(A)</bold> Boxplot of phage genome lengths (bp) of phages by their host ESKAPE pathogen. <bold>(B)</bold> Boxplot of open reading frames of phages by their host ESKAPE pathogen. <bold>(C)</bold> Scatter plot of phage genome lengths (bp) versus open reading frames, colored by each host ESKAPE pathogen with an overlayed regression line. <bold>(D)</bold> Scatter plot of <italic>E. coli</italic> phage genome lengths (bp) versus open reading frames with an overlayed regression line.</p>
</caption>
<graphic xlink:href="fmolb-11-1395450-g002.tif"/>
</fig>
</sec>
<sec id="s2-2">
<title>Gene function analysis of ESKAPE phages</title>
<p>To observe the function of diverse genes in the ESKAPE phage genomes, the fasta files were processed with Biopython to extract gene information such as gene ID, virus ID, gene name, virus name, virus genus, and family. We extracted the gene keywords such as &#x2018;DNA&#x2019; or &#x2018;polymerase&#x2019; from the gene names. A vast majority of the gene names contain the keyword &#x2018;hypothetical&#x2019;. A hypothetical gene is a predicted gene that is likely to be expressed in organisms, but which has never been characterized for biochemical function (<xref ref-type="bibr" rid="B18">Galperin, 2001</xref>). Therefore, keywords related to hypothetical genes were excluded from the subsequent processing.</p>
<p>To visualize these gene keywords and names, we generated a keyword heatmap and a horizontal bar chart for each ESKAPE gene genome, respectively. The genes with the keyword &#x2018;putative&#x2019; also have unknown functions like hypothetical genes, but they have similar properties to already existing genes (<xref ref-type="bibr" rid="B1">Alexandre et al., 1988</xref>). Since they are assumed to be functional genes, this keyword was kept for further study. The visualization of these genes was done in three sets; the first set includes only functional genes excluding hypothetical and putative genes (<xref ref-type="sec" rid="s10">Supplementary Figure S1</xref>), the second set includes only putative genes (<xref ref-type="sec" rid="s10">Supplementary Figure S2</xref>), and the third set includes all functional and putative genes (<xref ref-type="sec" rid="s10">Supplementary Figure S3</xref>).</p>
<p>Subsequently, we summarized the gene functions into a heatmap to compare the relative magnitudes of genes co-occurring in the ESKAPE phage genomes (<xref ref-type="fig" rid="F3">Figure 3</xref>). Gene keywords from each ESKAPE gene genome were downloaded as CSV files, which were processed into &#x2018;word&#x2019; and &#x2018;value&#x2019; columns. To calculate the frequency of common words within 22 CSV files (11 CSV files were downloaded from each set: all genes and only functional genes), we used the &#x2018;Counter&#x2019; object. Each word was treated as a unique key and its occurrences are computed individually, for example, &#x2018;holin&#x2019; or &#x2018;putative holin&#x2019;. We created a list of regular expression patterns by selecting the 32 words with the highest frequencies from the extracted unique keys. Each target word in the list was processed using &#x27;\b{}\b&#x27; to indicate a word boundary in the regular expression pattern. These patterns were then used to search for words in the previous CSV files that fully matched the pattern. Finally, we merged all the data into a single data frame and generated the heatmap, which provided us with a visual pattern to easily grasp the relationships between the data.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Heatmap of co-occurring functional genes from the ESKAPE phage genomes.</p>
</caption>
<graphic xlink:href="fmolb-11-1395450-g003.tif"/>
</fig>
</sec>
<sec id="s2-3">
<title>Genome visualization and phage lifestyle prediction</title>
<p>We used a genome viewer to visualize the genomic architecture of each phage genome. In this process, the untranslated reference sequence genome fasta file and translated reference sequence protein fasta file were separated into every single IDs of phages. The translated reference sequence protein fasta file contains the information about IDs of phage and locations where the protein of each phage is translated. To simplify reference sequence protein fasta files, phage IDs were extracted from the untranslated reference sequence genome fasta files through Python. Using the phage IDs in the list, new files containing all translation sites per phage ID were created. With the DNA feature viewer algorithm, linear graphs were made with whole genome data while omitting hypothetical proteins (<xref ref-type="sec" rid="s10">Supplementary Figure S4</xref>).</p>
<p>PHACTS is a computational approach that classifies the lifestyle of bacteriophages (<xref ref-type="bibr" rid="B33">McNair et al., 2012</xref>). The lifestyle of a phage can be classified as virulent (lytic) or temperate (lysogenic). For the virulent and temperate types, a phage genome is annotated as either &#x2018;Lytic&#x2019; or &#x2018;Lysogenic&#x2019;, respectively, with a computed probability. PHACTS predicts the lifestyle of a bacteriophage based on the genome content through two training sets and the Random Forest method (<xref ref-type="bibr" rid="B22">Ho, 2024a</xref>; <xref ref-type="bibr" rid="B23">Ho, 2024b</xref>). One of the training sets of the query protein sequences was selected from the newly edited fasta files. The other training set is based on the known phage sequence data from PHANTOME (<xref ref-type="bibr" rid="B4">Batstone et al., 2022</xref>), providing a complete phage genome with two training sets. The Random Forest algorithm of PHACTS classifies the lifestyle of phages by creating multiple decision trees. Then, the known phage sequence data is loaded into decision trees through the bootstrapping method, which randomly picks data by resampling. Randomly selected query protein sequences from the newly edited fasta files are matched with the known phage sequence data to complete the decision trees. Finally, the tree with the highest computed probability is chosen as the final prediction model. To keep the accuracy and stable runtime, replicate iterations are executed 10 times. Among 10 iterative replicates, the sole consensus result values are determined as the confident data (<xref ref-type="bibr" rid="B33">McNair et al., 2012</xref>).</p>
</sec>
<sec id="s2-4">
<title>Pangenome analysis of ESKAPE phage genomes</title>
<p>We used IPGA (v1.09) to analyze, compare, and visualize the pangenome of ESKAPE phages (<xref ref-type="bibr" rid="B31">Liu et al., 2022</xref>). This web tool features a scoring system that evaluates the reliability of profiles generated by different pangenome methods. We ran several pangenome packages with the phages of <italic>Salmonella enterica</italic> as an initial trial. Using IPGA, we compared the pangenome profiles created by different methods and found that PEPPAN (<xref ref-type="bibr" rid="B58">Zhou et al., 2020</xref>) was the only pangenome software that performed well with the ESKAPE phage genomes as input data. Thus, all the other phage genomes of the ESKAPE pathogens were analyzed using PEPPAN as the pangenome analysis option (<xref ref-type="fig" rid="F4">Figure 4</xref>; <xref ref-type="sec" rid="s10">Supplementary Figure S5</xref>). As pangenome analysis requires four or more individual genomes, the phages of <italic>H. influenzae</italic> with only two individual genomes were excluded from the pangenome analysis. In addition, IPGA also implements several downstream comparative analysis modules and genome analysis modules, including Average Nucleotide Identity (ANI) that measure nucleotide-level genomic similarity between the coding regions of genomes (<xref ref-type="bibr" rid="B29">Konstantinidis and Tiedje, 2005</xref>).</p>
<p>MMseqs2 is a deep learning-based software to search and cluster huge sequence sets, with a highly efficient clustering module to group similar sequences into clusters (<xref ref-type="bibr" rid="B53">Steinegger and S&#xf6;ding, 2017</xref>). For clustering whole-genome sequences of these ESKAPE phages, we used a clustering module of MMseqs2 that is highly efficient at grouping similar sequences into clusters. It employs an iterative clustering approach, progressively merging sequences into clusters while optimizing a predefined objective function to achieve accurate and scalable clustering results. We created a database containing the genomes of phages against each ESKAPE pathogen and clustered these genomes with a minimum sequence identity of 70%.</p>
</sec>
<sec id="s2-5">
<title>Clustering of phage protein sequences by similarity</title>
<p>For clustering protein sequences, a clustering in linear time called the Linclust algorithm in MMseqs2 was used for fast clustering, which reduces the time complexity <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x007C;">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x007C;">
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the parameter that indicates the final number of clusters. The Linclust algorithm operates by generating a table that consists of the k-mer, the sequence identifier, and the sequence position, and sorting the table by k-mer to identify groups of sequencing sharing the same k-mer in quasi-linear time. Then, the longest sequence is selected as the center sequence among the sequences that share the same k-mer. The groups are merged around the center sequence and the sequence groups are compared by the global Hamming distance and gapped local sequence alignment. Finally, the representative sequence and aligned sequences are determined through the incremental greedy algorithm.</p>
<p>We used two settings of sequence identity thresholds (50% and 90%, respectively) for clustering as shown below. For the sequence identity of 50%, the mode was set as 1 which enables the alignment to cover at least 50% between query and target.</p>
<p>mmseqs easy-cluster examples/DB.fasta clusterRes tmp --min-seq-id 0.5 -c 0.5 --cov-mode 1</p>
<p>For the sequence identity of 90%, the mode was set as 0 which enables the alignment to cover at least 80% between query and target.</p>
<p>mmseqs easy-cluster examples/DB.fasta clusterRes tmp --min-seq-id 0.9 -c 0.8 --cov-mode 0.</p>
<p>We compared the results from these two sets with the different sequence identity thresholds. We determined that the sequence identity threshold of 90% was too stringent, thus the following structure analysis was conducted with the set with the sequence identity threshold of 50%. The representative sequences were extracted from the resulting MMseqs2 file for AlphaFold structure prediction.</p>
</sec>
<sec id="s2-6">
<title>AlphaFold-predicted structures of representative inhibitor phage proteins</title>
<p>AlphaFold is a deep learning-based program that could predict three-dimensional structures of proteins (<xref ref-type="bibr" rid="B26">Jumper et al., 2021</xref>). We used AlphaFold to gain insight into the structure of representative inhibitor phage proteins (<xref ref-type="fig" rid="F5">Figure 5</xref>; <xref ref-type="sec" rid="s10">Supplementary Figure S6</xref>). We performed AlphaFold using Google Colab and generated protein structure predictions from the representative protein sequences. To identify only the inhibitor proteins of lytic phages, we selectively screened lytic phages of the ESKAPE pathogens (<xref ref-type="sec" rid="s10">Supplementary Table S2</xref>) and identified the &#x2018;Phage ID&#x2019; of the lytic phages that corresponded to the &#x2018;Protein ID&#x2019; of the inhibitor phage proteins (<xref ref-type="sec" rid="s10">Supplementary Table S3</xref>).</p>
<p>Subsequently, the PDB files of the predicted protein structures were downloaded, and we observed the three-dimensional representation of protein structures with the Pymol, including alpha helices, beta sheets, and tertiary structures. In addition, we analyzed the temperature factor column in the PDB file describing the per-residue Local Distance Difference Test (pLDDT) of each residue (<xref ref-type="bibr" rid="B26">Jumper et al., 2021</xref>). The pLDDT corresponds to the model&#x2019;s estimate of its score on the local Distance Difference Test (lDDT-C&#x3b1;), which is a measure of local accuracy. These scores are represented on a spectrum, with higher certainty depicted in red and lower certainty shown in blue.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<sec id="s3-1">
<title>Bacteriophages are highly underrepresented in the databases</title>
<p>In the Protein Data Bank (PDB), we discovered that the experimental structures of bacteriophage proteins are highly underrepresented when classified at the kingdom level (<xref ref-type="fig" rid="F1">Figure 1B</xref>). The bar chart of the PDB entries by natural source organism shows that the kingdom of Animalia is highly represented, particularly biological model organisms such as <italic>Homo sapiens</italic> (3,364 entries) and <italic>Mus musculus</italic> (1,265 entries). Furthermore, the kingdoms that contain pathogens against humans such as Bacteria and Eukaryotic viruses are well represented in the PDB database with entries of 5,262 and 1,165, respectively. The least represented kingdoms are Protista, Archaea, and Prokaryotic viruses (i.e., bacteriophages) with entries of 294, 415, and 309, respectively. For the ESKAPE phages, we found only few to no entries for <italic>Acinetobacter baumannii</italic> (0), <italic>Campylobacter jejuni</italic> (0), <italic>Escherichia coli</italic> (73), <italic>Enterococcus faecium</italic> (0), <italic>Haemophilus influenzae</italic> (0), <italic>Helicobacter pylori</italic> (4), <italic>Pseudomonas aeruginosa</italic> (10), <italic>Salmonella enterica</italic> (12), <italic>Shigella flexneri</italic> (4), <italic>Staphylococcus aureus</italic> (8), <italic>Streptococcus pneumoniae</italic> (3). Even considering that the protein contents of these organisms differ vastly, as the full human genome contains 24,000 proteins while a typical phage genome contains fewer than 100 proteins (<xref ref-type="table" rid="T1">Table 1</xref>), this underrepresentation of phage proteins in the PDB is still striking. For example, eukaryotic viruses with a similar number of proteins have 1,165 entries as compared to 309 entries of prokaryotic viruses in the PDB.</p>
<p>For the genome database, this study reveals a paucity of complete phage genomes associated with the WHO priority list pathogens in the NCBI Virus database (<xref ref-type="fig" rid="F1">Figure 1C</xref>). For example, <italic>H. influenzae</italic>, a Gram-negative bacterium causing pneumonia, meningitis, or bloodstream infections (<xref ref-type="bibr" rid="B15">Fleischmann et al., 1995</xref>), has only two known complete phage genomes. Furthermore, these phages exhibit temperate lifestyles (<xref ref-type="table" rid="T2">Table 2</xref>), limiting their suitability for phage therapy applications. Despite the evident advantages of phage therapy, several challenges persist, especially in the realm of fundamental research. A notable challenge is the inadequate comprehension of the diversity of lytic bacteriophages within their natural habitats, such as human microbiomes. This scarcity of diverse lyric phage genomes poses a substantial impediment to the development of effective phage therapy treatments tailored to specific bacterial infections.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Summary statistics of lytic phage genomes associated with the ESKAPE pathogens.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Host pathogen</th>
<th align="left">Most common phage type</th>
<th align="left">Most common DNA type</th>
<th align="left">GC content (mean &#xb1; s.d.)</th>
<th align="left">Number of ORF (mean &#xb1; s.d.)</th>
<th align="left">Genome length (mean &#xb1; s.d.)</th>
<th align="left">Most common lifestyle</th>
<th align="left">Probability of most common lifestyle (mean &#xb1; s.d.)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<italic>Acinetobacter baumannii</italic>
</td>
<td align="left">Friunavirus</td>
<td align="center">dsDNA</td>
<td align="left">38.91 &#xb1; 2.69</td>
<td align="left">112.40 &#xb1; 101.31</td>
<td align="left">78,735 &#xb1; 64,666</td>
<td align="left">Lytic</td>
<td align="left">0.53 &#xb1; 0.04</td>
</tr>
<tr>
<td align="left">
<italic>Campylobacter jejuni</italic>
</td>
<td align="left">Fletchervirus</td>
<td align="center">dsDNA</td>
<td align="left">37.01 &#xb1; 8.44</td>
<td align="left">48.10 &#xb1; 78.53</td>
<td align="left">88,282 &#xb1; 58,035</td>
<td align="left">Lytic</td>
<td align="left">0.51 &#xb1; 0.01</td>
</tr>
<tr>
<td align="left">
<italic>Escherichia coli</italic>
</td>
<td align="left">Gequatrovirus</td>
<td align="center">dsDNA</td>
<td align="left">42.28 &#xb1; 6.25</td>
<td align="left">73.97 &#xb1; 97.41</td>
<td align="left">61,027 &#xb1; 68,482</td>
<td align="left">Lytic</td>
<td align="left">0.58 &#xb1; 0.10</td>
</tr>
<tr>
<td align="left">
<italic>Enterococcus faecium</italic>
</td>
<td align="left">Siphoviridae</td>
<td align="center">dsDNA</td>
<td align="left">37.16 &#xb1; 4.07</td>
<td align="left">7.71 &#xb1; 33.24</td>
<td align="left">9,579 &#xb1; 33,455</td>
<td align="left">Lytic</td>
<td align="left">0.51 &#xb1; 0.01</td>
</tr>
<tr>
<td align="left">
<italic>Haemophilus influenzae</italic>
</td>
<td align="left">N/A</td>
<td align="center">N/A</td>
<td align="left">N/A</td>
<td align="left">N/A</td>
<td align="left">N/A</td>
<td align="left">N/A</td>
<td align="left">N/A</td>
</tr>
<tr>
<td align="left">
<italic>Helicobacter pylori</italic>
</td>
<td align="left">Schmidvirus</td>
<td align="center">dsDNA</td>
<td align="left">37.77 &#xb1; 1.65</td>
<td align="left">6.90 &#xb1; 12.40</td>
<td align="left">5,579 &#xb1; 10,797</td>
<td align="left">Lytic</td>
<td align="left">0.51 &#xb1; 0.01</td>
</tr>
<tr>
<td align="left">
<italic>Pseudomonas aeruginosa</italic>
</td>
<td align="left">Pbunavirus</td>
<td align="center">dsDNA</td>
<td align="left">56.26 &#xb1; 8.16</td>
<td align="left">35.36 &#xb1; 46.91</td>
<td align="left">31,593 &#xb1; 36,682</td>
<td align="left">Lytic</td>
<td align="left">0.52 &#xb1; 0.03</td>
</tr>
<tr>
<td align="left">
<italic>Staphylococcus aureus</italic>
</td>
<td align="left">Kayvirus</td>
<td align="center">dsDNA</td>
<td align="left">30.83 &#xb1; 2.66</td>
<td align="left">107.89 &#xb1; 88.85</td>
<td align="left">94,240 &#xb1; 61,839</td>
<td align="left">Lytic</td>
<td align="left">0.57 &#xb1; 0.06</td>
</tr>
<tr>
<td align="left">
<italic>Streptococcus pneumoniae</italic>
</td>
<td align="left">Cepunavirus</td>
<td align="center">dsDNA</td>
<td align="left">37.76 &#xb1; 4.25</td>
<td align="left">29.60 &#xb1; 28.04</td>
<td align="left">17,318 &#xb1; 15,731</td>
<td align="left">Lytic</td>
<td align="left">0.52 &#xb1; 0.04</td>
</tr>
<tr>
<td align="left">
<italic>Salmonella enterica</italic>
</td>
<td align="left">Kuttervirus</td>
<td align="center">dsDNA</td>
<td align="left">45.48 &#xb1; 6.37</td>
<td align="left">12.52 &#xb1; 27.69</td>
<td align="left">66,567 &#xb1; 61,489</td>
<td align="left">Lytic</td>
<td align="left">0.53 &#xb1; 0.04</td>
</tr>
<tr>
<td align="left">
<italic>Shigella flexneri</italic>
</td>
<td align="left">Tequatrovirus</td>
<td align="center">dsDNA</td>
<td align="left">41.46 &#xb1; 5.76</td>
<td align="left">133.92 &#xb1; 96.42</td>
<td align="left">96,430 &#xb1; 57,962</td>
<td align="left">Lytic</td>
<td align="left">0.62 &#xb1; 0.12</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Boxplots provided us with a concise overview of statistics (distribution, median, quartiles, and presence of outliers), allowing for easy comparison and interpretation (<xref ref-type="fig" rid="F2">Figure 2</xref>). The boxplots of the phage genome lengths by the host show that most phages have genome lengths below 50&#xa0;kbp (<xref ref-type="fig" rid="F2">Figure 2A</xref>). Notably, the phages of <italic>E. coli</italic>, <italic>P. aeruginosa</italic>, and <italic>S. enterica</italic> have high variations in the genome length, with some phages reaching above 200&#xa0;kbp in length (<xref ref-type="bibr" rid="B36">Michniewski et al., 2021</xref>). The boxplots of the number of ORFs by the host show that most phages have protein lengths below 100 bp (<xref ref-type="fig" rid="F2">Figure 2B</xref>). Notably, the phages of <italic>E. coli</italic> and <italic>P. aeruginosa</italic> have high variations in the protein length, with some proteins reaching above 400 bp in length. This result shows that most ESKAPE phages have small genome lengths packed with small proteins (<xref ref-type="bibr" rid="B17">Fremin et al., 2022</xref>; <xref ref-type="bibr" rid="B51">Silpe et al., 2023</xref>). The scatter plots with an overlayed regression line show that the relationship between the genome length and the number of ORFs is positive, but this relationship only has a coefficient of determination, or <italic>R</italic>
<sup>2</sup>, value of 0.63 for the ESKAPE phages, except that of a much lower value of 0.09 for the <italic>E. coli</italic> phages as shown in <xref ref-type="fig" rid="F2">Figures 2C, D</xref>, respectively. This <italic>R</italic>
<sup>2</sup> value measures the goodness of fit of this regression model, and it shows that only 63% of the variance in the number of ORFs can be explained by the genome length variables in the ESKAPE phages, except in the <italic>E. coli</italic> phages with a much lower value at 9%. The variation of the number of ORFs versus the genome length appears to be much higher in the <italic>E. coli</italic> phages, with only many small phages having a larger number of proteins than expected by the regression model, and <italic>vice versa</italic> (<xref ref-type="fig" rid="F2">Figure 2D</xref>).</p>
</sec>
<sec id="s3-2">
<title>Some ESKAPE phages have genes with antimicrobial activities</title>
<p>As the first exploratory analysis, we created heatmaps visualizing the top keywords in the annotations of the ESKAPE phage genomes. We searched for keywords containing &#x2018;inhibit&#x2019; or &#x2018;anti&#x2019; that are potentially related to the antimicrobial activities of phages against their hosts. The ESKAPE phage with the most antimicrobial keywords such as &#x2018;inhibit&#x2019; or &#x2018;anti&#x2019; within the functionally validated genes was associated with <italic>E. coli</italic> (<xref ref-type="sec" rid="s10">Supplementary Figure S1</xref>). Even after adjusting for the large dataset of available genomes, the <italic>E. coli</italic> phages have 10 or 20 times more genes with the antimicrobial keywords as compared to the <italic>P. aeruginosa</italic> phages with half the available genomes. The phages of <italic>S. enterica</italic> contain the next most abundant genes associated with antimicrobial activities. Notably, the phages of <italic>H. pylori</italic> and <italic>C. jeuni</italic> have almost no functional genes with antimicrobial activities, which may be explained by the lack of available phage genomes in the NCBI Virus database (<xref ref-type="fig" rid="F1">Figure 1</xref>). However, the phage genomes of <italic>C. jejuni</italic> contain a small number of putative antimicrobial genes, including &#x2018;anti-holin&#x2019;, in the putative gene list (<xref ref-type="sec" rid="s10">Supplementary Figure S2</xref>).</p>
<p>This finding remains the same when the putative genes were included in the analysis (<xref ref-type="sec" rid="s10">Supplementary Figure S2</xref>). The putative genes containing the keywords &#x2018;inhibit&#x2019; or &#x2018;anti&#x2019; show the potential for novel antimicrobial activities from the ESKAPE phages. For example, the &#x2018;anti-proliferative&#x2019; gene only appears in the list of putative genes, and, interestingly, this gene may be involved in the regulation of cell growth and development (<xref ref-type="bibr" rid="B27">Kalie et al., 2007</xref>; <xref ref-type="bibr" rid="B57">Zhang et al., 2023</xref>). Subsequently, we analyzed the most frequent co-occurrence of functional proteins in the ESKAPE phage genomes (<xref ref-type="fig" rid="F3">Figure 3</xref>). The heatmap shows the top 30 functional proteins that are shared between the ESKAPE phage genomes in this study. This heatmap shows that most annotations are related either to the replication function of bacteriophages, such as polymerase and helicase, or to the assembly function of bacteriophages, such as capsid and tail.</p>
<p>Finally, we analyzed the gene content of these phages further examining individual genes through bar graphs (<xref ref-type="sec" rid="s10">Supplementary Figure S3</xref>). The bar charts show the 30 most frequent genes in each ESKAPE phage. These bar charts reveal some genes of interest may have inhibitory functions. Several ESKAPE phages have several antimicrobial genes that may disrupt the vital functions of host bacteria. For example, <italic>A. baumannii</italic>, <italic>E. coli</italic>, <italic>S. enterica</italic>, and <italic>S. flexneri</italic> have anti-sigma factors that bind to sigma factors and inhibit transcriptional activity in regulating prokaryotic gene expression (<xref ref-type="bibr" rid="B38">Paget, 2015</xref>). Moreover, <italic>E. coli</italic>, <italic>S. enterica</italic>, and <italic>S. flexneri</italic> have host polymerase inhibitors that may disrupt the transcriptional activities of host bacteria (<xref ref-type="bibr" rid="B42">Pilotto et al., 2021</xref>). These genes are likely to be antibacterial as the disruption of transcription leads to cell cycle arrest (<xref ref-type="bibr" rid="B34">Merrikh et al., 2023</xref>). These phages also possess other genes that are involved in counter-defensive activities against the host defense mechanisms, such as anti-restriction genes (<xref ref-type="bibr" rid="B52">Spoerel et al., 1979</xref>), host protease inhibitors (<xref ref-type="bibr" rid="B35">Meyn et al., 1977</xref>), anti-repressor (<xref ref-type="bibr" rid="B24">Horiuchi et al., 1974</xref>), and anti-crispr genes (<xref ref-type="bibr" rid="B6">Bondy-Denomy et al., 2012</xref>; <xref ref-type="bibr" rid="B39">Park et al., 2022a</xref>; <xref ref-type="bibr" rid="B40">Park et al., 2022b</xref>; <xref ref-type="bibr" rid="B48">Shim, 2022</xref>).</p>
<sec id="s3-2-1">
<title>Databases have a limited number of lytic ESKAPE phages</title>
<p>We visualized the genomic architecture of the ESKAPE phages to explore the gene content. The genome maps of these ESKAPE phages revealed that many of these phages have similar gene content, and many of these phages are nearly identical. Remarkably, these phage genomes also share taxonomic relationships, resulting in even fewer unique genomes for potential therapeutic use. For instance, our analysis indicates two phage genomes infecting <italic>H. influenzae</italic> are nearly identical in gene content (<xref ref-type="sec" rid="s10">Supplementary Figure S4</xref>).</p>
<p>Given the lack of antimicrobial genes in the ESKAPE phage genomes, we explored the lifestyle of these phages (<xref ref-type="table" rid="T2">Table 2</xref>; <xref ref-type="sec" rid="s10">Supplementary Table S2</xref>). Identifying the lifestyle of a phage traditionally relies on labor-intensive and costly culturing techniques. However, these methods are not only time-consuming but also impractical for phage genomes derived from environmental sequencing. We used a computational approach to predict the lifestyle of the ESKAPE phages (<xref ref-type="bibr" rid="B33">McNair et al., 2012</xref>). Phages demonstrate two main lifestyles: virulent (lytic) and temperate (lysogenic). This computational approach leverages the gene content inherent to phage genomes to predict the lifestyle of a phage.</p>
<p>According to the computational prediction, there is only a limited number of lytic phages against all the ESKAPE pathogens (<xref ref-type="sec" rid="s10">Supplementary Table S2</xref>). For example, the phages of <italic>C. jejuni</italic>, <italic>E. faecium,</italic> and <italic>H. pylori</italic> only have one lytic phage in the database. Other ESKAPE pathogens such as <italic>A. baumannii</italic> and <italic>S. pneumoniae</italic> also have four and two lytic phages, respectively. More importantly, <italic>H. influenzae</italic> has no lytic phages that are predicted by the computational method. The medical importance of bacteriophages lies in their potential applications in phage therapy, a field that explores the use of bacteriophages to combat bacterial infections. Bacteriophages exhibit specificity in targeting bacterial strains, making them attractive candidates for precision medicine in treating bacterial infections. Their ability to infect and lyse bacteria provides a natural and tailored approach to controlling pathogenic bacteria, including antibiotic-resistant strains. The biological features of the lytic phages were summarized into the most common phage type, the most common DNA type, GC content, and the number of open-reading frames (ORF) for each host ESKAPE pathogen (<xref ref-type="table" rid="T2">Table 2</xref>). The curated dataset of bacteriophages with lytic lifestyles that are potential candidates for phage therapy is shared for each class of ESKAPE pathogen (<xref ref-type="sec" rid="s10">Supplementary Table S2</xref>).</p>
</sec>
</sec>
<sec id="s3-3">
<title>Pangenome analysis shows the ESKAPE phages share many core genes</title>
<p>For the ESKAPE phages that share the same host, we used the clustering method to automatically cluster their genomes by similarity at the nucleotide level. When we clustered these phages with the minimum sequence identity of 70%, it revealed that the number of individual phages can be reduced into a small number of clusters. For example, 263 individual phages infecting <italic>A. baumannii</italic> can be reduced into 3 clusters with the clustering method (<xref ref-type="fig" rid="F4">Figure 4A</xref>). This shows that the NCBI virus database only contains 3 unique phages against the high-priority pathogen of <italic>A. baumannii</italic>. Similarly, the genome maps of phages infecting <italic>A. baumannii</italic> can also be divided into a few categories based on visualization (<xref ref-type="sec" rid="s10">Supplementary Figure S4</xref>). Notably, many phages share the same gene content and even the same gene arrangement. However, the public databases do not compute the similarity between these phages as there is no consensus on how to classify these phages into different &#x2018;strains&#x2019;.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Pangenome analysis of the ESKAPE pathogens. <bold>(A)</bold> Average Nucleotide Identity (ANI) in <italic>S. pneumoniae</italic> phages. <bold>(B)</bold> Core genome clusters in <italic>S. pneumoniae</italic> phages. <bold>(C)</bold> Pangenome clusters in <italic>S. pneumoniae</italic> phages. <bold>(D)</bold> Set sizes and intersection sizes in <italic>S. pneumoniae</italic> phages.</p>
</caption>
<graphic xlink:href="fmolb-11-1395450-g004.tif"/>
</fig>
<p>The objective of pangenome analysis is to assess the diversity of all genes and genomic structures across genomes within a particular clade. The primary and pivotal step in this analysis involves clustering orthologous genes. These gene clusters are subsequently categorized into three groups based on their presence in the specified sets of genomes: core genes, accessory genes, and unique genes. Multiple packages or web services have been developed for pangenome analyses of eukaryotic and prokaryotic genomes (<xref ref-type="bibr" rid="B30">Li et al., 2003</xref>; <xref ref-type="bibr" rid="B16">Fouts et al., 2012</xref>; <xref ref-type="bibr" rid="B45">Santos et al., 2013</xref>; <xref ref-type="bibr" rid="B37">Page et al., 2015</xref>; <xref ref-type="bibr" rid="B8">Chen et al., 2018</xref>; <xref ref-type="bibr" rid="B13">Ding et al., 2018</xref>; <xref ref-type="bibr" rid="B14">Emms and Kelly, 2019</xref>; <xref ref-type="bibr" rid="B10">DeSalle et al., 2020</xref>; <xref ref-type="bibr" rid="B20">Gautreau et al., 2020</xref>; <xref ref-type="bibr" rid="B55">Tonkin-Hill et al., 2020</xref>; <xref ref-type="bibr" rid="B58">Zhou et al., 2020</xref>; <xref ref-type="bibr" rid="B19">Galperin et al., 2021</xref>), but we found no method that is specifically designed for phage genomes. We chose an integrated prokaryotes genome and pangenome analysis web service called IPGA that allows phage genomes as input for pangenome analysis, downstream analysis, and visualization of the target genomes (<xref ref-type="bibr" rid="B31">Liu et al., 2022</xref>).</p>
<p>The first step of IPGA was a quality control module that removes all low-quality genomes and performs a taxonomic assignment for each genome. IPGA then predicted genes of all filtered genomes and used them as the input of the pangenome analysis module (<xref ref-type="fig" rid="F4">Figure 4</xref>; <xref ref-type="sec" rid="s10">Supplementary Figure S5</xref>). Not all phage genomes of the ESKAPE pathogen had pangenome analysis results, such as <italic>H. influenzae</italic> and <italic>E. faecium</italic>, due to the lack of individual genomes. The ESKAPE phages of <italic>E. coli</italic> and <italic>S. pneumoniae</italic> failed to give any meaningful results due to the large input dataset. The pangenome results show that the ESKAPE phage genomes have a high number of core genes that are shared as the intersection between the phages infecting the same host. For example, the core genome clusters in <italic>S. pneumoniae</italic> phages decrease rapidly (<xref ref-type="fig" rid="F4">Figure 4B</xref>) while the pangenome clusters in <italic>S. pneumoniae</italic> phages increase slowly (<xref ref-type="fig" rid="F4">Figure 4C</xref>). This relative difference in the incline and decline rates indicates that more core genes are shared between the phage genomes of this pathogen than the accessory or unique genes. The set sizes and intersection sizes also reflect the size of the core genome versus the size of the pangenome in the <italic>S. pneumoniae</italic> phages (<xref ref-type="fig" rid="F4">Figure 4D</xref>). This trend is observed in the pangenome analysis of the other ESKAPE phages, including <italic>C. jejuni</italic>, <italic>H. pylori</italic>, and <italic>S. flexneri</italic> (<xref ref-type="sec" rid="s10">Supplementary Figure S5</xref>).</p>
<p>We also conducted the downstream comparative genomic analysis modules on the filtered genomes and gene clusters, including the phylogenetic analysis module, core gene allele analysis, and average nucleotide identity (ANI) calculation module. Average Nucleotide Identity (ANI) is a measure of nucleotide-level genomic similarity between the coding regions of two genomes. The ANI statistics show that many phage genomes share high genomic similarity at the nucleotide level. For example, the ANI analysis of <italic>S. pneumoniae</italic> phages shows that there are 6 clusters of phage genomes with genomic similarity below 70% at the nucleotide level (<xref ref-type="fig" rid="F4">Figure 4A</xref>).</p>
<sec id="s3-3-1">
<title>Protein structure analyses show the underexplored structure landscape</title>
<p>After the pangenome analysis of the ESKAPE phage genomes, we clustered the functional and putative protein sequences with antimicrobial activities. Subsequently, we predicted the three-dimensional protein structure of a representative protein from each cluster. When these representative proteins were visualized, we found that these proteins share several secondary and tertiary structural components despite being in different clusters in terms of genetic sequence (<xref ref-type="sec" rid="s10">Supplementary Figure S6</xref>). For example, the inhibitors of host energy from <italic>S. enterica</italic> phages share similar tertiary structures (<xref ref-type="fig" rid="F5">Figure 5A</xref>). Furthermore, the inhibitors of host transcription from <italic>E. coli</italic> phages and <italic>S. enterica</italic> phages also share similar tertiary structures (<xref ref-type="fig" rid="F5">Figure 5B</xref>). Also, the inhibitors of host translation from <italic>S. enterica</italic> phages follow the same trend (<xref ref-type="fig" rid="F5">Figure 5C</xref>). Interestingly, small proteins such as the inhibitors of the host toxin/antitoxin system from <italic>S. enterica</italic> and the anti-sigma factors from <italic>E. coli</italic> phages have unique structures (<xref ref-type="fig" rid="F5">Figure 5D</xref>). This protein structure analysis of the representative proteins with potential antimicrobial activities reveals the underexplored surface of the protein structure landscape, which are small proteins with diverse structural components (<xref ref-type="fig" rid="F1">Figure 1A</xref>).</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>AlphaFold-predicted inhibitor proteins from lytic phages of the ESKAPE pathogens. <bold>(A)</bold> Inhibitors of host energy from <italic>S. enterica</italic> phages. <bold>(B)</bold> Inhibitors of host transcription from <italic>E. coli</italic> phages and <italic>S. enterica</italic> phages. <bold>(C)</bold> Inhibitors of host translation from <italic>S. enterica</italic> phages. <bold>(D)</bold> Inhibitor of host toxin/antitoxin system from <italic>S. enterica</italic> and anti-sigma factor from <italic>E. coli</italic> phages. The pLDDT score represents the model&#x2019;s estimate of its performance on the Local Distance Difference Test.</p>
</caption>
<graphic xlink:href="fmolb-11-1395450-g005.tif"/>
</fig>
<p>AlphaFold generates a per-residue model confidence score known as pLDDT, ranging from 0 to 100, where regions scoring below 50 pLDDT may lack structural integrity when considered in isolation. Notably, the predicted structures of all the proteins exhibit consistently low confidence levels, predominantly below the threshold of 50 pLDDT across various regions (<xref ref-type="fig" rid="F5">Figure 5</xref>). This pattern suggests limitations in AlphaFold&#x2019;s ability to accurately predict the structures of phage proteins. A probable contributing factor could be the relatively small number of phage proteins included in the training dataset of this computational model. This observation underscores the importance of expanding the diversity of proteins represented in training datasets to improve the accuracy and applicability of structure prediction algorithms like AlphaFold. Particularly, there is a notable gap in the exploration of antimicrobial proteins derived from lytic phages, which are increasingly recognized for their potential role in phage therapy (<xref ref-type="bibr" rid="B49">Shim, 2023</xref>). Despite recent findings highlighting their significance, these proteins remain understudied, as evidenced by the limited confidence in their predicted structures.</p>
</sec>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<p>The underrepresentation of bacteriophages in biological databases despite being the most abundant biological entities in the biosphere raises concerns that a significant portion of this vital biosphere remains unexplored. This study seeks to quantify the extent of the underrepresentation of phage data within public databases and to discuss the implications of such biased datasets. Specifically, it aims to assess the impact on medically relevant fields such as phage therapy, as well as on data-driven computational models like deep learning-based protein structure prediction programs. In this study, we revealed the extent of underrepresentation of bacteriophages in the Reference Sequence (RefSeq) of the NCBI Virus. This database provides a comprehensive and non-redundant set of sequences and a stable reference for genome annotation, forming a foundation for medical and functional studies. We examined the phage genomes infecting 11 bacterial species categorized under the ESKAPE acronym, designated as the top priority pathogens by the WHO for new drug development due to widespread multidrug resistance (<xref ref-type="bibr" rid="B49">Shim, 2023</xref>). The gene content analysis shows that several ESKAPE pathogens, such as <italic>C. jejuni</italic>, <italic>E. faecium</italic>, and <italic>H. influenzae</italic>, have only a few phages that are likely to have antimicrobial activities. Notably, <italic>H. pylori</italic> has no unique phages with functionally annotated genes encoding for antimicrobial activities (<xref ref-type="sec" rid="s10">Supplementary Figure S1</xref>). More importantly, these bacteria have no putative antimicrobial genes (<xref ref-type="sec" rid="s10">Supplementary Figure S2</xref>).</p>
<p>Next, we conducted the lifestyle analysis of these ESKAPE phages to confirm that there is only a handful of lytic phages for most pathogens (<xref ref-type="table" rid="T2">Table 2</xref>). The lifestyle of bacteriophages holds profound implications across various fields, including phage therapy, genomics, and microbiology. In the lytic lifecycle, the phage infects a bacterial host cell, hijacks its machinery to replicate its own genetic material and produce progeny phages, and ultimately causes the host cell to burst, releasing the newly formed phages to infect neighboring cells. This process results in the immediate destruction of the host cell (<xref ref-type="bibr" rid="B50">Shim et al., 2021</xref>). On the other hand, in the lysogenic lifecycle, the phage inserts its genetic material into the host cell&#x2019;s genome, becoming a prophage. The prophage is replicated along with the host cell&#x2019;s DNA during cell division, remaining latent within the host cell without causing immediate harm. Under certain conditions, such as exposure to stressors, the prophage may become activated, entering the lytic cycle and causing the host cell to lyse (<xref ref-type="bibr" rid="B28">Knowles et al., 2016</xref>). The lack of lytic phages in the genomic repertoire reduces the number of treatment options in phage therapy (<xref ref-type="bibr" rid="B49">Shim, 2023</xref>). Phage therapy, leveraging the natural predation of bacteriophages on bacteria, is gaining traction as a viable alternative or complementary strategy to traditional antibiotics (<xref ref-type="bibr" rid="B56">Young and Gill, 2015</xref>; <xref ref-type="bibr" rid="B21">Gordillo et al., 2019</xref>). As antibiotics struggle to maintain efficacy against evolving bacterial defenses, bacteriophages offer a tailored and evolving solution, capable of adapting to bacterial mutations (<xref ref-type="bibr" rid="B49">Shim, 2023</xref>).</p>
<p>Given the lack of phages with antimicrobial activities, we used the pangenome analysis to reveal how many of the ESKAPE phages are unique in terms of nucleotide identity and gene content. From whole-genome clustering and average nucleotide identity computation, we found that the ESKAPE phages of the same host share many core genes. Additionally, the ESKAPE phages form a small number of clusters in the whole-genome phylogenetic analysis. The average nucleotide identity analysis also highlights a deficiency in unique phage genomes within the reference sequence database, which ideally should offer a comprehensive and non-redundant collection of sequences. To enhance the classification of phage genomes, we propose the adoption of clustering methods that prioritize gene content rather than gene arrangement, considering the rapid evolutionary mechanism of phages.</p>
<p>Last, we used the deep learning-based structure program to predict the three-dimensional structures of the representative proteins with potential antimicrobial activities in the ESKAPE phages. After clustering with the multiple sequence alignment method, we selected representative sequences for each cluster (<xref ref-type="sec" rid="s10">Supplementary Figure S6</xref>). From the visualization, we found that the ESKAPE phages only possess a few antimicrobial proteins with unique structures (<xref ref-type="fig" rid="F5">Figure 5</xref>). Notably, the unique structures of these antimicrobial proteins are mostly small proteins such as anti-sigma proteins or toxin/antitoxin proteins. The protein structure landscape is not complete without exploring the small proteins from phages (<xref ref-type="fig" rid="F1">Figure 1A</xref>). These small proteins derived from phages are increasingly recognized for their remarkable diversity (<xref ref-type="bibr" rid="B26">Jumper et al., 2021</xref>; <xref ref-type="bibr" rid="B40">Park et al., 2022b</xref>).</p>
<p>The sequencing of bacteriophages presents distinct challenges compared to other microbial entities, such as bacteria and archaea. Bacteriophages exhibit an extraordinary degree of genetic diversity. Unlike bacteria and archaea, for which reference genomes are relatively abundant, the lack of comprehensive reference databases for bacteriophages complicates the alignment and assembly processes during sequencing (<xref ref-type="bibr" rid="B46">Shim, 2019a</xref>). Bacteriophages exhibit rapid rates of evolution, leading to genomic changes that occur over short periods (<xref ref-type="bibr" rid="B47">Shim, 2019b</xref>). This evolutionary dynamism can complicate the assembly and analysis of phage genomes, particularly when attempting to capture the full spectrum of genetic variation. Environmental samples often contain a multitude of different bacteriophages, and some phages can even infect the same bacterial host. This phenomenon, known as mixed infections, introduces complexities in the interpretation of sequencing data, making it challenging to distinguish individual phage genomes in a mixture (<xref ref-type="bibr" rid="B32">Mathew et al., 2019</xref>). Other challenges arise from a combination of factors, including limitations in phage-specific extraction protocols and computational tools. Addressing these challenges in bacteriophage sequencing requires the development of specialized protocols, bioinformatics tools, and reference databases tailored to the unique characteristics of these viral entities.</p>
<p>Lastly, we emphasize the implications of biased datasets on data-driven computational models, particularly deep learning-based tools that have been trained on the currently available datasets for decision-making and generative processes (<xref ref-type="bibr" rid="B26">Jumper et al., 2021</xref>; <xref ref-type="bibr" rid="B40">Park et al., 2022b</xref>). The challenge of underrepresented datasets poses a substantial concern, particularly in the context of machine learning models. This issue becomes particularly pronounced when the training data used to train these models is biased, leading to skewed and potentially inaccurate results during the model&#x2019;s predictive or classification tasks. In machine learning, the efficacy and reliability of a model are highly contingent on the quality and representativeness of the data it is trained on. When certain groups or categories within the dataset are underrepresented, the model may not adequately learn the patterns and characteristics associated with those groups (<xref ref-type="bibr" rid="B11">Deviyani, 2022</xref>; <xref ref-type="bibr" rid="B25">Jones et al., 2024</xref>). This lack of representation can lead to a biased understanding of the data, resulting in a model that is less capable of making accurate predictions or classifications for the underrepresented groups. The consequences of biased training data are far-reaching and can manifest in various ways. For instance, in predictive modeling of protein structures, the model may struggle to generalize well to instances that belong to underrepresented classes, leading to poor performance for phage proteins.</p>
<p>Bacteriophages have undergone a paradigm shift in recent years. Initially perceived as having limited medical relevance compared to bacteria, bacteriophages are now recognized as essential players in various medical and therapeutic contexts. This shift in perspective is primarily attributed to a deeper understanding of the intricate relationships between bacteriophages and their bacterial hosts. Recent advances in sequencing technologies and methodologies are continually improving our ability to overcome these obstacles and unravel the genetic intricacies of bacteriophages (<xref ref-type="bibr" rid="B41">Park et al., 2023</xref>). In our future research endeavors, we are dedicated to tackling the issue of underrepresentation of phage-related data through comprehensive genome sampling initiatives, leveraging the latest advancements in long-read sequencing technology. By harnessing the capabilities of long-read sequencing technologies, we aim to overcome the inherent limitations of traditional short-read sequencing methods, which often fail to capture the full complexity and diversity of phage genomes.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://github.com/hshimlab">https://github.com/hshimlab</ext-link>.</p>
</sec>
<sec id="s6">
<title>Author contributions</title>
<p>JL: Writing&#x2013;original draft, Visualization, Validation, Formal Analysis. BH: Writing&#x2013;review and editing, Writing&#x2013;original draft, Formal Analysis, Data curation. HS: Writing&#x2013;review and editing, Writing&#x2013;original draft, Visualization, Validation, Supervision, Software, Resources, Project administration, Methodology, Investigation, Funding acquisition, Formal Analysis, Data curation, Conceptualization.</p>
</sec>
<sec sec-type="funding-information" id="s7">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. The research and development activities described in this study were funded by GUGC and California State University, Fresno.</p>
</sec>
<ack>
<p>We thank the members of the Centre for Biotech Data Science at GUGC and the Department of Biology at California State University, Fresno for their support and motivation.</p>
</ack>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s10">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fmolb.2024.1395450/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fmolb.2024.1395450/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alexandre</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Guyaux</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Murphy</surname>
<given-names>N. B.</given-names>
</name>
<name>
<surname>Coquelet</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Pays</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Steinert</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>1988</year>). <article-title>Putative genes of a variant-specific antigen gene transcription unit in Trypanosoma brucei</article-title>. <source>Mol. Cell. Biol.</source> <volume>8</volume>, <fpage>2367</fpage>&#x2013;<lpage>2378</lpage>. <pub-id pub-id-type="doi">10.1128/mcb.8.6.2367</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Al-Shayeb</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Sachdeva</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.-X.</given-names>
</name>
<name>
<surname>Ward</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Munk</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Devoto</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Clades of huge phages from across Earth&#x2019;s ecosystems</article-title>. <source>Nature</source> <volume>578</volume>, <fpage>425</fpage>&#x2013;<lpage>431</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-020-2007-4</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<collab>Antimicrobial Resistance Collaborators</collab> (<year>2022</year>). <article-title>Global burden of bacterial antimicrobial resistance in 2019: a systematic analysis</article-title>. <source>Lancet</source> <volume>399</volume>, <fpage>629</fpage>&#x2013;<lpage>655</lpage>. <pub-id pub-id-type="doi">10.1016/S0140-6736(21)02724-0</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Batstone</surname>
<given-names>R. T.</given-names>
</name>
<name>
<surname>Burghardt</surname>
<given-names>L. T.</given-names>
</name>
<name>
<surname>Heath</surname>
<given-names>K. D.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Phenotypic and genomic signatures of interspecies cooperation and conflict in naturally occurring isolates of a model plant symbiont</article-title>. <source>Proc. Biol. Sci.</source> <volume>289</volume>, <fpage>20220477</fpage>. <pub-id pub-id-type="doi">10.1098/rspb.2022.0477</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Berman</surname>
<given-names>H. M.</given-names>
</name>
<name>
<surname>Westbrook</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Gilliland</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Bhat</surname>
<given-names>T. N.</given-names>
</name>
<name>
<surname>Weissig</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2000</year>). <article-title>The protein Data Bank</article-title>. <source>Nucleic Acids Res.</source> <volume>28</volume>, <fpage>235</fpage>&#x2013;<lpage>242</lpage>. <pub-id pub-id-type="doi">10.1093/nar/28.1.235</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bondy-Denomy</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Pawluk</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Maxwell</surname>
<given-names>K. L.</given-names>
</name>
<name>
<surname>Davidson</surname>
<given-names>A. R.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Bacteriophage genes that inactivate the CRISPR/Cas bacterial immune system</article-title>. <source>Nature</source> <volume>493</volume>, <fpage>429</fpage>&#x2013;<lpage>432</lpage>. <pub-id pub-id-type="doi">10.1038/nature11723</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<collab>ChatGPT</collab> (<year>2023</year>). <article-title>ChatGPT: a comprehensive review on background, applications, key challenges, bias, ethics, limitations and future scope</article-title>. <source>Internet Things Cyber-Physical Syst.</source> <volume>3</volume>, <fpage>121</fpage>&#x2013;<lpage>154</lpage>. <pub-id pub-id-type="doi">10.1016/j.iotcps.2023.04.003</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>PGAweb: a web server for bacterial pan-genome analysis</article-title>. <source>Front. Microbiol.</source> <volume>9</volume>, <fpage>1910</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2018.01910</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Clokie</surname>
<given-names>M. R. J.</given-names>
</name>
<name>
<surname>Millard</surname>
<given-names>A. D.</given-names>
</name>
<name>
<surname>Letarov</surname>
<given-names>A. V.</given-names>
</name>
<name>
<surname>Heaphy</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Phages in nature</article-title>. <source>Bacteriophage</source> <volume>1</volume>, <fpage>31</fpage>&#x2013;<lpage>45</lpage>. <pub-id pub-id-type="doi">10.4161/bact.1.1.14942</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>DeSalle</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Tessler</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Rosenfeld</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <source>Phylogenomics: a primer</source>. <publisher-loc>Boca Raton, FL, USA</publisher-loc>: <publisher-name>CRC Press</publisher-name>.</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Deviyani</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Assessing dataset bias in computer vision</article-title>. <pub-id pub-id-type="doi">10.13140/RG.2.2.19950.89924</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dill</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Ozkan</surname>
<given-names>S. B.</given-names>
</name>
<name>
<surname>Shell</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Weikl</surname>
<given-names>T. R.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>The protein folding problem</article-title>. <source>Annu. Rev. Biophys.</source> <volume>37</volume>, <fpage>289</fpage>&#x2013;<lpage>316</lpage>. <pub-id pub-id-type="doi">10.1146/annurev.biophys.37.092707.153558</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ding</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Baumdicker</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Neher</surname>
<given-names>R. A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>panX: pan-genome analysis and exploration</article-title>. <source>Nucleic Acids Res.</source> <volume>46</volume>, <fpage>e5</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkx977</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Emms</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Kelly</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>OrthoFinder: phylogenetic orthology inference for comparative genomics</article-title>. <source>Genome Biol.</source> <volume>20</volume>, <fpage>238</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-019-1832-y</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fleischmann</surname>
<given-names>R. D.</given-names>
</name>
<name>
<surname>Adams</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>White</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Clayton</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Kirkness</surname>
<given-names>E. F.</given-names>
</name>
<name>
<surname>Kerlavage</surname>
<given-names>A. R.</given-names>
</name>
<etal/>
</person-group> (<year>1995</year>). <article-title>Whole-genome random sequencing and assembly of Haemophilus influenzae rd</article-title>. <source>Science</source> <volume>269</volume>, <fpage>496</fpage>&#x2013;<lpage>512</lpage>. <pub-id pub-id-type="doi">10.1126/science.7542800</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fouts</surname>
<given-names>D. E.</given-names>
</name>
<name>
<surname>Brinkac</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Beck</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Inman</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sutton</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>PanOCT: automated clustering of orthologs using conserved gene neighborhood for pan-genomic analysis of bacterial strains and closely related species</article-title>. <source>Nucleic Acids Res.</source> <volume>40</volume>, <fpage>e172</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gks757</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fremin</surname>
<given-names>B. J.</given-names>
</name>
<name>
<surname>Bhatt</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>Kyrpides</surname>
<given-names>N. C.</given-names>
</name>
</person-group>
<collab>Global Phage Small Open Reading Frame (GP-SmORF) Consortium</collab> (<year>2022</year>). <article-title>Thousands of small, novel genes predicted in global phage genomes</article-title>. <source>Cell. Rep.</source> <volume>39</volume>, <fpage>110984</fpage>. <pub-id pub-id-type="doi">10.1016/j.celrep.2022.110984</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Galperin</surname>
<given-names>M. Y.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Conserved &#x201c;hypothetical&#x201d; proteins: new hints and new puzzles</article-title>. <source>Comp. Funct. Genomics</source> <volume>2</volume>, <fpage>14</fpage>&#x2013;<lpage>18</lpage>. <pub-id pub-id-type="doi">10.1002/cfg.66</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Galperin</surname>
<given-names>M. Y.</given-names>
</name>
<name>
<surname>Wolf</surname>
<given-names>Y. I.</given-names>
</name>
<name>
<surname>Makarova</surname>
<given-names>K. S.</given-names>
</name>
<name>
<surname>Vera Alvarez</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Landsman</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Koonin</surname>
<given-names>E. V.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>COG database update: focus on microbial diversity, model organisms, and widespread pathogens</article-title>. <source>Nucleic Acids Res.</source> <volume>49</volume>, <fpage>D274</fpage>&#x2013;<lpage>D281</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkaa1018</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gautreau</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Bazin</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gachet</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Planel</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Burlot</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Dubois</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>PPanGGOLiN: depicting microbial diversity via a partitioned pangenome graph</article-title>. <source>PLoS Comput. Biol.</source> <volume>16</volume>, <fpage>e1007732</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1007732</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gordillo Altamirano</surname>
<given-names>F. L.</given-names>
</name>
<name>
<surname>Barr</surname>
<given-names>J. J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Phage therapy in the postantibiotic era</article-title>. <source>Clin. Microbiol. Rev.</source> <volume>32</volume>, <fpage>e00066</fpage>. <pub-id pub-id-type="doi">10.1128/CMR.00066-18</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ho</surname>
<given-names>T. K.</given-names>
</name>
</person-group> (<year>2024a</year>). <source>Random decision forests</source>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://ieeexplore.ieee.org/abstract/document/598994">https://ieeexplore.ieee.org/abstract/document/598994</ext-link>.</comment>
</citation>
</ref>
<ref id="B23">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ho</surname>
<given-names>T. K.</given-names>
</name>
</person-group> (<year>2024b</year>). <source>The random subspace method for constructing decision forests</source>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://ieeexplore.ieee.org/abstract/document/709601">https://ieeexplore.ieee.org/abstract/document/709601</ext-link>.</comment>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Horiuchi</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Sakamoto</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Murotsu</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>1974</year>). <article-title>Studies on lambda virulent mutants. III. Action of the anti- and vir-repressor (cro-product) of lambda phage on the related lambdoid phages</article-title>. <source>Mol. Gen. Genet.</source> <volume>133</volume>, <fpage>57</fpage>&#x2013;<lpage>61</lpage>. <pub-id pub-id-type="doi">10.1007/BF00268677</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jones</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Castro</surname>
<given-names>D. C.</given-names>
</name>
<name>
<surname>De Sousa Ribeiro</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Oktay</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>McCradden</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Glocker</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>A causal perspective on dataset bias in machine learning for medical imaging</article-title>. <source>Nat. Mach. Intell.</source>, <fpage>1</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1038/s42256-024-00797-8</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jumper</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Evans</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Pritzel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Green</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Figurnov</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ronneberger</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Highly accurate protein structure prediction with AlphaFold</article-title>. <source>Nature</source> <volume>596</volume>, <fpage>583</fpage>&#x2013;<lpage>589</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-021-03819-2</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kalie</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Jaitin</surname>
<given-names>D. A.</given-names>
</name>
<name>
<surname>Abramovich</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Schreiber</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>An interferon alpha2 mutant optimized by phage display for IFNAR1 binding confers specifically enhanced antitumor activities</article-title>. <source>J. Biol. Chem.</source> <volume>282</volume>, <fpage>11602</fpage>&#x2013;<lpage>11611</lpage>. <pub-id pub-id-type="doi">10.1074/jbc.M610115200</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Knowles</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Silveira</surname>
<given-names>C. B.</given-names>
</name>
<name>
<surname>Bailey</surname>
<given-names>B. A.</given-names>
</name>
<name>
<surname>Barott</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Cantu</surname>
<given-names>V. A.</given-names>
</name>
<name>
<surname>Cobi&#xe1;n-G&#xfc;emes</surname>
<given-names>A. G.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Lytic to temperate switching of viral communities</article-title>. <source>Nature</source> <volume>531</volume>, <fpage>466</fpage>&#x2013;<lpage>470</lpage>. <pub-id pub-id-type="doi">10.1038/nature17193</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Konstantinidis</surname>
<given-names>K. T.</given-names>
</name>
<name>
<surname>Tiedje</surname>
<given-names>J. M.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Genomic insights that advance the species definition for prokaryotes</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A.</source> <volume>102</volume>, <fpage>2567</fpage>&#x2013;<lpage>2572</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.0409727102</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Stoeckert</surname>
<given-names>C. J. Jr</given-names>
</name>
<name>
<surname>Roos</surname>
<given-names>D. S.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>OrthoMCL: identification of ortholog groups for eukaryotic genomes</article-title>. <source>Genome Res.</source> <volume>13</volume>, <fpage>2178</fpage>&#x2013;<lpage>2189</lpage>. <pub-id pub-id-type="doi">10.1101/gr.1224503</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>IPGA: a handy integrated prokaryotes genome and pan-genome analysis web service</article-title>. <source>iMeta</source> <volume>1</volume>, <fpage>e55</fpage>. <pub-id pub-id-type="doi">10.1002/imt2.55</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mathew</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Smatti</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Al Ansari</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Nasrallah</surname>
<given-names>G. K.</given-names>
</name>
<name>
<surname>Al Thani</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Yassine</surname>
<given-names>H. M.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Mixed viral-bacterial infections and their effects on gut microbiota and clinical illnesses in children</article-title>. <source>Sci. Rep.</source> <volume>9</volume>, <fpage>865</fpage>&#x2013;<lpage>912</lpage>. <pub-id pub-id-type="doi">10.1038/s41598-018-37162-w</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McNair</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Bailey</surname>
<given-names>B. A.</given-names>
</name>
<name>
<surname>Edwards</surname>
<given-names>R. A.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>PHACTS, a computational approach to classifying the lifestyle of phages</article-title>. <source>Bioinformatics</source> <volume>28</volume>, <fpage>614</fpage>&#x2013;<lpage>618</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bts014</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Merrikh</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Grossman</surname>
<given-names>A. D.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J. D.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Replication-transcription conflicts in bacteria</article-title>. <source>Nat. Rev. Microbiol.</source> <volume>10</volume>, <fpage>449</fpage>&#x2013;<lpage>458</lpage>. <pub-id pub-id-type="doi">10.1038/nrmicro2800</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meyn</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Rossman</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Troll</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>1977</year>). <article-title>A protease inhibitor blocks SOS functions in <italic>Escherichia coli</italic>: antipain prevents lambda repressor inactivation, ultraviolet mutagenesis, and filamentous growth</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A.</source> <volume>74</volume>, <fpage>1152</fpage>&#x2013;<lpage>1156</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.74.3.1152</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Michniewski</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rihtman</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Cook</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Wilson</surname>
<given-names>W. H.</given-names>
</name>
<name>
<surname>Scanlan</surname>
<given-names>D. J.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>A new family of &#x201c;megaphages&#x201d; abundant in the marine environment</article-title>. <source>ISME Commun.</source> <volume>1</volume>, <fpage>58</fpage>&#x2013;<lpage>64</lpage>. <pub-id pub-id-type="doi">10.1038/s43705-021-00064-6</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Page</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Cummins</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Hunt</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>V. K.</given-names>
</name>
<name>
<surname>Reuter</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Holden</surname>
<given-names>M. T. G.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Roary: rapid large-scale prokaryote pan genome analysis</article-title>. <source>Bioinformatics</source> <volume>31</volume>, <fpage>3691</fpage>&#x2013;<lpage>3693</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btv421</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paget</surname>
<given-names>M. S.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Bacterial sigma factors and anti-sigma factors: structure, function and distribution</article-title>. <source>Biomolecules</source> <volume>5</volume>, <fpage>1245</fpage>&#x2013;<lpage>1265</lpage>. <pub-id pub-id-type="doi">10.3390/biom5031245</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Park</surname>
<given-names>H.-M.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Berani</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Bang</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Vankerschaver</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Van Messem</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2022a</year>). <article-title>
<italic>In silico</italic> optimization of RNA-protein interactions for CRISPR-Cas13-based antimicrobials</article-title>. <source>Biol. Direct</source> <volume>17</volume>, <fpage>27</fpage>. <pub-id pub-id-type="doi">10.1186/s13062-022-00339-5</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Park</surname>
<given-names>H.-M.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Vankerschaver</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Van Messem</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>De Neve</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Shim</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2022b</year>). <article-title>Rethinking protein drug design with highly accurate structure prediction of anti-CRISPR proteins</article-title>. <source>Pharmaceuticals</source> <volume>15</volume>, <fpage>310</fpage>. <pub-id pub-id-type="doi">10.3390/ph15030310</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Park</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Shim</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Sequencing, fast and slow: profiling microbiomes in human samples with nanopore sequencing</article-title>. <source>Appl. Biosci.</source> <volume>2</volume>, <fpage>437</fpage>&#x2013;<lpage>458</lpage>. <pub-id pub-id-type="doi">10.3390/applbiosci2030028</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pilotto</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Fouqueau</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Lukoyanova</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Sheppard</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Lucas-Staat</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>D&#xed;az-Sant&#xed;n</surname>
<given-names>L. M.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Structural basis of RNA polymerase inhibition by viral and host factors</article-title>. <source>Nat. Commun.</source> <volume>12</volume>, <fpage>5523</fpage>&#x2013;<lpage>5615</lpage>. <pub-id pub-id-type="doi">10.1038/s41467-021-25666-5</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="book">
<source>Prioritization of pathogens to guide discovery, research and development of new antibiotics for drug-resistant bacterial infections, including tuberculosis</source>. <publisher-loc>Geneva, Switzerland</publisher-loc>: <publisher-name>World Health Organization</publisher-name>; <year>2019</year>.</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Santajit</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Indrawattana</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Mechanisms of antimicrobial resistance in ESKAPE pathogens</article-title>. <source>Biomed. Res. Int.</source> <volume>2016</volume>, <fpage>2475067</fpage>. <pub-id pub-id-type="doi">10.1155/2016/2475067</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Santos</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Barbosa</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Fiaux</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zurita-Turk</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chaitankar</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Kamapantula</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>PANNOTATOR: an automated tool for annotation of pan-genomes</article-title>. <source>Genet. Mol. Res.</source> <volume>12</volume>, <fpage>2982</fpage>&#x2013;<lpage>2989</lpage>. <pub-id pub-id-type="doi">10.4238/2013.August.16.2</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Shim</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2019a</year>). &#x201c;<article-title>Futuristic methods in virus genome evolution using the third-generation DNA sequencing and artificial neural networks</article-title>,&#x201d; in <source>Global virology III: virology in the 21st century</source>, <fpage>485</fpage>&#x2013;<lpage>513</lpage>.</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shim</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2019b</year>). <article-title>Feature learning of virus genome evolution with the nucleotide skip-gram neural network</article-title>. <source>Evol. Bioinform Online</source> <volume>15</volume>, <fpage>1176934318821072</fpage>. <pub-id pub-id-type="doi">10.1177/1176934318821072</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shim</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Investigating the genomic background of CRISPR-Cas genomes for CRISPR-based antimicrobials</article-title>. <source>arXiv [q-bio.GN]</source> <volume>18</volume>, <fpage>11769343221103887</fpage>. <pub-id pub-id-type="doi">10.1177/11769343221103887</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shim</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Three innovations of next-generation antibiotics: evolvability, specificity, and non-immunogenicity</article-title>. <source>Antibiot. (Basel)</source> <volume>12</volume>, <fpage>204</fpage>. <pub-id pub-id-type="doi">10.3390/antibiotics12020204</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shim</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Shivram</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Lei</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Doudna</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Banfield</surname>
<given-names>J. F.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Diverse ATPase proteins in mobilomes constitute a large potential sink for prokaryotic host ATP</article-title>. <source>Front. Microbiol.</source> <volume>12</volume>, <fpage>691847</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2021.691847</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Silpe</surname>
<given-names>J. E.</given-names>
</name>
<name>
<surname>Duddy</surname>
<given-names>O. P.</given-names>
</name>
<name>
<surname>Johnson</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Beggs</surname>
<given-names>G. A.</given-names>
</name>
<name>
<surname>Hussain</surname>
<given-names>F. A.</given-names>
</name>
<name>
<surname>Forsberg</surname>
<given-names>K. J.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Small protein modules dictate prophage fates during polylysogeny</article-title>. <source>Nature</source> <volume>620</volume>, <fpage>625</fpage>&#x2013;<lpage>633</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-023-06376-y</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Spoerel</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Herrlich</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Bickle</surname>
<given-names>T. A.</given-names>
</name>
</person-group> (<year>1979</year>). <article-title>A novel bacteriophage defence mechanism: the anti-restriction protein</article-title>. <source>Nature</source> <volume>278</volume>, <fpage>30</fpage>&#x2013;<lpage>34</lpage>. <pub-id pub-id-type="doi">10.1038/278030a0</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Steinegger</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>S&#xf6;ding</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>MMseqs2 enables sensitive protein sequence searching for the analysis of massive data sets</article-title>. <source>Nat. Biotechnol.</source> <volume>35</volume>, <fpage>1026</fpage>&#x2013;<lpage>1028</lpage>. <pub-id pub-id-type="doi">10.1038/nbt.3988</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Suttle</surname>
<given-names>C. A.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Viruses in the sea</article-title>. <source>Nature</source> <volume>437</volume>, <fpage>356</fpage>&#x2013;<lpage>361</lpage>. <pub-id pub-id-type="doi">10.1038/nature04160</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tonkin-Hill</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>MacAlasdair</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Ruis</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Weimann</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Horesh</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Lees</surname>
<given-names>J. A.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Producing polished prokaryotic pangenomes with the Panaroo pipeline</article-title>. <source>Genome Biol.</source> <volume>21</volume>, <fpage>180</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-020-02090-4</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Young</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gill</surname>
<given-names>J. J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Phage therapy redux&#x2014;what is to be done?</article-title> <source>Science</source> <volume>350</volume>, <fpage>1163</fpage>&#x2013;<lpage>1164</lpage>. <pub-id pub-id-type="doi">10.1126/science.aad6791</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>L.-L.</given-names>
</name>
<name>
<surname>Qian</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Diao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>A pan-cancer analysis of anti-proliferative protein family genes for therapeutic targets in cancer</article-title>. <source>Sci. Rep.</source> <volume>13</volume>, <fpage>21607</fpage>&#x2013;<lpage>21615</lpage>. <pub-id pub-id-type="doi">10.1038/s41598-023-48961-1</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Charlesworth</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Achtman</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Accurate reconstruction of bacterial pan- and core genomes with PEPPAN</article-title>. <source>Genome Res.</source> <volume>30</volume>, <fpage>1667</fpage>&#x2013;<lpage>1679</lpage>. <pub-id pub-id-type="doi">10.1101/gr.260828.120</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>