<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">878012</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2022.878012</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>
<italic>In Silico</italic> Characterization of Uncharacterized Proteins From Multiple Strains of <italic>Clostridium Difficile</italic>
</article-title>
<alt-title alt-title-type="left-running-head">Abbasi et al.</alt-title>
<alt-title alt-title-type="right-running-head">In Silico Characterization of Uncharacterized Proteins</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Abbasi</surname>
<given-names>Bilal Ahmed</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/913741/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Dharan</surname>
<given-names>Aishwarya</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1683891/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Mishra</surname>
<given-names>Astha</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Saraf</surname>
<given-names>Devansh</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1861688/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ahamad</surname>
<given-names>Irsad</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Suravajhala</surname>
<given-names>Prashanth</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/55577/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Valadi</surname>
<given-names>Jayaraman</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/126476/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Bioclues.org</institution>, <addr-line>Hyderabad</addr-line>, <country>India</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Amrita School of Biotechnology, Amrita Vishwa Vidyapeetham</institution>, <addr-line>Clappana</addr-line>, <country>India</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>School of Computational and Data Sciences, Vidyashilp University</institution>, <addr-line>Bengaluru</addr-line>, <country>India</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Department of Computer Science, FLAME University</institution>, <addr-line>Pune</addr-line>, <country>India</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/392035/overview">Nunzio D&#x2019;Agostino</ext-link>, University of Naples Federico II, Italy</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1218286/overview">Abu Saim Mohammad Saikat</ext-link>, Bangabandhu Sheikh Mujibur Rahman Science and Technology University, Bangladesh</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/149883/overview">Shymaa Enany</ext-link>, Suez Canal University, Egypt</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Prashanth Suravajhala, <email>prash@bioclues.org</email>; Jayaraman Valadi, <email>valadi@gmail.com</email>
</corresp>
<fn fn-type="equal" id="fn1">
<label>
<sup>&#x2020;</sup>
</label>
<p>These authors have contributed equally to this work</p>
</fn>
<fn fn-type="other">
<p>This article was submitted to Computational Genomics, a section of the journal Frontiers in Genetics</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>11</day>
<month>08</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>878012</elocation-id>
<history>
<date date-type="received">
<day>17</day>
<month>02</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>22</day>
<month>06</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Abbasi, Dharan, Mishra, Saraf, Ahamad, Suravajhala and Valadi.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Abbasi, Dharan, Mishra, Saraf, Ahamad, Suravajhala and Valadi</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>
<italic>Clostridium difficile</italic> (<italic>C. difficile</italic>) is a multi-strain, spore-forming, Gram-positive, opportunistic enteropathogen bacteria, majorly associated with nosocomial infections, resulting in severe diarrhoea and colon inflammation. Several antibiotics including penicillin, tetracycline, and clindamycin have been employed to control <italic>C. difficile</italic> infection, but studies have suggested that injudicious use of antibiotics has led to the development of resistance in <italic>C. difficile</italic> strains. However, many proteins from its genome are still considered uncharacterized proteins that might serve crucial functions and assist in the biological understanding of the organism. In this study, we aimed to annotate and characterise the 6&#xa0;<italic>C. difficile</italic> strains using <italic>in silico</italic> approaches. We first analysed the complete genome of 6&#xa0;<italic>C. difficile</italic> strains using standardised approaches and analysed hypothetical proteins (HPs) employing various bioinformatics approaches coalescing, including identifying contigs, coding sequences, phage sequences, CRISPR-Cas9 systems, antimicrobial resistance determination, membrane helices, instability index, secretory nature, conserved domain, and vaccine target properties like comparative homology analysis, allergenicity, antigenicity determination along with structure prediction and binding-site analysis. This study provides crucial supporting information about the functional characterization of the HPs involved in the pathophysiology of the disease. Moreover, this information also aims to assist in mechanisms associated with bacterial pathogenesis and further design candidate inhibitors and <italic>bona fide</italic> pharmaceutical targets.</p>
</abstract>
<kwd-group>
<kwd>clostridium difficile</kwd>
<kwd>uncharacterized proteins</kwd>
<kwd>essential genes</kwd>
<kwd>annotation</kwd>
<kwd>function abbreviations C. difficile-clostridium difficile CDI-C. <italic>difficile</italic> infection</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>
<italic>Clostridium difficile</italic> is a multi-strain, spore-forming, Gram-positive anaerobic bacterium posing a global threat to post-operative individuals. Infamously known for antibiotic-associated diarrhoea, it is one of the most important causes of healthcare-associated infections worldwide, leading to a quarter of reported cases of infectious diarrhoea and a broad spectrum of gastrointestinal complications, including sepsis and pseudomembranous colitis (<xref ref-type="bibr" rid="B5">Barbut and Petit, 2001</xref>). Studies have suggested that it is a crucial part of healthy human gut flora as it overgrows and imbalances intestinal microflora with unnecessary antibiotic therapies (<xref ref-type="bibr" rid="B2">Abt et al., 2016</xref>). With the progression of antibiotic-based therapeutics accompanied by sub-standard hygiene in hospitals, the incidence of <italic>C. difficile</italic> infection (CDI) has significantly increased since the 20th century (<xref ref-type="bibr" rid="B14">Czepiel et al., 2019</xref>). Being a major causative pathogen, <italic>C. difficile</italic> contributes to almost half a million cases with 29,000 deaths per annum in the United States alone and impacting Latin America, Europe, and the Asian regions (<xref ref-type="bibr" rid="B23">Goudarzi et al., 2014</xref>; <xref ref-type="bibr" rid="B31">Lessa et al., 2015</xref>). Whereas in India, the incidence and prevalence rates of CDI-associated diarrhoea in hospitalised patients ranges from 3 to 29% and 7.1&#x2013;26.6%, respectively (<xref ref-type="bibr" rid="B47">Segar et al., 2017</xref>).</p>
<p>
<italic>C. difficile</italic> possesses a huge, diversified pangenome with high levels of evolutionary plasticity accumulated over time due to gene flux and recombination in response to environmental changes. Literature suggests the evolutionary rate of <italic>C. difficile</italic> to be 3.2 &#xd7; 10<sup>&#x2013;7</sup> mutations per nucleotide per year, resulting around 1.4 mutations per genome per year that drives and reshapes the genetic diversity of the pathogen (<xref ref-type="bibr" rid="B17">Didelot et al., 2012</xref>). Additionally, the ratio of the nucleotide substitution rate to result of mutation (r/m) has been estimated around 0.2 or higher. These rates are comparatively lower to other guts pathogens (<xref ref-type="bibr" rid="B24">He et al., 2010</xref>). <italic>C. difficile</italic> infection involves an opportunistic colonisation of the intestinal tract leading to nosocomial, antibiotic-associated severe diarrhoea with or without colitis, fever with chills, and abdominal pain (<xref ref-type="bibr" rid="B6">Bartlett, 2002</xref>; <xref ref-type="bibr" rid="B29">Korman, 2015</xref>). <italic>C. difficile</italic> infection occurs via transmission of spores that are resistant to acid, heat and antibiotics. Antibiotics like metronidazole and oral vancomycin have been recommended as a cure for the acute infection. Other antibiotics including penicillin, tetracycline, and clindamycin, have been employed to control CDI, but studies have suggested that imprudent overuse has led to the development of antimicrobial resistance in <italic>C. difficile</italic> strains (<xref ref-type="bibr" rid="B37">Nelson et al., 1994</xref>). Current treatments for CDI consist of supportive care, discontinuation of unnecessary antibiotics and specific antimicrobial therapies.</p>
<p>Furthermore, novel methodologies including fidaxomicin therapy, and faecal microbiota transplantation-mediated therapy have shown prominent results. Faecal microbiota transplantation has shown significant efficacy to overcome CDI and reduce its recurrence (<xref ref-type="bibr" rid="B23">Goudarzi et al., 2014</xref>). The appearance of hyper-virulent antibiotic-resistant strains with the production of antimicrobial peptides from activated immune cells and inflamed epithelial cells allows the residual <italic>C. difficile</italic> to re-expand, following the end of treatment (<xref ref-type="bibr" rid="B55">Vindigni and Surawicz, 2015</xref>). Growth and development of the bacteria can be prevented at the genetic level effectively, by reducing the prevalence of <italic>C. difficile</italic> and also limiting the rates of recurrence (<xref ref-type="bibr" rid="B30">Leber et al., 2017</xref>). The pathophysiology of the disease, including the transmission and physicochemical pathways employed by <italic>C. difficile,</italic> has aroused several researchers in the past few years to investigate the proteins involved in their virulence (<xref ref-type="bibr" rid="B44">Rineh et al., 2014</xref>; <xref ref-type="bibr" rid="B50">Smits et al., 2016</xref>).</p>
<p>Advanced high-throughput technologies like genome sequencing, gene editing, and functional annotation might be helpful to understand the biology of <italic>C. difficile</italic> and its genomic composition. In the recent past, domains like genomics, transcriptomics and proteomics studies have assisted in gaining insights to the mechanism of microbial adaptation (<xref ref-type="bibr" rid="B46">Sebaihia et al., 2006</xref>; <xref ref-type="bibr" rid="B51">Stabler et al., 2009</xref>; <xref ref-type="bibr" rid="B8">Boetzkes et al., 2012</xref>; <xref ref-type="bibr" rid="B10">Cafardi et al., 2013</xref>). Moreover, there are still challenges while decoding these mechanisms, and the bioinformatics tool aids in our understanding via functional annotation, protein-protein interactions, and pathway analysis. Functional annotation of uncharacterized proteins is a crucial step in deciphering the role of proteins. An uncharacterized or hypothetical protein (HP) is defined as the one that is predicted to be expressed in an organism but no proper function is known (<xref ref-type="bibr" rid="B52">Suravajhala et al., 2015</xref>). Most of these hypothetical proteins are expected to play essential roles and their annotation can unveil novel functional pathways. The utilisation of <italic>in silico</italic> approaches to predict annotations of HPs has been successful in numerous bacterial species (<xref ref-type="bibr" rid="B48">Singh et al., 2015</xref>; <xref ref-type="bibr" rid="B54">Varma et al., 2015</xref>; <xref ref-type="bibr" rid="B42">Prabhu et al., 2020</xref>). Additionally, there are still several challenges in annotating such proteins, given the scanty organelle information known and the pervasive nature of the subcellular location of these proteins. Earlier, methods to annotate HPs by us (<xref ref-type="bibr" rid="B26">Ijaq et al., 2019</xref>) could be useful, but given the bacterial system, a coherent need for employing several computational tools, viz. determine the conserved domain, subcellular localization, secretory nature, physicochemical characterization, identification of prophage sequence and CRISPR-Cas9 system, detection of antimicrobial resistance, comparative homology analysis, virulence factors, antigenicity analysis, allergenicity determination along with structure prediction and binding-site analysis would allow us to annotate the possible functions for the HPs. Deciphering the role of complete gene coding regions in the genome is crucial to merge the gaps in the proteome to fully understand the pathogenicity. This study aims to determine the functional annotations of HPs of <italic>C. difficile</italic> to have a clear implication.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>Materials and Methodology</title>
<sec id="s2-1">
<title>Data Retrieval</title>
<p>A total of 2,512 genomes of <italic>C. difficile</italic> were available in the NCBI database (25 Aug 2021). A robust methodology was used to narrow down these 2,512 genomes to retrieve the complete genome of six strains of <italic>C. difficile</italic>, namely, BR81, R20291, CF5, M120, 196, and 2,007,855. Initially, this obtained data was standardised using RAST pipeline, which is an automated service that gives high-quality genome annotations for complete or nearly complete bacterial and archaeal genomes (<xref ref-type="bibr" rid="B39">Overbeek et al., 2014</xref>). Finally, HPs were extracted from proteomes of these <italic>C. difficile</italic> strains using a python script. The protocol used in this study is depicted (<xref ref-type="fig" rid="F1">Figure 1</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Methodology used for characterization of HPs for <italic>Clostridium</italic> strains. Briefly, we have jotted down the HPs using python script and characterised the six strains of <italic>C. difficile</italic> using a cohort of tools (Please see methodology). The subsequent annotation is tabulated and analysed.</p>
</caption>
<graphic xlink:href="fgene-13-878012-g001.tif"/>
</fig>
</sec>
<sec id="s2-2">
<title>Identification of Prophage and CRISPR-Cas9 System</title>
<p>CRISPR-Cas9 system acts as an adaptive immune system in microbes against prophages. It uses RNA guided nucleases to cleave foreign genetic elements (<xref ref-type="bibr" rid="B43">Ran et al., 2013</xref>). It is also responsible for a continuous saga of evolution between phages and bacteria via addition or deletion of spacers into the genome of host bacteria and via mutations or deletion in phage genomes (<xref ref-type="bibr" rid="B16">Deveau et al., 2010</xref>). Thus, prophage aids in understanding the evolution of bacterial genomes. The PHASTER server was employed for the identification of prophage sequences in the whole genomes of bacterial strains (<xref ref-type="bibr" rid="B4">Arndt et al., 2016</xref>). It queries virus and prophage/bacterial databases to identify potent prophage sequences and rank the hits according to score; as intact (score&#x3e;90), questionable (score ranging between 70 and 90), or incomplete (score&#x3c;70). Additionally, CRISPRCasFinder was employed to identify CRISPR-Cas9 related genes in six strains of Closteroides using the default settings (<xref ref-type="bibr" rid="B13">Couvin et al., 2018</xref>).</p>
</sec>
<sec id="s2-3">
<title>Identification of Antimicrobial Resistance</title>
<p>In order to identify AMR genes, two different procedures were utilised. Firstly, the identified HPs from the six CD strains, and RAST based characterization of WGS were searched for the presence of AMR genes using the tools AMRFinderPlus v3.10 and ResFinder v 4.1 (<xref ref-type="bibr" rid="B9">Bortolaia et al., 2020</xref>; <xref ref-type="bibr" rid="B20">Feldgarden et al., 2021</xref>). Default settings with a maximum coverage length of 80% and percent identity set at 90% were used for AMRFinder Plus. Default settings with % ID threshold set at 80% were used for ResFinder.</p>
</sec>
<sec id="s2-4">
<title>Physicochemical Characterization of Hypothetical Proteins</title>
<p>ExPASy&#x2019;s ProtParam tool evaluated various physicochemical properties for the obtained HPs. The ProtParam tool determines these properties based on the amino-acid sequence. For the present study, we computed properties like theoretical isoelectric point (pI), molecular weight, instability index, and grand average of hydropathicity (GRAVY) value. The instability index estimates whether a protein will be stable in the test tube or not. Proteins having an instability index lesser than 40 are predicted to be stable, whereas a value greater than 40 indicates the protein to be unstable. A negative GRAVY value implies the protein is non-polar, while a positive value means the protein is polar (<xref ref-type="bibr" rid="B21">Gasteiger et al., 2005</xref>).</p>
</sec>
<sec id="s2-5">
<title>Identification of Subcellular Localization and Secretory Nature</title>
<p>Each bacterial protein is localised into different subcellular locations like cytoplasm, plasma membrane, outer membrane etc<italic>.</italic> and can perform different functions. Thus, subcellular localization is a chief criterion for identification of potential bacterial drug targets (<xref ref-type="bibr" rid="B38">Omeershffudin and Kumar, 2019</xref>). PSORTb 3.0 tool was used for assigning the subcellular localization of HPs (<xref ref-type="bibr" rid="B56">Yu et al., 2010</xref>). The tool utilises a support vector machine that gives scores related to subcellular classifiers of each protein based on their amino acid sequences and evaluates their probability of finding the final location. Additionally, another tool, SignalP 5.0 server was used to predict whether the HPs from 6&#xa0;<italic>C. difficile</italic> strains are secretory or non-secretory proteins in nature. The server predicts the presence of signal peptides and the location of their cleavage sites in proteins. In bacteria, it can discriminate between three signal peptides, Sec/SPI, Sec/SPII, and Tat/SPI, based on how they are transported and cleaved (<xref ref-type="bibr" rid="B41">Petersen et al., 2011</xref>).</p>
</sec>
<sec id="s2-6">
<title>Functional Domain Prediction</title>
<p>NCBI Conserved Domain Search Service (CDD) was implemented to investigate the domains of the selected HPs. It identifies the conserved domains present in protein sequences by performing Reverse Position Specific (RPS)-BLAST against position specific scoring matrix (PSSM) resulting from conserved domain alignments present in the conserved domain database (<xref ref-type="bibr" rid="B34">Marchler-Bauer et al., 2015</xref>).</p>
</sec>
<sec id="s2-7">
<title>Comparative Homology Analysis</title>
<p>The homology analysis of HPs against the human proteome was performed using the BLASTp tool. The proteins with &#x2265;35% identity, &#x2265;35% query coverage, and &#x3c;10e-5E value were considered homologous to human proteins. The hypothetical non-homologous protein can be used to design potential vaccine candidates against <italic>C. difficile</italic> since those will avoid generating potential cross-reactivity (<xref ref-type="bibr" rid="B3">Altschul et al., 1990</xref>).</p>
</sec>
<sec id="s2-8">
<title>Prediction of Virulence Factors</title>
<p>Bacterial virulence factors are the molecules, cell structures, or regulatory pathways that allow the microbes to replicate and spread within the host by evading or suppressing the host&#x2019;s immune response. These can serve as targets for identifying new therapies against the disease. The Virulence Factor Database (VFDB) was used for determining whether the identified HPs are virulent factors or not (<xref ref-type="bibr" rid="B12">Chen et al., 2005</xref>).</p>
</sec>
<sec id="s2-9">
<title>Antigenicity Analysis</title>
<p>Identification of novel antigens associated with infectious diseases are essential for invention of new diagnostic tests as well as designing subunit vaccines against them (<xref ref-type="bibr" rid="B32">Liang and Felgner, 2012</xref>). Thus, it is important to identify if the HPs from six strains of <italic>C. difficile</italic> are antigenic in nature. To predict the antigenicity of the HP, an online server, Vaxijen was used with the default settings for Gram positive bacteria (<xref ref-type="bibr" rid="B18">Doytchinova and Flower, 2007</xref>).</p>
</sec>
<sec id="s2-10">
<title>Structure Prediction and Active Site Determination</title>
<p>With the results of the previous step, the highest antigenic proteins were identified and further subjected to structure modelling via iTASSER structure prediction server (<xref ref-type="bibr" rid="B45">Roy et al., 2010</xref>). These three-dimensional proteins were further employed and investigated for active site determination using CastP server (<xref ref-type="bibr" rid="B53">Tian et al., 2018</xref>).</p>
</sec>
<sec id="s2-11">
<title>Allergenicity Analysis</title>
<p>AllergenOnline database was used to predict whether the HPs are allergic to humans in nature. This information helps in determining whether the HPs are potentially allergenic. Non-allergenic proteins can be utilised for designing vaccine candidates (<xref ref-type="bibr" rid="B22">Goodman et al., 2016</xref>).</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<sec id="s3-1">
<title>Data Retrieval</title>
<p>In this study, six complete genomes from <italic>C. difficile</italic> were utilised. All the genomes were retrieved from the NCBI database (<ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/genome/">https://www.ncbi.nlm.nih.gov/genome/</ext-link>) and standardised using RAST (<xref ref-type="bibr" rid="B39">Overbeek et al., 2014</xref>). The key characteristics of the strains used in this study are listed here (<xref ref-type="table" rid="T1">Table 1</xref> and <xref ref-type="fig" rid="F2">Figure 2</xref>).</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Characteristics of selected genomes of <italic>Clostriodioles</italic> strains.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">S. No</th>
<th align="center">Description</th>
<th align="center">Accession ID</th>
<th align="center">Size (Mb)</th>
<th align="center">Proteins (HP)</th>
<th align="center">GC%</th>
<th align="center">CDS</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">01</td>
<td>
<italic>C. difficile</italic> BR81</td>
<td align="center">CP019870.1</td>
<td align="char" char=".">4,124,384</td>
<td align="char" char="(">3,547 (356)</td>
<td align="char" char=".">28.7</td>
<td align="char" char=".">3,683</td>
</tr>
<tr>
<td align="left">02</td>
<td>
<italic>C. difficile</italic> R20291</td>
<td align="center">NZ_CP029423.1</td>
<td align="char" char=".">4,204,902</td>
<td align="char" char="(">3,647 (409)</td>
<td align="char" char=".">28.9</td>
<td align="char" char=".">3,802</td>
</tr>
<tr>
<td align="left">03</td>
<td>
<italic>C. difficile</italic> CF5</td>
<td align="center">NC_017,173.1</td>
<td align="char" char=".">4,159,517</td>
<td align="char" char="(">3,587 (413)</td>
<td align="char" char=".">28.5</td>
<td align="char" char=".">3,797</td>
</tr>
<tr>
<td align="left">04</td>
<td>
<italic>C. difficile</italic> M120</td>
<td align="center">FN665653.1</td>
<td align="char" char=".">4,047,729</td>
<td align="char" char="(">3,446 (404)</td>
<td align="char" char=".">28.7</td>
<td align="char" char=".">3,697</td>
</tr>
<tr>
<td align="left">05</td>
<td>
<italic>C. difficile</italic> 196</td>
<td align="center">NC_013,315.1</td>
<td align="char" char=".">4,110,554</td>
<td align="char" char="(">3,552 (374)</td>
<td align="char" char=".">28.6</td>
<td align="char" char=".">3,715</td>
</tr>
<tr>
<td align="left">06</td>
<td>
<italic>C. difficile</italic> 2,007,855</td>
<td align="center">NC_017,178.1</td>
<td align="char" char=".">4,179,867</td>
<td align="char" char="(">3,614 (377)</td>
<td align="char" char=".">28.7</td>
<td align="char" char=".">3,806</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Overview of Subsystem Category Distribution for six strains <bold>(A)</bold> <italic>C. difficile</italic> strain CF5. <bold>(B)</bold> <italic>C. difficile</italic> strain BR81 <bold>(C)</bold> <italic>C. difficile</italic> strain R20291 <bold>(D)</bold> <italic>C. difficile</italic> strain 196 <bold>(E)</bold> <italic>C. difficile</italic> strain 2,007,855 <bold>(F)</bold> <italic>C. difficile</italic> strain M120. Identification of Prophage and CRISPR-Cas9 systems.</p>
</caption>
<graphic xlink:href="fgene-13-878012-g002.tif"/>
</fig>
</sec>
<sec>
<title>Identification of Prophage and CRISPR-Cas9 System</title>
<p>The PHASTER tool was employed to identify the phage genes, if any, present in six strains of <italic>C. difficile</italic>. Four Intact Phage sequences (score&#x3e;90) were found in <italic>C. difficile</italic> strain of R20291, CF5, 196, and 2,007,855, the details of which are provided here (<xref ref-type="table" rid="T2">Table 2</xref>). CRISPRCasFinder was employed to identify any CRISPR/Cas system located in the 6&#xa0;<italic>C. difficile</italic> strains. The significance level of the predicted systems was evaluated based on the evidence level. Of all the <italic>C</italic>. <italic>difficile</italic> selected, the majority of the sequences had more than one spacer sequence and an Evidence level of 3-4, which suggests the presence of CRISPR/Cas genes (<xref ref-type="sec" rid="s10">Supplementary Table S1</xref>).</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Intact Prophage region identified in <italic>Clostriodium difficile</italic> strains.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Genome</th>
<th align="center">Intact Region</th>
<th align="center">Region length (Kb)</th>
<th align="center">Score</th>
<th align="center">Total protein</th>
<th align="center">Position</th>
<th align="center">Common phage</th>
<th align="center">GC%</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<italic>C. difficile</italic> R20291</td>
<td align="char" char=".">1</td>
<td align="char" char=".">55.9</td>
<td align="char" char=".">140</td>
<td align="char" char=".">71</td>
<td align="char" char="ndash">1,684,408&#x2013;1,740,383</td>
<td align="center">PHAGE_Clostr_phiMMP01_NC_028,883 (32)</td>
<td align="char" char=".">28.64</td>
</tr>
<tr>
<td align="left">
<italic>C. difficile</italic> CF5</td>
<td align="char" char=".">1</td>
<td align="char" char=".">56.2</td>
<td align="char" char=".">130</td>
<td align="char" char=".">74</td>
<td align="char" char="ndash">1,707,633&#x2013;1,763,916</td>
<td align="center">PHAGE_Clostr_phiMMP03_NC_028,959 (30)</td>
<td align="char" char=".">29.02</td>
</tr>
<tr>
<td align="left">
<italic>C. difficile</italic> 196</td>
<td align="char" char=".">1</td>
<td align="char" char=".">57.7</td>
<td align="char" char=".">140</td>
<td align="char" char=".">72</td>
<td align="char" char="ndash">1,673,218&#x2013;1,730,955</td>
<td align="center">PHAGE_Clostr_phiMMP01_NC_028,883 (31)</td>
<td align="char" char=".">28.48</td>
</tr>
<tr>
<td align="left">
<italic>C. difficile</italic> 2,007,855</td>
<td align="char" char=".">1</td>
<td align="char" char=".">55.9</td>
<td align="char" char=".">140</td>
<td align="char" char=".">71</td>
<td align="char" char="ndash">1,666,638&#x2013;1,722,613</td>
<td align="center">PHAGE_Clostr_phiMMP01_NC_028,883 (32)</td>
<td align="char" char=".">28.64</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-2">
<title>Identification of Antimicrobial Resistance Genes</title>
<p>The AMR analysis from NCBI HPs and RAST yielded different results. From the identified HPs obtained by NCBI, no AMR genes were found by AMRFinderPlus. The analysis of the WGS by AMRFinderPlus revealed the presence of AMR genes <italic>vanZ1</italic> and <italic>blaCDD</italic> which confer resistance to Vancomycin and Beta-Lactam respectively, It also yielded genes with virulence factors <italic>tcdE, tcdB, tcdR,</italic> that were common in all six CD strains. RAST based amino acid sequences revealed that AMR genes <italic>vanZ1</italic>, <italic>blaR1</italic> (Beta-Lactam) and <italic>blaCDD</italic> and genes with virulence factor <italic>tcdE</italic> and <italic>tcdB</italic> were found in all 6&#xa0;<italic>C. difficile</italic> strains. Additionally, RAST based classified HP amino acid sequences had the virulence variant <italic>tcdR</italic> common in all 6&#xa0;<italic>C. difficile</italic> strains. ResFinder yielded no AMR genes and virulent factors however in WGS, ResFinder found AMR genes <italic>ant(6)-Ib</italic>, <italic>ant(6)-Ia,</italic> which confer resistance against aminoglycoside and <italic>tet(M)</italic>, <italic>tet(44)</italic> which confer resistance to Tetracycline in <italic>C. difficile</italic> M120 strain and <italic>erm(B)</italic> which confers resistance to Macrolide and <italic>aac(6&#x2032;)-Im, aph(2&#x2033;)-Ib</italic> confers resistance to Aminoglycosides in <italic>C. difficile</italic> 2,007,855 (<xref ref-type="table" rid="T3">Table 3</xref>).</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>AMR genes determination in WGS strains of <italic>C. difficile</italic> 2,007,855 and M120 using ResFinder tool.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Strain</th>
<th align="center">Resistance Gene</th>
<th align="center">Identity%</th>
<th align="center">Position in Contig</th>
<th align="center">Alignment length</th>
<th align="center">Phenotype</th>
<th align="center">Accession No.</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<italic>C. difficile</italic> 2,007,855</td>
<td>
<italic>erm(B)</italic>
</td>
<td align="char" char=".">100</td>
<td align="char" char=".">3,136,169..3,136,906</td>
<td align="char" char=".">738</td>
<td>Macrolide resistance</td>
<td>U18931</td>
</tr>
<tr>
<td align="left">
<italic>C. difficile</italic> 2,007,855</td>
<td>
<italic>aac(6&#x2032;)-Im</italic>
</td>
<td align="char" char=".">96.65</td>
<td align="char" char=".">3,810,802..3,811,338</td>
<td align="char" char=".">537</td>
<td>Aminoglycoside resistance</td>
<td>AF337947</td>
</tr>
<tr>
<td align="left">
<italic>C. difficile</italic> 2,007,855</td>
<td>
<italic>aph(2&#x2033;)-Ib</italic>
</td>
<td align="char" char=".">98.67</td>
<td align="char" char=".">3,811,382..3,812,281</td>
<td align="char" char=".">900</td>
<td>Aminoglycoside resistance</td>
<td>KF652098</td>
</tr>
<tr>
<td align="left">
<italic>C. difficile</italic> M120</td>
<td>
<italic>ant(6)-Ib</italic>
</td>
<td align="char" char=".">100</td>
<td align="char" char=".">480,747..481,604</td>
<td align="char" char=".">858</td>
<td>Aminoglycoside resistance</td>
<td>FN594949</td>
</tr>
<tr>
<td align="left">
<italic>C. difficile</italic> M120</td>
<td>
<italic>ant(6)-Ia</italic>
</td>
<td align="char" char=".">100</td>
<td align="char" char=".">468,126..468,989</td>
<td align="char" char=".">864</td>
<td>Aminoglycoside resistance</td>
<td>KF421157</td>
</tr>
<tr>
<td align="left">
<italic>C. difficile</italic> M120</td>
<td>
<italic>tet(M)</italic>
</td>
<td align="char" char=".">98.85</td>
<td align="char" char=".">2,175,639..2,177,558</td>
<td align="char" char=".">1920</td>
<td>Tetracycline resistance</td>
<td>EU182585</td>
</tr>
<tr>
<td align="left">
<italic>C. difficile</italic> M120</td>
<td>
<italic>tet(44)</italic>
</td>
<td align="char" char=".">98.02</td>
<td align="char" char=".">478,510..480,432</td>
<td align="char" char=".">1923</td>
<td>Tetracycline resistance</td>
<td>NZ_ABDU01000081</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-3">
<title>Characterization of Physicochemical Properties</title>
<p>The Instability Index (II), isoelectric point, GRAVY value, and molecular weight of HPs from six strains were determined using the ProtParam tool. In all six strains of <italic>C. difficile</italic>, approximately 70% of the HP sequences had II below 40, indicating that the majority of HPs were stable. <italic>C. difficile</italic> strains R20291 and M120 had more than 280 HP sequences having II values below 40. <italic>C. difficile</italic> strains 196, 2,007,855, and CF5 had approximately 260&#x2013;275 HP sequences with II values below 40. <italic>C. difficile</italic> strain BR81 had the lowest number of HP sequences, around 240, with an II value below 40. Moreover, HPs from all six strains had theoretical pI ranging from 4.05 to 11.99, while around 70% were found to have negative GRAVY values, indicating that they are non-polar in nature (<xref ref-type="sec" rid="s10">Supplementary Figure S1</xref>). Additionally, detailed information and physicochemical characterization are shown in (<xref ref-type="sec" rid="s10">Supplementary Sheet S1</xref>).</p>
</sec>
<sec id="s3-4">
<title>Identification of Subcellular Localization and Secretory Nature Determination</title>
<p>The subcellular localization of proteins was identified by the PSORTb tool, which classified the HPs from all six strains of <italic>C. difficile</italic> into four categories, namely, cytoplasmic, cytoplasmic membrane, extracellular and unknown, based on their location in the bacterial cell. In all the identified strains of <italic>C. difficile</italic>, approximately 32&#x2013;38% and 23&#x2013;27% HPs were localised in the cytoplasm and cytoplasmic membrane, respectively. Meanwhile, 1&#x2013;3% and 37&#x2013;40% of all HPs in these six strains were located in the extracellular space, or their location is unknown (<xref ref-type="table" rid="T4">Table 4</xref>). SignalP 5.0 server was employed to predict the secretory nature of HPs from 6&#xa0;<italic>C. difficile</italic> strains. Approximately, 87&#x2013;92% of HPs from each strain were non-secretory, while the remaining proteins were secretory.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Subcellular localisation of <italic>C. difficile</italic> strains hypothetical proteins determined by PSORTb.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left">Strain</th>
<th rowspan="2" align="center">Total HPs</th>
<th colspan="4" align="center">Subcellular Location as Given Be PSORTb</th>
</tr>
<tr>
<th align="center">Cytoplasmic</th>
<th align="center">Cytoplasmic membrane</th>
<th align="center">Extracellular</th>
<th align="center">Unknown</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<italic>C. difficile</italic> BR81</td>
<td align="char" char=".">356</td>
<td align="center">122 (34.27%)</td>
<td align="center">86 (24.16%)</td>
<td align="center">10 (2.81%)</td>
<td align="center">138 (38.76%)</td>
</tr>
<tr>
<td align="left">
<italic>C. difficile</italic> R20291</td>
<td align="char" char=".">409</td>
<td align="center">142 (34.72%)</td>
<td align="center">99 (24.21%)</td>
<td align="center">9 (2.20%)</td>
<td align="center">159 (38.88%)</td>
</tr>
<tr>
<td align="left">
<italic>C. difficile</italic> CF5</td>
<td align="char" char=".">413</td>
<td align="center">155 (37.53%)</td>
<td align="center">96 (23.24%)</td>
<td align="center">8 (1.94%)</td>
<td align="center">154 (37.29%)</td>
</tr>
<tr>
<td align="left">
<italic>C. difficile</italic> M120</td>
<td align="char" char=".">404</td>
<td align="center">130 (32.18%)</td>
<td align="center">110 (27.23%)</td>
<td align="center">4 (0.99%)</td>
<td align="center">160 (39.60%)</td>
</tr>
<tr>
<td align="left">
<italic>C. difficile</italic> 196</td>
<td align="char" char=".">374</td>
<td align="center">130 (34.76%)</td>
<td align="center">88 (25.53%)</td>
<td align="center">8 (2.14%)</td>
<td align="center">148 (39.57%)</td>
</tr>
<tr>
<td align="left">
<italic>C. difficile</italic> 2,007,855</td>
<td align="char" char=".">377</td>
<td align="center">133 (35.28%)</td>
<td align="center">90 (23.87%)</td>
<td align="center">9 (2.39%)</td>
<td align="center">145 (38.46%)</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-5">
<title>Functional Domain Prediction</title>
<p>Domains are distinct, recurring, functional and structural units of protein, the extent of which can be determined by sequence and structure analysis and are crucial in molecular evolution. Conserved domains contain highly conserved sequence patterns or motifs, which might be detected in a polypeptide sequences. The data obtained from NCBI Batch CDD search tool showed that <italic>C. difficile</italic> BR81, <italic>C. difficile</italic> M120, and <italic>C. difficile</italic> CF5 HPs had nine specific hit types/conserved domains. The functional signature identified in <italic>C. difficile</italic> BR81 strain belongs to nine specific superfamilies namely, Beta_helix, Chalcone_N, GH113_mannanase-like, HDC_protein (x2), M34_PPEP, PBECR3, Pectate_lyase_3, SPASM. Similarly, HPs of <italic>C. difficile</italic> M120 strain had nine specific superfamilies including Beta_helix_3, Chalcone_N, GH113_mannanase-like, HDC_protein (x3), M34_PPEP, PBECR3, SPASM. HPs of <italic>C. difficile</italic> 196 strain had seven specific superfamilies namely Chalcone_N, Glyco_hydro_129, HDC_protein (x2), M34_PPEP, PBECR3, SPASM. Additionally, HPs of <italic>C. difficile</italic> CF5 strain had nine specific superfamilies like ABC_trans_CmpB, C80_toxinA_B-like, Gly_rich, HDC_protein (x3), M34_PPEP, PBECR3, SPASM. Moreover, HPs of <italic>C. difficile</italic> 2,007,855 strain had eight specific superfamilies Chalcone_N, GH113_mannanase-like, Glyco_hydro_129, HDC_protein (x2), M34_PPEP, PBECR3, SPASM. Lastly, HPs of <italic>C. difficile</italic> R20291 strain had eleven specific superfamilies Chalcone_N, DUF5685, DUF5699, DUF5780, GH113_mannanase-like, Glyco_hydro_129, HDC_protein (x2), M34_PPEP, PBECR3, SPASM.</p>
<p>Furthermore, this analysis could be effective in predicting the functional role of HPs determined on the basis of their conserved domains and motifs. Likewise, the common conserved domains identified in HPs of shortlisted strains were HDC_protein, M34_PPEP, PBECR3 and SPASM. The most recurring superfamily Histidine decarboxylase (HDC_protein) is the sole member of the histamine synthesis pathway, producing histamine in a one-step reaction. Histamine cannot be generated by any other known enzyme (<xref ref-type="bibr" rid="B35">Mohammad et al., 2009</xref>). M34_PPEP includes the enzyme Pro-Pro endopeptidase (PPEP-1), an extracellular metalloprotease belonging to peptidase family M34. It aids <italic>C. difficile</italic> in switching from an adhesive to a motile phenotype by cleaving cell surface proteins (<xref ref-type="bibr" rid="B33">Lu et al., 2020</xref>). PBECR3 (phage-Barnase-EndoU-ColicinE5/D-RelE like nuclease3) is an endoRNase found in polyvalent proteins of phages and conjugative elements (<xref ref-type="bibr" rid="B28">Iyer et al., 2017</xref>). SPASM occurs as an additional C-terminal domain in many peptide-modifying enzymes of the radical S-adenosylmethionine (SAM) superfamily (<xref ref-type="bibr" rid="B33">Lu et al., 2020</xref>).</p>
</sec>
<sec id="s3-6">
<title>Comparative Homology Analysis</title>
<p>Proteins dissimilar to human proteome are prioritised in therapeutic and vaccine designing, since homologous proteins can cause side effects and cross-reactivity. Those proteins with &#x2265;35% identity, query coverage &#x2265;35%, and E value &#x3c; 10e-5 were considered. Approximately, 99.7% of the identified HPs across all the selected strains were non-homologous. This indicates that they can be further evaluated for vaccine and other pharmaceutical properties.</p>
</sec>
<sec id="s3-7">
<title>Prediction of Virulence</title>
<p>Virulent proteins assist bacteria in colonising the host and pathogenesis, and these proteins could be cytoplasmic, membranous, or secretory. They help in adhesion, adaptation to the changing environment, and protection against host immune response. Therefore, prioritising these proteins is necessary since they are potential drug targets and immunogenic vaccine candidates. An approximate, 0.25&#x2013;1.45% of HPs from each strain were found to be virulent in nature, while approximately 99% of proteins from each strain showed no virulence factor.</p>
</sec>
<sec id="s3-8">
<title>Antigenicity Analysis</title>
<p>Antigenicity analysis was determined using the VaxiJen server for 6&#xa0;<italic>C. difficile</italic> strains. We found that around 39.83% of the <italic>C. difficile</italic> 196 HPs to be antigenic, 40.84% of the <italic>C. difficile</italic> 2,007,855 to be antigenic, 41.01% of <italic>C. difficile</italic> BR81 to be antigenic proteins, 36.07% of <italic>C. difficile</italic> CF5 to be antigenic proteins, 40.09% of the <italic>C. difficile</italic> M120 to be antigenic proteins whereas, 39.85% of the <italic>C. difficile</italic> R20291 to be antigenic proteins. These data suggest that HPs could be further investigated for vaccine properties. Further, we prioritized top antigenic protein from each strain to examine their structure and binding analysis. WP_104,732,835.1 (CD20291strain), WP_009,906,007.1 (CDM120strain), WP_003,429,932.1 (CDCF5strain), WP_021,396,478.1 (CD196strain), WP_021,389,778.1 (CDBR81strain) and WP_003,423,063.1 (CD2007855strain) turned out to be promising candidate and further processed for structure prediction analysis.</p>
</sec>
<sec id="s3-9">
<title>Structure Prediction and Active Site Determination</title>
<p>Antigenicity determination allows identification of highly antigenic proteins, which can also assist in filtering out the best potential vaccine candidate (<xref ref-type="bibr" rid="B1">Abbasi et al., 2022</xref>). A threshold was selected and six highly antigenic proteins were modelled using the iTASSER server that may be explored for new drug designing strategies. Further, these 3D structural models were subjected to identify active sites by employing CastP server. After pre-processing, the top-ranked potential receptor binding sites and respective residue were identified. All the sites and cavities are shown (<xref ref-type="fig" rid="F3">Figure 3</xref>) and a list of respective residues are provided in <xref ref-type="sec" rid="s10">Supplementary Table S2</xref>.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Identification of active site pockets (highlighted in red cavity) for the shortlisted antigenic proteins <bold>(A)</bold> WP_104,732,835.1 <bold>(B)</bold> WP_009,906,007.1 <bold>(C)</bold> WP_003,429,932.1 <bold>(D)</bold> WP_021,396,478.1 <bold>(E)</bold> WP_021,389,778.1, and <bold>(F)</bold> WP_003,423,063.1.</p>
</caption>
<graphic xlink:href="fgene-13-878012-g003.tif"/>
</fig>
</sec>
<sec id="s3-10">
<title>Allergenicity Analysis</title>
<p>Therapeutic molecules like vaccines and drugs also have the potential to cause allergic reactions. Therefore, it is essential to check if the protein candidate used acts as an allergen or not. We found that all six <italic>Clostridioides</italic> strains HPs are non-allergens in nature.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<p>Understanding the genomic epidemiology and annotating proteins remains the most impactful strategies to detect, characterise, and monitor pathogens that impact human health. Though traditional biochemical and molecular experiments can be used to assign proper functions for genes, they are expensive, tedious, and have resulted in only 50&#x2013;60% gene annotations (<xref ref-type="bibr" rid="B49">Sivashankari and Shanmughavel, 2006</xref>). Despite continuous research efforts, a large portion of the proteome is still represented by uncharacterized proteins designated as HPs. They are predicted from the nucleic acid sequences and have unknown functions (<xref ref-type="bibr" rid="B54">Varma et al., 2015</xref>). Thus, automated gene/protein annotation using bioinformatics tools can overcome this challenge of the post-genome era where complete genome sequences of several organisms are available. Several researchers have gradually worked on structural or functional annotation of HPs from different microbes like <italic>Staphylococcus aureus</italic>, <italic>Vibrio cholera</italic>, <italic>Blumeria graminis</italic> and <italic>Serratia marcescens</italic> and others (<xref ref-type="bibr" rid="B27">Islam et al., 2015</xref>; <xref ref-type="bibr" rid="B54">Varma et al., 2015</xref>; <xref ref-type="bibr" rid="B15">da Costa et al., 2018</xref>; <xref ref-type="bibr" rid="B42">Prabhu et al., 2020</xref>). But still, they are shrouded in the mystery and there is a dire need to expedite the process (<xref ref-type="bibr" rid="B38">Omeershffudin and Kumar, 2019</xref>). Some of them include deriving information from sequence similarity analysis, interaction of proteins with other proteins or ligands, gene expression profiles, conserved domains/motifs, phylogenetic analysis, phosphorylation regions and active site residue similarity analysis. The most orthodox way of speculating protein function involves sequence similarity analysis using BLAST tool (<xref ref-type="bibr" rid="B54">Varma et al., 2015</xref>).</p>
<p>Quite a few studies have elucidated the genomic epidemiology of <italic>C. difficile</italic>, but none of them have focused on the HPs (<xref ref-type="bibr" rid="B10">Cafardi et al., 2013</xref>; <xref ref-type="bibr" rid="B19">Ezhilarasan et al., 2013</xref>; <xref ref-type="bibr" rid="B7">Basak et al., 2021</xref>). Recently, Basak et al., have characterised an <italic>in silico</italic> vaccine using the immunoinformatics approaches via utilising CotE, SlpA and FliC proteins, which were responsible for gastrointestinal tract colonisation, TLR4 interaction, cytokine production, and plays a major role in the adherence of the bacterial cell, which ultimately triggers the innate immune response (<xref ref-type="bibr" rid="B40">P&#xe9;chin&#xe9; et al., 2005</xref>; <xref ref-type="bibr" rid="B25">Hong et al., 2017</xref>; <xref ref-type="bibr" rid="B36">Mori and Takahashi, 2018</xref>). Another study has also emphasised on extracellular factors involved in the pathogenesis of <italic>C. difficile</italic>. A unique HP named, CD630_28,300 was found to share sequence similarity with zinc metallopeptidase, which demonstrated the binding of zinc with CD630_28,300 and its ability to disrupt the human fibronectin network by cleaving the fibronectin and fibrinogen <italic>in vitro</italic> in a zinc-dependent manner (<xref ref-type="bibr" rid="B10">Cafardi et al., 2013</xref>). Researchers have also employed similarity searches between pathogen and host, essentiality analysis, metabolic functional association, and choke point analysis. They identified 19 promising drug targets which were non-homologous to host proteins, and participated in four pathogen specific pathways, of which the peptidoglycan biosynthesis was found to be the highest contributor to the list of potential target proteins. MurG enzyme from the peptidoglycan biosynthesis pathway was found as one of the potential targets (<xref ref-type="bibr" rid="B19">Ezhilarasan et al., 2013</xref>).</p>
<p>In this study, the <italic>C. difficile</italic> HPs identified from NCBI database were subjected to various <italic>in silico</italic> experiments like, physicochemical properties, subcellular localisation identification, transmembrane helices detection, comparative homology analysis, antigenicity analysis, allergenicity analysis, secretory nature detection, and AMR identification. Furthermore, the top antigenic HPs were shortlisted and subjected to structure prediction and binding site analysis. The analysis of six <italic>Clostriodioles</italic> strains resulted in identifying approximately 11% of HPs from around 3,500 proteins coded by nearly 4.1&#xa0;Mb genome size of each strain. The physicochemical properties of HP from 6&#xa0;<italic>C. difficile</italic> strains, namely<italic>,</italic> BR81, R20291, CF5, M120, 196, and 2,007,855, were analysed. In all these strains of <italic>C. difficile</italic>, approximately 70% of the HPs were found to be stable as they had instability index (II) value below 40. Isoelectric point (pI) and grand average of hydropathicity (GRAVY) value were other important physicochemical parameters that were determined. The pI of HP ranges from 4.05 to 11.99 in all the strains. Isoelectric point (pI) is that pH where the amino acid of protein has a net zero charge and hence does not move in a direct current electrical field. At pI solubility of protein is lowest and electro focussing system mobility is zero, thereby making proteins stable and compact at this pH. This information can be utilised to develop buffer system for protein purification by isoelectric focussing (<xref ref-type="bibr" rid="B27">Islam et al., 2015</xref>). The GRAVY number of proteins is the measure of its hydrophilicity or hydrophobicity which are combined in a hydropathy scale. A positive value indicates proteins are hydrophobic while a negative value indicates that they are hydrophilic (<xref ref-type="bibr" rid="B11">Chang and Yang, 2013</xref>). In present study, around 70% HP of these strains had negative GRAVY values, demonstrating that they are hydrophilic in nature.</p>
<p>Since proteins located on the cell membrane can act as potential vaccine targets and those in the cytoplasmic matrix can act as potential drug targets, therefore, knowledge from subcellular localization is an important parameter for functional characterization of a protein (<xref ref-type="bibr" rid="B42">Prabhu et al., 2020</xref>). Moreover, research suggests the role of cell surface proteins in Clostridial pathogenesis, yet not many cell surface or secreted proteins of the nosocomial pathogen <italic>C. difficile</italic> have been identified or functionally characterised (<xref ref-type="bibr" rid="B10">Cafardi et al., 2013</xref>). Protein subcellular localization of HPs from all six strains of <italic>C. difficile</italic> were examined by PSORTb tool which categorises them into four categories, namely, cytoplasmic, cytoplasmic membrane, extracellular and unknown, based on their location in the bacterial cell. Approximately, 32&#x2013;38% and 23&#x2013;27% HPs were localised in the cytoplasm and cytoplasmic membrane, respectively. Meanwhile, 1&#x2013;3% and 37&#x2013;40% of all HPs in these six strains were located in the extracellular space, or their location was unknown.</p>
<p>Prediction of signal peptides is a key feature to determine the transportation system of particular proteins and their cleavage site. All non-cytoplasmic proteins have signal peptides that facilitate the transport of proteins across the membrane to a designated cellular location or organelles (<xref ref-type="bibr" rid="B42">Prabhu et al., 2020</xref>). We have used SignalP 5.0 server and found that almost 87&#x2013;92% of HPs from each strain did not have signal peptides while the remaining 8&#x2013;13% proteins had signal peptides indicating their involvement in a secretory pathway. Membrane proteins are also involved in various biological processes like signalling, transport, energy transduction and pathogenesis and can act as potential drug targets. Thus, it is important to predict membrane proteins to develop potent drug molecules (<xref ref-type="bibr" rid="B42">Prabhu et al., 2020</xref>).</p>
<p>Consequently, to check whether these HPs <italic>C. difficile</italic> can act as potential vaccine targets, we evaluated their sequence homology with humans, virulence factor, antigenicity and allergenicity. For a protein to be considered a potential candidate, it should be non-homologous to human proteins to avoid cross-reactivity with them. Also, they should be antigenic, non-allergenic and can have presence of virulence factors, our study demonstrated that almost 99.7% HP from all strains were non homologous but only a minute fraction of them around 0.25&#x2013;1.45% has virulent factor. While around 36&#x2013;41% of all HP were antigenic and none of them were allergens. Additionally, AMR analysis identified that the glycopeptide resistance protein <italic>vanZ1</italic> gene, glycosylating toxin <italic>TcdB</italic> gene, holin-like glycosylating toxin export protein <italic>TcdE</italic>, glycosylating toxin sigma factor <italic>TcdR</italic>, and CDD family class D beta-lactamase <italic>blaCDD</italic> genes were common in all strains according to the AMR and virulent factor data generated by AMRFinderPlus for WGS sequences, of which the <italic>vanZ1</italic> and the <italic>blaCDD</italic> genes were AMR, rest of the above mentioned genes were virulent. The AMR genes identified by ResFinder in only the WGS strains of M120 and 2,007,855 were also identified as AMR genes by the AMRFinderPlus. In the protein sequence from RAST three genes <italic>vanZ1</italic>, <italic>tcdB</italic>, and <italic>blaR1</italic> were found in all six chains, of which the <italic>tcdB</italic> is a virulence factor. In the HPs from RAST, the AMRFinderPlus was able to identify genes with virulence factors and the <italic>tcdR</italic> gene was the common gene in all six chains. The HPs play a role in virulence and could be used as potential targets for drug discovery or as antigen to develop vaccines. These properties of virulence, stability, polarity, presence in cytoplasmic and extracellular regions, minimal number of transmembrane helices present, non-homology with the human proteome, antigenicity and non-allergenicity which these HPs show, further experimental and computational studies can be done to assess the potentiality of these HPs as targets for drug discovery and as vaccine candidates.</p>
</sec>
<sec sec-type="conclusion" id="s5">
<title>Conclusion</title>
<p>Identifying protein functions is crucial for understanding various biological processes. Here, we implemented <italic>in silico</italic> approaches to predict the function of HPs from six strains of the <italic>C</italic>. <italic>difficile.</italic> While employing various tools to annotate and characterise HPs, characteristic predictions like subcellular localization, secretory nature and physicochemical properties were suitable to understand particular features. Further, identification of AMR genes, prophage sequences, CRISPR-Cas9 genes, and elucidation of immunoinformatics properties has allowed us to identify proteins that might play important roles in the complex mechanisms and biological processes. In conclusion, the pipeline used in this study allowed us to screen and characterise candidate HPs for assigning protein functions and further broaden up the possibilities for better downstream validation.</p>
</sec>
</body>
<back>
<sec id="s6" sec-type="data-availability">
<title>Data Availability Statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s10">Supplementary Material</xref>, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s7">
<title>Author Contributions</title>
<p>BAA and JV conceptualised the idea for this study. BAA collected all the data, AD and AM contributed partly to collection of data. BAA, AD and AM performed the analyses, AD, IA and AM created the tables and figures, BAA wrote the original draft, AD, AM and DS also contributed to writing. DS and IA proofread the data. All authors participated in the discussions on the interpretation of results and the conclusions before approving the manuscript. JV and PS co-supervised and interpreted the data, proofread the manuscript, BAA and JV wrote and revised the manuscript.</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ack>
<p>All authors acknowledge <ext-link ext-link-type="uri" xlink:href="http://Bioclues.org">Bioclues.org</ext-link> for providing an open-platform for knowledge sharing.</p>
</ack>
<sec id="s10">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2022.878012/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2022.878012/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table3.XLSX" id="SM1" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image1.JPEG" id="SM2" mimetype="application/JPEG" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="DataSheet1.docx" id="SM3" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<sec id="s11">
<title>Abbreviations</title>
<p>C. difficile, <italic>Clostridium difficile</italic>; CDI, <italic>C. difficile</italic> infection; CDD, Conserved Domain Search; AMR, Antimicrobial resistance; II, Instability Index; GRAVY, Grand Average of Hydropathicity Value; CRISPR, Clustered regularly interspaced short palindromic repeats; Cas9, CRISPR associated protein 9; HP, Hypothetical Protein; NCBI, National Centre for Biotechnology Information; PHASTER, PHAge Search Tool Enhanced Release.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abbasi</surname>
<given-names>B. A.</given-names>
</name>
<name>
<surname>Saraf</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Sharma</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Sinha</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sood</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Identification of Vaccine Targets &#x26; Design of Vaccine against SARS-CoV-2 Coronavirus Using Computational and Deep Learning-Based Approaches</article-title>. <source>PeerJ</source> <volume>10</volume>, <fpage>e13380</fpage>. <pub-id pub-id-type="doi">10.7717/peerj.13380</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/35611169/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.7717/peerj.13380">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Identification+of+Vaccine+Targets+&#x26;+Design+of+Vaccine+against+SARS-CoV-2+Coronavirus+Using+Computational+and+Deep+Learning-Based+Approaches&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abt</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>McKenney</surname>
<given-names>P. T.</given-names>
</name>
<name>
<surname>Pamer</surname>
<given-names>E. G.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>
<italic>Clostridium difficile</italic> Colitis: Pathogenesis and Host Defence</article-title>. <source>Nat. Rev. Microbiol.</source> <volume>14</volume>, <fpage>609</fpage>&#x2013;<lpage>620</lpage>. <pub-id pub-id-type="doi">10.1038/nrmicro.2016.108</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/27573580/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/nrmicro.2016.108">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Clostridium+difficile+Colitis:+Pathogenesis+and+Host+Defence&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Altschul</surname>
<given-names>S. F.</given-names>
</name>
<name>
<surname>Gish</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Miller</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Myers</surname>
<given-names>E. W.</given-names>
</name>
<name>
<surname>Lipman</surname>
<given-names>D. J.</given-names>
</name>
</person-group> (<year>1990</year>). <article-title>Basic Local Alignment Search Tool</article-title>. <source>J. Mol. Biol.</source> <volume>215</volume>, <fpage>403</fpage>&#x2013;<lpage>410</lpage>. <pub-id pub-id-type="doi">10.1016/s0022-2836(05)80360-2</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/2231712/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/s0022-2836(05)80360-2">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Basic+Local+Alignment+Search+Tool&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arndt</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Grant</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>Marcu</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sajed</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Pon</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>PHASTER: A Better, Faster Version of the PHAST Phage Search Tool</article-title>. <source>Nucleic Acids Res.</source> <volume>44</volume>, <fpage>W16</fpage>&#x2013;<lpage>W21</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkw387</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/27141966/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/nar/gkw387">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=PHASTER:+A+Better,+Faster+Version+of+the+PHAST+Phage+Search+Tool&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barbut</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Petit</surname>
<given-names>J.-C.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Epidemiology of <italic>Clostridium Difficile</italic>-Associated Infections</article-title>. <source>Clin. Microbiol. Infect.</source> <volume>7</volume>, <fpage>405</fpage>&#x2013;<lpage>410</lpage>. <pub-id pub-id-type="doi">10.1046/j.1198-743x.2001.00289.x</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/11591202/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1046/j.1198-743x.2001.00289.x">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Epidemiology+of+Clostridium+Difficile-Associated+Infections&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bartlett</surname>
<given-names>J. G.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Antibiotic-Associated Diarrhea</article-title>. <source>N. Engl. J. Med.</source> <volume>346</volume>, <fpage>334</fpage>&#x2013;<lpage>339</lpage>. <pub-id pub-id-type="doi">10.1056/nejmcp011603</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/11821511/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1056/nejmcp011603">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Antibiotic-Associated+Diarrhea&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Basak</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Deb</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Narsaria</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Kar</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Castiglione</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Sanyal</surname>
<given-names>I.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>
<italic>In Silico</italic> Designing of Vaccine Candidate against <italic>Clostridium difficile</italic>
</article-title>. <source>Sci. Rep.</source> <volume>11</volume>, <fpage>14215</fpage>&#x2013;<lpage>14222</lpage>. <pub-id pub-id-type="doi">10.1038/s41598-021-93305-6</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/34244557/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41598-021-93305-6">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=In+Silico+Designing+of+Vaccine+Candidate+against+Clostridium+difficile&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Boetzkes</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Felkel</surname>
<given-names>K. W.</given-names>
</name>
<name>
<surname>Zeiser</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jochim</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Just</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Pich</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Secretome Analysis of <italic>Clostridium difficile</italic> Strains</article-title>. <source>Arch. Microbiol.</source> <volume>194</volume>, <fpage>675</fpage>&#x2013;<lpage>687</lpage>. <pub-id pub-id-type="doi">10.1007/s00203-012-0802-5</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/22398929/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1007/s00203-012-0802-5">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Secretome+Analysis+of+Clostridium+difficile+Strains&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bortolaia</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Kaas</surname>
<given-names>R. S.</given-names>
</name>
<name>
<surname>Ruppe</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Roberts</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Schwarz</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Cattoir</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>ResFinder 4.0 for Predictions of Phenotypes from Genotypes</article-title>. <source>J. Antimicrob. Chemother.</source> <volume>75</volume>, <fpage>3491</fpage>&#x2013;<lpage>3500</lpage>. <pub-id pub-id-type="doi">10.1093/jac/dkaa345</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/32780112/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/jac/dkaa345">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=ResFinder+4.0+for+Predictions+of+Phenotypes+from+Genotypes&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cafardi</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Biagini</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Martinelli</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Leuzzi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Rubino</surname>
<given-names>J. T.</given-names>
</name>
<name>
<surname>Cantini</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Identification of a Novel Zinc Metalloprotease through a Global Analysis of <italic>Clostridium difficile</italic> Extracellular Proteins</article-title>. <source>PLoS One</source> <volume>8</volume>, <fpage>e81306</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0081306</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/24303041/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1371/journal.pone.0081306">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Identification+of+a+Novel+Zinc+Metalloprotease+through+a+Global+Analysis+of+Clostridium+difficile+Extracellular+Proteins&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chang</surname>
<given-names>K. Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J.-R.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Analysis and Prediction of Highly Effective Antiviral Peptides Based on Random Forests</article-title>. <source>PLoS One</source> <volume>8</volume>, <fpage>e70166</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0070166</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/23940542/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1371/journal.pone.0070166">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Analysis+and+Prediction+of+Highly+Effective+Antiviral+Peptides+Based+on+Random+Forests&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2005</year>). <article-title>VFDB: A Reference Database for Bacterial Virulence Factors</article-title>. <source>Nucleic Acids Res.</source> <volume>33</volume>, <fpage>D325</fpage>&#x2013;<lpage>D328</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gki008</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/15608208/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/nar/gki008">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=VFDB:+A+Reference+Database+for+Bacterial+Virulence+Factors&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Couvin</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Bernheim</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Toffano-Nioche</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Touchon</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Michalik</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>N&#xe9;ron</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>CRISPRCasFinder, an Update of CRISRFinder, Includes a Portable Version, Enhanced Performance and Integrates Search for Cas Proteins</article-title>. <source>Nucleic Acids Res.</source> <volume>46</volume>, <fpage>W246</fpage>&#x2013;<lpage>W251</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gky425</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/29790974/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/nar/gky425">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=CRISPRCasFinder,+an+Update+of+CRISRFinder,+Includes+a+Portable+Version,+Enhanced+Performance+and+Integrates+Search+for+Cas+Proteins&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Czepiel</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Dr&#xf3;&#x17c;d&#x17c;</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pituch</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Kuijper</surname>
<given-names>E. J.</given-names>
</name>
<name>
<surname>Perucki</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Mielimonka</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>
<italic>Clostridium Difficile</italic> Infection: Review</article-title>. <source>Eur. J. Clin. Microbiol. Infect. Dis.</source> <volume>38</volume>, <fpage>1211</fpage>&#x2013;<lpage>1221</lpage>. <pub-id pub-id-type="doi">10.1007/s10096-019-03539-6</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/30945014/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1007/s10096-019-03539-6">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Clostridium+Difficile+Infection:+Review&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>da Costa</surname>
<given-names>W. L. O.</given-names>
</name>
<name>
<surname>Ara&#xfa;jo</surname>
<given-names>C. L. d. A.</given-names>
</name>
<name>
<surname>Dias</surname>
<given-names>L. M.</given-names>
</name>
<name>
<surname>Pereira</surname>
<given-names>L. C. d. S.</given-names>
</name>
<name>
<surname>Alves</surname>
<given-names>J. T. C.</given-names>
</name>
<name>
<surname>Ara&#xfa;jo</surname>
<given-names>F. A.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Functional Annotation of Hypothetical Proteins from the <italic>Exiguobacterium Antarcticum</italic> Strain B7 Reveals Proteins Involved in Adaptation to Extreme Environments, Including High Arsenic Resistance</article-title>. <source>PLoS One</source> <volume>13</volume>, <fpage>e0198965</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0198965</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/29940001/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1371/journal.pone.0198965">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Functional+Annotation+of+Hypothetical+Proteins+from+the+Exiguobacterium+Antarcticum+Strain+B7+Reveals+Proteins+Involved+in+Adaptation+to+Extreme+Environments,+Including+High+Arsenic+Resistance&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Deveau</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Garneau</surname>
<given-names>J. E.</given-names>
</name>
<name>
<surname>Moineau</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>CRISPR/Cas System and its Role in Phage-Bacteria Interactions</article-title>. <source>Annu. Rev. Microbiol.</source> <volume>64</volume>, <fpage>475</fpage>&#x2013;<lpage>493</lpage>. <pub-id pub-id-type="doi">10.1146/annurev.micro.112408.134123</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/20528693/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1146/annurev.micro.112408.134123">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=CRISPR/Cas+System+and+its+Role+in+Phage-Bacteria+Interactions&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Didelot</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Eyre</surname>
<given-names>D. W.</given-names>
</name>
<name>
<surname>Cule</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ip</surname>
<given-names>C. L.</given-names>
</name>
<name>
<surname>Ansari</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Griffiths</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Microevolutionary Analysis of <italic>Clostridium Difficile</italic> Genomes to Investigate Transmission</article-title>. <source>Genome Biol.</source> <volume>13</volume> (<issue>12</issue>), <fpage>1188</fpage>&#x2013;<lpage>R213</lpage>. <pub-id pub-id-type="doi">10.1186/gb-2012-13-12-r118</pub-id> <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1186/gb-2012-13-12-r118">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Microevolutionary+Analysis+of+Clostridium+Difficile+Genomes+to+Investigate+Transmission&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Doytchinova</surname>
<given-names>I. A.</given-names>
</name>
<name>
<surname>Flower</surname>
<given-names>D. R.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>VaxiJen: A Server for Prediction of Protective Antigens, Tumour Antigens and Subunit Vaccines</article-title>. <source>BMC Bioinform.</source> <volume>8</volume>, <fpage>4</fpage>&#x2013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-8-4</pub-id> <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1186/1471-2105-8-4">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=VaxiJen:+A+Server+for+Prediction+of+Protective+Antigens,+Tumour+Antigens+and+Subunit+Vaccines&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ezhilarasan</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Sharma</surname>
<given-names>O. P.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>
<italic>In Silico</italic> identification of Potential Drug Targets in <italic>Clostridium difficile</italic> R20291: Modeling and Virtual Screening Analysis of a Candidate Enzyme MurG</article-title>. <source>Med. Chem. Res.</source> <volume>22</volume>, <fpage>2692</fpage>&#x2013;<lpage>2705</lpage>. <pub-id pub-id-type="doi">10.1007/s00044-012-0262-0</pub-id> <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1007/s00044-012-0262-0">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=In+Silico+identification+of+Potential+Drug+Targets+in+Clostridium+difficile+R20291:+Modeling+and+Virtual+Screening+Analysis+of+a+Candidate+Enzyme+MurG&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Feldgarden</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Brover</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Gonzalez-Escalona</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Frye</surname>
<given-names>J. G.</given-names>
</name>
<name>
<surname>Haendiges</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Haft</surname>
<given-names>D. H.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>AMRFinderPlus and the Reference Gene Catalog Facilitate Examination of the Genomic Links Among Antimicrobial Resistance, Stress Response, and Virulence</article-title>. <source>Sci. Rep.</source> <volume>11</volume>, <fpage>12728</fpage>&#x2013;<lpage>12729</lpage>. <pub-id pub-id-type="doi">10.1038/s41598-021-91456-0</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/34135355/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/s41598-021-91456-0">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=AMRFinderPlus+and+the+Reference+Gene+Catalog+Facilitate+Examination+of+the+Genomic+Links+Among+Antimicrobial+Resistance,+Stress+Response,+and+Virulence&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B21">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Gasteiger</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Hoogland</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Gattiker</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Duvaud</surname>
<given-names>S. e.</given-names>
</name>
<name>
<surname>Wilkins</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Appel</surname>
<given-names>R. D.</given-names>
</name>
<etal/>
</person-group> (<year>2005</year>). &#x201c;<article-title>Protein Identification and Analysis Tools on the ExPASy Server</article-title>,&#x201d; in <source>The Proteomics Protocols Handbook</source>, <fpage>571</fpage>&#x2013;<lpage>607</lpage>. <pub-id pub-id-type="doi">10.1385/1-59259-890-0:571</pub-id> <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1385/1-59259-890-0:571">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Protein+Identification+and+Analysis+Tools+on+the+ExPASy+Server&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goodman</surname>
<given-names>R. E.</given-names>
</name>
<name>
<surname>Ebisawa</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ferreira</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Sampson</surname>
<given-names>H. A.</given-names>
</name>
<name>
<surname>van Ree</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Vieths</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>AllergenOnline: A Peer-Reviewed, Curated Allergen Database to Assess Novel Food Proteins for Potential Cross-Reactivity</article-title>. <source>Mol. Nutr. Food Res.</source> <volume>60</volume>, <fpage>1183</fpage>&#x2013;<lpage>1198</lpage>. <pub-id pub-id-type="doi">10.1002/mnfr.201500769</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/26887584/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1002/mnfr.201500769">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=AllergenOnline:+A+Peer-Reviewed,+Curated+Allergen+Database+to+Assess+Novel+Food+Proteins+for+Potential+Cross-Reactivity&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goudarzi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Seyedjavadi</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Goudarzi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Mehdizadeh Aghdam</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Nazeri</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>
<italic>Clostridium Difficile</italic> Infection: Epidemiology, Pathogenesis, Risk Factors, and Therapeutic Options</article-title>. <source>Sci. (Cairo)</source> <volume>2014</volume>, <fpage>916826</fpage>. <pub-id pub-id-type="doi">10.1155/2014/916826</pub-id> <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1155/2014/916826">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Clostridium+Difficile+Infection:+Epidemiology,+Pathogenesis,+Risk+Factors,+and+Therapeutic+Options&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>He</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sebaihia</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lawley</surname>
<given-names>T. D.</given-names>
</name>
<name>
<surname>Stabler</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Dawson</surname>
<given-names>L. F.</given-names>
</name>
<name>
<surname>Martin</surname>
<given-names>M. J.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>Evolutionary Dynamics of <italic>Clostridium Difficile</italic> over Short and Long Time Scales</article-title>. <source>Proc. Natl. Acad. Sci. U.S.A.</source> <volume>107</volume> (<issue>16</issue>), <fpage>7527</fpage>&#x2013;<lpage>7532</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.0914322107</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/20368420/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1073/pnas.0914322107">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Evolutionary+Dynamics+of+Clostridium+Difficile+over+Short+and+Long+Time+Scales&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hong</surname>
<given-names>H. A.</given-names>
</name>
<name>
<surname>Ferreira</surname>
<given-names>W. T.</given-names>
</name>
<name>
<surname>Hosseini</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Anwar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hitri</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Wilkinson</surname>
<given-names>A. J.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>The Spore Coat Protein CotE Facilitates Host Colonization by <italic>Clostridium difficile</italic>
</article-title>. <source>J. Infect. Dis.</source> <volume>216</volume>, <fpage>1452</fpage>&#x2013;<lpage>1459</lpage>. <pub-id pub-id-type="doi">10.1093/infdis/jix488</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/28968845/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/infdis/jix488">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=The+Spore+Coat+Protein+CotE+Facilitates+Host+Colonization+by+Clostridium+difficile&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ijaq</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Malik</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Das</surname>
<given-names>P. S.</given-names>
</name>
<name>
<surname>Meena</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Bethi</surname>
<given-names>N.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>A Model to Predict the Function of Hypothetical Proteins through a Nine-point Classification Scoring Schema</article-title>. <source>BMC Bioinform.</source> <volume>20</volume> (<issue>1</issue>), <fpage>14</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-018-2554-y</pub-id> <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1186/s12859-018-2554-y">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=A+Model+to+Predict+the+Function+of+Hypothetical+Proteins+through+a+Nine-point+Classification+Scoring+Schema&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Islam</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Shahik</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Sohel</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Patwary</surname>
<given-names>N. I. A.</given-names>
</name>
<name>
<surname>Hasan</surname>
<given-names>M. A.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>In Silico Structural and Functional Annotation of Hypothetical Proteins ofVibrio Cholerae O139</article-title>. <source>Genomics Inf.</source> <volume>13</volume>, <fpage>53</fpage>. <pub-id pub-id-type="doi">10.5808/gi.2015.13.2.53</pub-id> <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.5808/gi.2015.13.2.53">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=In+Silico+Structural+and+Functional+Annotation+of+Hypothetical+Proteins+ofVibrio+Cholerae+O139&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Iyer</surname>
<given-names>L. M.</given-names>
</name>
<name>
<surname>Burroughs</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Anand</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>de Souza</surname>
<given-names>R. F.</given-names>
</name>
<name>
<surname>Aravind</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Polyvalent Proteins, a Pervasive Theme in the Intergenomic Biological Conflicts of Bacteriophages and Conjugative Elements</article-title>. <source>J. Bacteriol.</source> <volume>199</volume> (<issue>15</issue>), <fpage>e00245</fpage>&#x2013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.1128/JB.00245-17</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/28559295/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1128/JB.00245-17">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Polyvalent+Proteins,+a+Pervasive+Theme+in+the+Intergenomic+Biological+Conflicts+of+Bacteriophages+and+Conjugative+Elements&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B29">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Korman</surname>
<given-names>T. M.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Diagnosis and Management of <italic>Clostridium difficile</italic> Infection</article-title>,&#x201d; in <source>Seminars in Respiratory and Critical Care Medicine</source> (<publisher-loc>Leipzig, Germany</publisher-loc>: <publisher-name>Thieme Medical Publishers</publisher-name>), <fpage>31</fpage>&#x2013;<lpage>43</lpage>. <pub-id pub-id-type="doi">10.1055/s-0034-1398741</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/25643269/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1055/s-0034-1398741">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Diagnosis+and+Management+of+Clostridium+difficile+Infection&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Leber</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hontecillas</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Abedi</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Tubau-Juni</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Zoccoli-Rodriguez</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Stewart</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Modeling New Immunoregulatory Therapeutics as Antimicrobial Alternatives for Treating <italic>Clostridium difficile</italic> Infection</article-title>. <source>Artif. Intell. Med.</source> <volume>78</volume>, <fpage>1</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1016/j.artmed.2017.05.003</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/28764868/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.artmed.2017.05.003">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Modeling+New+Immunoregulatory+Therapeutics+as+Antimicrobial+Alternatives+for+Treating+Clostridium+difficile+Infection&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lessa</surname>
<given-names>F. C.</given-names>
</name>
<name>
<surname>Mu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Bamberg</surname>
<given-names>W. M.</given-names>
</name>
<name>
<surname>Beldavs</surname>
<given-names>Z. G.</given-names>
</name>
<name>
<surname>Dumyati</surname>
<given-names>G. K.</given-names>
</name>
<name>
<surname>Dunn</surname>
<given-names>J. R.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Burden of <italic>Clostridium difficile</italic> Infection in the United States</article-title>. <source>N. Engl. J. Med.</source> <volume>372</volume>, <fpage>825</fpage>&#x2013;<lpage>834</lpage>. <pub-id pub-id-type="doi">10.1056/nejmoa1408913</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/25714160/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1056/nejmoa1408913">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Burden+of+Clostridium+difficile+Infection+in+the+United+States&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Felgner</surname>
<given-names>P. L.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Predicting Antigenicity of Proteins in a Bacterial Proteome; a Protein Microarray and Na&#xef;ve Bayes Classification Approach</article-title>. <source>Chem. Biodivers.</source> <volume>9</volume>, <fpage>977</fpage>&#x2013;<lpage>990</lpage>. <pub-id pub-id-type="doi">10.1002/cbdv.201100360</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/22589097/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1002/cbdv.201100360">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Predicting+Antigenicity+of+Proteins+in+a+Bacterial+Proteome;+a+Protein+Microarray+and+Na&#xef;ve+Bayes+Classification+Approach&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chitsaz</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Derbyshire</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Geer</surname>
<given-names>R. C.</given-names>
</name>
<name>
<surname>Gonzales</surname>
<given-names>N. R.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>CDD/SPARCLE: The Conserved Domain Database in 2020</article-title>. <source>Nucleic Acids Res.</source> <volume>48</volume> (<issue>D1</issue>), <fpage>D265</fpage>&#x2013;<lpage>D268</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkz991</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/31777944/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/nar/gkz991">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=CDD/SPARCLE:+The+Conserved+Domain+Database+in+2020&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Marchler-Bauer</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Derbyshire</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Gonzales</surname>
<given-names>N. R.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chitsaz</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Geer</surname>
<given-names>L. Y.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>CDD: NCBI&#x27;s Conserved Domain Database</article-title>. <source>Nucleic Acids Res.</source> <volume>43</volume>, <fpage>D222</fpage>&#x2013;<lpage>D226</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gku1221</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/25414356/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/nar/gku1221">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=CDD:+NCBI&#x27;s+Conserved+Domain+Database&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mohammad</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tripathi</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Sobia</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Histamine, Histamine Receptors, and Their Role in Immunomodulation: An Updated Systematic. Section of Immunology</article-title>. <source>Open Immunol. J.</source> <volume>2</volume>, <fpage>9</fpage>&#x2013;<lpage>41</lpage>. <pub-id pub-id-type="doi">10.2174/1874226200902010009</pub-id> <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.2174/1874226200902010009">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Histamine,+Histamine+Receptors,+and+Their+Role+in+Immunomodulation:+An+Updated+Systematic.+Section+of+Immunology&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mori</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Takahashi</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Characteristics and Immunological Roles of Surface Layer Proteins in <italic>Clostridium difficile</italic>
</article-title>. <source>Ann. Lab. Med.</source> <volume>38</volume>, <fpage>189</fpage>&#x2013;<lpage>195</lpage>. <pub-id pub-id-type="doi">10.3343/alm.2018.38.3.189</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/29401552/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3343/alm.2018.38.3.189">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Characteristics+and+Immunological+Roles+of+Surface+Layer+Proteins+in+Clostridium+difficile&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nelson</surname>
<given-names>D. E.</given-names>
</name>
<name>
<surname>Auerbach</surname>
<given-names>S. B.</given-names>
</name>
<name>
<surname>Baltch</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Desjardin</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Beck-Sague</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Rheal</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>1994</year>). <article-title>Epidemic Clostridium Difficile-Associated Diarrhea: Role of Second- and Third-Generation Cephalosporins</article-title>. <source>Infect. Control Hosp. Epidemiol.</source> <volume>15</volume>, <fpage>88</fpage>&#x2013;<lpage>94</lpage>. <pub-id pub-id-type="doi">10.1086/646867</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/8201240/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1086/646867">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Epidemic+Clostridium+Difficile-Associated+Diarrhea:+Role+of+Second-+and+Third-Generation+Cephalosporins&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Omeershffudin</surname>
<given-names>U. N. M.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>
<italic>In Silico</italic> Approach for Mining of Potential Drug Targets from Hypothetical Proteins of Bacterial Proteome</article-title>. <source>Int. J. Mol. Biol. Open Access</source> <volume>4</volume>, <fpage>145</fpage>&#x2013;<lpage>152</lpage>. <pub-id pub-id-type="doi">10.15406/ijmboa.2019.04.00111</pub-id> <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.15406/ijmboa.2019.04.00111">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=In+Silico+Approach+for+Mining+of+Potential+Drug+Targets+from+Hypothetical+Proteins+of+Bacterial+Proteome&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Overbeek</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Olson</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Pusch</surname>
<given-names>G. D.</given-names>
</name>
<name>
<surname>Olsen</surname>
<given-names>G. J.</given-names>
</name>
<name>
<surname>Davis</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Disz</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>The SEED and the Rapid Annotation of Microbial Genomes Using Subsystems Technology (RAST)</article-title>. <source>Nucl. Acids Res.</source> <volume>42</volume>, <fpage>D206</fpage>&#x2013;<lpage>D214</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkt1226</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/24293654/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/nar/gkt1226">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=The+SEED+and+the+Rapid+Annotation+of+Microbial+Genomes+Using+Subsystems+Technology+(RAST)&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>P&#xe9;chin&#xe9;</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gleizes</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Janoir</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Gorges-Kergot</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Barc</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Delm&#xe9;e</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2005</year>). <article-title>Immunological Properties of Surface Proteins of <italic>Clostridium difficile</italic>
</article-title>. <source>J. Med. Microbiol.</source> <volume>54</volume>, <fpage>193</fpage>&#x2013;<lpage>196</lpage>. <pub-id pub-id-type="doi">10.1099/jmm.0.45800-0</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/15673516/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1099/jmm.0.45800-0">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Immunological+Properties+of+Surface+Proteins+of+Clostridium+difficile&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Petersen</surname>
<given-names>T. N.</given-names>
</name>
<name>
<surname>Brunak</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Von Heijne</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Nielsen</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Signal P 4.0: Discriminating Signal Peptides from Transmembrane Regions</article-title>. <source>Nat. Methods</source> <volume>8</volume>, <fpage>785</fpage>&#x2013;<lpage>786</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.1701</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/21959131/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/nmeth.1701">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Signal+P+4.0:+Discriminating+Signal+Peptides+from+Transmembrane+Regions&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Prabhu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Rajamanikandan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Anusha</surname>
<given-names>S. B.</given-names>
</name>
<name>
<surname>Chowdary</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Veerapandiyan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Jeyakanthan</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>
<italic>In Silico</italic> functional Annotation and Characterization of Hypothetical Proteins from <italic>Serratia marcescens</italic> FGI94</article-title>. <source>Biol. Bull. Russ. Acad. Sci.</source> <volume>47</volume>, <fpage>319</fpage>&#x2013;<lpage>331</lpage>. <pub-id pub-id-type="doi">10.1134/s1062359020300019</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/32834707/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1134/s1062359020300019">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=In+Silico+functional+Annotation+and+Characterization+of+Hypothetical+Proteins+from+Serratia+marcescens+FGI94&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ran</surname>
<given-names>F. A.</given-names>
</name>
<name>
<surname>Hsu</surname>
<given-names>P. D.</given-names>
</name>
<name>
<surname>Wright</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Agarwala</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Scott</surname>
<given-names>D. A.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Genome Engineering Using the CRISPR-Cas9 System</article-title>. <source>Nat. Protoc.</source> <volume>8</volume>, <fpage>2281</fpage>&#x2013;<lpage>2308</lpage>. <pub-id pub-id-type="doi">10.1038/nprot.2013.143</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/24157548/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/nprot.2013.143">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Genome+Engineering+Using+the+CRISPR-Cas9+System&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rineh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kelso</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Vatansever</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Tegos</surname>
<given-names>G. P.</given-names>
</name>
<name>
<surname>Hamblin</surname>
<given-names>M. R.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Clostridium Difficileinfection: Molecular Pathogenesis and Novel Therapeutics</article-title>. <source>Expert Rev. Anti Infective Ther.</source> <volume>12</volume>, <fpage>131</fpage>&#x2013;<lpage>150</lpage>. <pub-id pub-id-type="doi">10.1586/14787210.2014.866515</pub-id> <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1586/14787210.2014.866515">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Clostridium+Difficileinfection:+Molecular+Pathogenesis+and+Novel+Therapeutics&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Roy</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kucukural</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>I-TASSER: A Unified Platform for Automated Protein Structure and Function Prediction</article-title>. <source>Nat. Protoc.</source> <volume>5</volume>, <fpage>725</fpage>&#x2013;<lpage>738</lpage>. <pub-id pub-id-type="doi">10.1038/nprot.2010.5</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/20360767/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/nprot.2010.5">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=I-TASSER:+A+Unified+Platform+for+Automated+Protein+Structure+and+Function+Prediction&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sebaihia</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wren</surname>
<given-names>B. W.</given-names>
</name>
<name>
<surname>Mullany</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Fairweather</surname>
<given-names>N. F.</given-names>
</name>
<name>
<surname>Minton</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Stabler</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2006</year>). <article-title>The Multidrug-Resistant Human Pathogen <italic>Clostridium difficile</italic> Has a Highly Mobile, Mosaic Genome</article-title>. <source>Nat. Genet.</source> <volume>38</volume>, <fpage>779</fpage>&#x2013;<lpage>786</lpage>. <pub-id pub-id-type="doi">10.1038/ng1830</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/16804543/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/ng1830">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=The+Multidrug-Resistant+Human+Pathogen+Clostridium+difficile+Has+a+Highly+Mobile,+Mosaic+Genome&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Segar</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Easow</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Srirangaraj</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hanifah</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Joseph</surname>
<given-names>N. M.</given-names>
</name>
<name>
<surname>Seetha</surname>
<given-names>K. S.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Prevalence of <italic>Clostridium difficile</italic> Infection Among the Patients Attending a Tertiary Care Teaching Hospital</article-title>. <source>Indian J. Pathol. Microbiol.</source> <volume>60</volume>, <fpage>221</fpage>&#x2013;<lpage>225</lpage>. <pub-id pub-id-type="doi">10.4103/0377-4929.208383</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/28631639/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.4103/0377-4929.208383">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Prevalence+of+Clostridium+difficile+Infection+Among+the+Patients+Attending+a+Tertiary+Care+Teaching+Hospital&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Singh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Singal</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Nath</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>I. K.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Functional Annotation and Classification of the Hypothetical Proteins of Neisseria Meningitidis H44/76</article-title>. <source>Bio</source> <volume>3</volume>, <fpage>57</fpage>&#x2013;<lpage>64</lpage>. <pub-id pub-id-type="doi">10.11648/j.bio.20150305.16</pub-id> <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.11648/j.bio.20150305.16">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Functional+Annotation+and+Classification+of+the+Hypothetical+Proteins+of+Neisseria+Meningitidis+H44/76&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sivashankari</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shanmughavel</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Functional Annotation of Hypothetical Proteins - A Review</article-title>. <source>Bioinformation</source> <volume>1</volume>, <fpage>335</fpage>&#x2013;<lpage>338</lpage>. <pub-id pub-id-type="doi">10.6026/97320630001335</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/17597916/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.6026/97320630001335">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Functional+Annotation+of+Hypothetical+Proteins+-+A+Review&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Smits</surname>
<given-names>W. K.</given-names>
</name>
<name>
<surname>Lyras</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Lacy</surname>
<given-names>D. B.</given-names>
</name>
<name>
<surname>Wilcox</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Kuijper</surname>
<given-names>E. J.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>
<italic>Clostridium difficile</italic> Infection</article-title>. <source>Nat. Rev. Dis. Prim.</source> <volume>2</volume>, <fpage>16020</fpage>&#x2013;<lpage>20</lpage>. <pub-id pub-id-type="doi">10.1038/nrdp.2016.20</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/27158839/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/nrdp.2016.20">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Clostridium+difficile+Infection&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stabler</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Dawson</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Martin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Valiente</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Corton</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>Comparative Genome and Phenotypic Analysis of <italic>Clostridium difficile</italic> 027 Strains Provides Insight into the Evolution of a Hypervirulent Bacterium</article-title>. <source>Genome Biol.</source> <volume>10</volume>, <fpage>R102</fpage>&#x2013;<lpage>R115</lpage>. <pub-id pub-id-type="doi">10.1186/gb-2009-10-9-r102</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/19781061/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1186/gb-2009-10-9-r102">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Comparative+Genome+and+Phenotypic+Analysis+of+Clostridium+difficile+027+Strains+Provides+Insight+into+the+Evolution+of+a+Hypervirulent+Bacterium&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Suravajhala</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Benso</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Valadi</surname>
<given-names>J. K.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Annotation and Curation of Uncharacterized Proteins: Systems Biology Approaches</article-title>. <source>Front. Genet.</source> <volume>6</volume>, <fpage>224</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2015.00224</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/26175751/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2015.00224">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=Annotation+and+Curation+of+Uncharacterized+Proteins:+Systems+Biology+Approaches&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tian</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Lei</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>CASTp 3.0: Computed Atlas of Surface Topography of Proteins</article-title>. <source>Nucleic Acids Res.</source> <volume>46</volume> (<issue>W1</issue>), <fpage>W363</fpage>&#x2013;<lpage>W367</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gky473</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/29860391/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/nar/gky473">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=CASTp+3.0:+Computed+Atlas+of+Surface+Topography+of+Proteins&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Varma</surname>
<given-names>P. B. S.</given-names>
</name>
<name>
<surname>Adimulam</surname>
<given-names>Y. B.</given-names>
</name>
<name>
<surname>Kodukula</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>In Silico Functional Annotation of a Hypothetical Protein from <italic>Staphylococcus Aureus</italic>
</article-title>. <source>J. Infect. Public Health</source> <volume>8</volume>, <fpage>526</fpage>&#x2013;<lpage>532</lpage>. <pub-id pub-id-type="doi">10.1016/j.jiph.2015.03.007</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/26025048/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.jiph.2015.03.007">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=In+Silico+Functional+Annotation+of+a+Hypothetical+Protein+from+Staphylococcus+Aureus&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vindigni</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Surawicz</surname>
<given-names>C. M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>
<italic>C. difficile</italic> Infection: Changing Epidemiology and Management Paradigms</article-title>. <source>Clin. Transl. Gastroenterol.</source> <volume>6</volume>, <fpage>e99</fpage>. <pub-id pub-id-type="doi">10.1038/ctg.2015.24</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/26158611/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1038/ctg.2015.24">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=C.+difficile+Infection:+Changing+Epidemiology+and+Management+Paradigms&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>N. Y.</given-names>
</name>
<name>
<surname>Wagner</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>Laird</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Melli</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Rey</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lo</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>PSORTb 3.0: Improved Protein Subcellular Localization Prediction with Refined Localization Subcategories and Predictive Capabilities for All Prokaryotes</article-title>. <source>Bioinformatics</source> <volume>26</volume>, <fpage>1608</fpage>&#x2013;<lpage>1615</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btq249</pub-id> <ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/20472543/">PubMed Abstract</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/bioinformatics/btq249">CrossRef Full Text</ext-link> &#x7c; <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&#x0026;as_sdt=0%2C5&#x0026;q=PSORTb+3.0:+Improved+Protein+Subcellular+Localization+Prediction+with+Refined+Localization+Subcategories+and+Predictive+Capabilities+for+All+Prokaryotes&#x0026;btnG=">Google Scholar</ext-link>
</citation>
</ref>
</ref-list>
</back>
</article>