<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Microbiol.</journal-id>
<journal-title>Frontiers in Microbiology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Microbiol.</abbrev-journal-title>
<issn pub-type="epub">1664-302X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmicb.2022.1076797</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Microbiology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Pan-genome association study of <italic>Mycobacterium tuberculosis</italic> lineage-4 revealed specific genes related to the high and low prevalence of the disease in patients from the North-Eastern area of Medell&#x00ED;n, Colombia</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Hurtado-P&#x00E1;ez</surname> <given-names>Uriel</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1952659/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>&#x00C1;lvarez Zuluaga</surname> <given-names>Nataly</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2143068/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Arango Isaza</surname> <given-names>Rafael Eduardo</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/602448/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Contreras-Moreira</surname> <given-names>Bruno</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/163689/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Rouzaud</surname> <given-names>Fran&#x00E7;ois</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Robledo</surname> <given-names>Jaime</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1257422/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Corporaci&#x00F3;n para Investigaciones Biol&#x00F3;gicas (CIB)</institution>, <addr-line>Medell&#x00ED;n</addr-line>, <country>Colombia</country></aff>
<aff id="aff2"><sup>2</sup><institution>Facultad de Ciencias, Universidad Nacional de Colombia (UNAL)</institution>, <addr-line>Medell&#x00ED;n</addr-line>, <country>Colombia</country></aff>
<aff id="aff3"><sup>3</sup><institution>Estaci&#x00F3;n Experimental de Aula Dei&#x2013;Consejo Superior de Investigaciones Cient&#x00ED;ficas (EEAD-CSIC)</institution>, <addr-line>Zaragoza</addr-line>, <country>Spain</country></aff>
<aff id="aff4"><sup>4</sup><institution>Fundaci&#x00F3;n ARAID</institution>, <addr-line>Zaragoza</addr-line>, <country>Spain</country></aff>
<aff id="aff5"><sup>5</sup><institution>Ministry of Agriculture</institution>, <addr-line>Occitania</addr-line>, <country>France</country></aff>
<aff id="aff6"><sup>6</sup><institution>Escuela de Ciencias de la Salud, Universidad Pontificia Bolivariana (UPB)</institution>, <addr-line>Medell&#x00ED;n</addr-line>, <country>Colombia</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Antony T. Vincent, Laval University, Canada</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Indu Kumari, Regional Centre for Biotechnology (RCB), India; Arshad Rizvi, Department of Microbiology and Immunology, School of Medicine, Emory University, United States</p></fn>
<corresp id="c001">&#x002A;Correspondence: Uriel Hurtado-P&#x00E1;ez, <email>uahurtadop@unal.edu.co</email></corresp>
<fn fn-type="other" id="fn004"><p>This article was submitted to Evolutionary and Genomic Microbiology, a section of the journal Frontiers in Microbiology</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>04</day>
<month>01</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>1076797</elocation-id>
<history>
<date date-type="received">
<day>22</day>
<month>10</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>12</day>
<month>12</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2023 Hurtado-P&#x00E1;ez, &#x00C1;lvarez Zuluaga, Arango Isaza, Contreras-Moreira, Rouzaud and Robledo.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Hurtado-P&#x00E1;ez, &#x00C1;lvarez Zuluaga, Arango Isaza, Contreras-Moreira, Rouzaud and Robledo</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p><italic>Mycobacterium tuberculosis</italic> (<italic>Mtb</italic>) lineage 4 is responsible for the highest burden of tuberculosis (TB) worldwide. This lineage has been the most prevalent lineage in Colombia, especially in the North-Eastern (NE) area of Medellin, where it has been shown to have a high prevalence of LAM9 SIT42 and Haarlem1 SIT62 sublineages. There is evidence that regardless of environmental factors and host genetics, differences among sublineages of <italic>Mtb</italic> strains play an important role in the course of infection and disease. Nevertheless, the genetic basis of the success of a sublineage in a specific geographic area remains uncertain. We used a pan-genome-wide association study (pan-GWAS) of 47 <italic>Mtb</italic> strains isolated from NE Medellin between 2005 and 2008 to identify the genes responsible for the phenotypic differences among high and low prevalence sublineages. Our results allowed the identification of 12 variants in 11 genes, of which 4 genes showed the strongest association to low prevalence (<italic>mmpL12</italic>, <italic>PPE29</italic>, <italic>Rv1419</italic>, and <italic>Rv1762c</italic>). The first three have been described as necessary for invasion and intracellular survival. Polymorphisms identified in low prevalence isolates may suggest related to a fitness cost of <italic>Mtb</italic>, which might reflect a decrease in their capacity to be transmitted or to cause an active infection. These results contribute to understanding the success of some sublineages of lineage-4 in a specific geographical area.</p>
</abstract>
<kwd-group>
<kwd><italic>M. tuberculosis</italic></kwd>
<kwd>sublineage</kwd>
<kwd>transmission</kwd>
<kwd>prevalence</kwd>
<kwd>pan-GWAS</kwd>
<kwd>pan-genome</kwd>
<kwd>variant</kwd>
</kwd-group>
<counts>
<fig-count count="8"/>
<table-count count="3"/>
<equation-count count="0"/>
<ref-count count="77"/>
<page-count count="19"/>
<word-count count="12221"/>
</counts>
</article-meta>
</front>
<body>
<sec id="S1" sec-type="intro">
<title>Introduction</title>
<p>Tuberculosis (TB) is a major public health problem worldwide and the most prevalent infectious-contagious disease in the history of humankind (<xref ref-type="bibr" rid="B35">Galagan, 2014</xref>; <xref ref-type="bibr" rid="B21">Couvin and Rastogi, 2015</xref>; <xref ref-type="bibr" rid="B74">World Health Organization [WHO], 2018</xref>). According to the World Health Organization&#x2019;s (WHO&#x2019;s) &#x201C;Global TB report 2022&#x201D; the TB is one of the leading causes of death from a single infectious agent (above HIV/AIDS). Globally, in 2021 there were an estimated 10.6 million new cases with active disease and about 1.4 million deaths (<xref ref-type="bibr" rid="B75">World Health Organization [WHO], 2022</xref>). The disease is caused by members of the <italic>Mycobacterium tuberculosis</italic> complex (MTBC), which includes five human-adapted lineages representing <italic>Mtb</italic> sensu stricto, L1 (The Philippines and Indian Ocean), L2 (East Asia), L3 (India and East Africa), L4 (Europe and Americas), and L7 (Ethiopia), and two other human-adapted lineages defined to as <italic>M. africanum</italic> L5 (West African 1), L6 (West African 2), and more than eight animal-adapted lineages globally distributed, which have evolved from common ancestors from different geographic areas (<xref ref-type="bibr" rid="B51">Ngabonziza et al., 2020</xref>). There are studies in MTBC that have associated genotypes with different clinical presentations of the disease, the geographic distribution, and their prevalence (<xref ref-type="bibr" rid="B26">David et al., 2012</xref>; <xref ref-type="bibr" rid="B49">Merker et al., 2015</xref>). Indeed, it has been shown that modern lineages (L2 and L4) are more successful in their spread capacity, being more prevalent than ancient lineages (L1, L3, L5, and L6), likely due to greater virulence and shorter latency periods (<xref ref-type="bibr" rid="B37">Gonzalo-Asensio et al., 2018</xref>; <xref ref-type="bibr" rid="B51">Ngabonziza et al., 2020</xref>).</p>
<p>However, although modern <italic>Mtb</italic> lineages are more virulent and spread faster, their sublineages do not always behave similarly. It partially depends on environmental factors such as antibiotic resistance, host demography, and genetic heterogeneity, the latter with the presence of dominant sublineages with epidemic behavior, which can acquire functional advantages over other strains in their ability to transmit and cause disease (<xref ref-type="bibr" rid="B20">Coscolla and Gagneux, 2014</xref>; <xref ref-type="bibr" rid="B33">Folkvardsen et al., 2018</xref>). In Colombia, lineage-4 was found as the most predominant. A previous molecular epidemiology study showed the population structure of <italic>Mtb</italic> in certain geographic regions of the country with a dominance of the LAM and Haarlem sublineages, particularly in Medell&#x00ED;n and Cali (<xref ref-type="bibr" rid="B58">Realpe et al., 2014</xref>). This study found that in North-Eastern (NE) area of Medell&#x00ED;n the sublineages LAM9 SIT42, H1 SIT62 predominated, compared with other sublineages present in the same geographical area. However, there is still limited information about the genetic background of <italic>Mtb</italic> associated with the high or low prevalence of clinical isolates, which generates the need to increase knowledge in this field that contribute to the control of the <italic>Mtb</italic> transmission (<xref ref-type="bibr" rid="B33">Folkvardsen et al., 2018</xref>).</p>
<p>Due to the development of whole-genome sequencing (WGS), it is possible to perform comparative genomic analysis of several strains. These have now become an alternative in order to answer microbiological questions related to outbreaks, evolution, antibiotic resistance, pathogenicity, and transmission (<xref ref-type="bibr" rid="B69">Uchiya et al., 2017</xref>). Analyses of multiple genomes of individuals from the same species have revealed wide intra-species diversity, largely due to differences in the gene and transposable element repertoire of the strains (<xref ref-type="bibr" rid="B67">Tettelin et al., 2008</xref>). Here, we first explore the <italic>Mtb</italic> pan-genome by estimating the genomic diversity of 47 <italic>Mtb</italic> clinical isolates of lineage-4. This was done by identifying genes shared among all isolates under study (core genome), and genes present in some but not all the strains studied (dispensable genome), as well as the strain-specific genes (<xref ref-type="bibr" rid="B66">Tettelin et al., 2005</xref>). In addition, we report a pan-genome-wide association study (pan-GWAS) where genetic variants related to the phenotype of the high or low prevalence of <italic>Mtb</italic> clinical isolates were identified.</p>
</sec>
<sec id="S2" sec-type="materials|methods">
<title>Materials and methods</title>
<sec id="S2.SS1">
<title>Study population</title>
<p><italic>Mycobacterium tuberculosis</italic> strains were isolated from patients with pulmonary TB in the laboratory of Corporaci&#x00F3;n para Investigaciones Biol&#x00F3;gicas as part of the Colombian center for TB research developed in Colombia between 2005 and 2008. A total of 324 isolates were genotyped by spoligotyping (Spoligo-International-Type&#x2014;SIT) and 24-loci Mycobacterial Interspersed Repetitive Units-Variable-Number of Tandem Repeats (MIRU-VNTR). Out of all typed isolates, 135 (41.7%) of them were present in the NE area of Medell&#x00ED;n that has shown the highest incidence in the city per 100,000 inhabitants (81 in 2018), with an incidence average rate of 70 per 100,000 in last decade (<xref ref-type="bibr" rid="B61">Secretar&#x00ED;a Seccional de Salud y Protecci&#x00F3;n Social de Antioquia, 2017</xref>; <xref ref-type="bibr" rid="B2">Almanza et al., 2019</xref>). <italic>Mtb</italic> LAM9 SIT42 and Haarlem1 SIT62 sublineages had the highest prevalence (73.3%) compared with other sublineages that exhibit lower prevalence (less than 10%). Simple random sampling with Epidat v4.1 was performed on the different genotypes, selecting 27 high prevalence and 20 low prevalence isolates for this study (<xref ref-type="table" rid="T1">Table 1</xref>).</p>
<table-wrap position="float" id="T1">
<label>TABLE 1</label>
<caption><p><italic>Mycobacterium tuberculosis</italic> (<italic>Mtb</italic>) isolates recovered in the city of Medell&#x00ED;n between 2005 and 2006.</p></caption>
<table cellspacing="5" cellpadding="5" frame="box" rules="all">
<thead>
<tr>
<td valign="top" align="left" colspan="7" style="color:#ffffff;background-color: #7f8080;">Selected isolates in Medell&#x00ED;n (324)</td>
</tr>
<tr>
<td valign="top" align="left" colspan="5" style="color:#ffffff;background-color: #7f8080;">North-Eastern area (135)</td>
<td valign="top" align="left" colspan="2" style="color:#ffffff;background-color: #7f8080;">Selected isolates</td>
</tr>
<tr>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Clade</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">SIT</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">N. isolates</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Isolates (%)</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Prevalence</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">SRS<xref ref-type="table-fn" rid="t1fns1">&#x002A;</xref></td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Isolates 48</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">LAM9</td>
<td valign="top" align="center">42</td>
<td valign="top" align="center">49</td>
<td valign="top" align="center">36.30%</td>
<td valign="top" align="center">High</td>
<td valign="top" align="center">11</td>
<td valign="top" align="center" rowspan="2">27</td>
</tr>
<tr>
<td valign="top" align="left">H1</td>
<td valign="top" align="center">62</td>
<td valign="top" align="center">50</td>
<td valign="top" align="center">37.00%</td>
<td valign="top" align="center">High</td>
<td valign="top" align="center">16</td>
</tr>
<tr>
<td valign="top" align="left">H3</td>
<td valign="top" align="center">Diff<xref ref-type="table-fn" rid="t1fns1">&#x002A;</xref> (4)</td>
<td valign="top" align="center">11</td>
<td valign="top" align="center">&#x003C;8.1%</td>
<td valign="top" align="center">Low</td>
<td valign="top" align="center">9</td>
<td valign="top" align="center" rowspan="3">20</td>
</tr>
<tr>
<td valign="top" align="left">H1</td>
<td valign="top" align="center">Diff<xref ref-type="table-fn" rid="t1fns1">&#x002A;</xref> (6)</td>
<td valign="top" align="center">11</td>
<td valign="top" align="center">&#x003C;8.1%</td>
<td valign="top" align="center">Low</td>
<td valign="top" align="center">6</td>
</tr>
<tr>
<td valign="top" align="left">LAM</td>
<td valign="top" align="center">Diff<xref ref-type="table-fn" rid="t1fns1">&#x002A;</xref> (11)</td>
<td valign="top" align="center">14</td>
<td valign="top" align="center">&#x003C;10.4%</td>
<td valign="top" align="center">Low</td>
<td valign="top" align="center">5</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="t1fns1"><p>The 41.7% were recovered in the NE area of Medell&#x00ED;n. The word Diff&#x002A; corresponds to the number of isolates with different SIT and SRS&#x002A; corresponds to a simple random sample using Epidat v4.1.</p></fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="S2.SS2">
<title>DNA isolation and whole genome sequencing</title>
<p>All 47 <italic>Mtb</italic> isolates were grown in Middlebrook 7H11 medium at 37&#x00B0;C in a 5% CO2 atmosphere for 3 weeks. DNA isolation was performed as previously described (<xref ref-type="bibr" rid="B70">van Soolingen et al., 1994</xref>). In brief, three loopfuls of cells were suspended in 400 ml of 1&#x00D7; Tris-EDTA buffer 1X (10 mM Tris&#x2013;HCl pH 8, 0, 1 mM EDTA). Heat inactivated at 80&#x00B0;C for 45 min. Then, 50 &#x03BC;L of lysozyme (10 mg/ml) was added, and incubated for 15 min at 37&#x00B0;C. Then, 75 &#x03BC;l of 10% SDS and 6 &#x03BC;l proteinase K (10 mg/ml) was added and incubated 10 min at 65&#x00B0;C. Followed by the addition of 100 &#x03BC;L CTAB (cetyltrimethylammonium bromide) 10% and NaCl al 5% preheated solution at 65&#x00B0;C, vortexed and incubated at 10 min at 65&#x00B0;C. Protein/lipids separation was performed with 750 &#x03BC;L chloroform/isoamyl alcohol (24: 1, v/v) vortexed, and centrifuge for 8 min at 12,000 g. The supernatant was transferred to a 1.5 &#x03BC;l tube and 450 &#x03BC;L of 2-propanol was added for to nucleic acid precipitation at &#x2212;20&#x00B0;C for 30 min. After centrifugation, the pellet was washed with 70% EtOH and subsequently air-dried and re-suspended in 20 &#x03BC;l Tris-EDTA (TE) buffer at 37&#x00B0;C. Genomic DNA of the 47 <italic>Mtb</italic> isolates was sequenced at the Biotechnology Center of the University of Wisconsin using Illumina HiSeq 2500 (150, paired-end) with TruSeq Nano library prep according to the manufacturer protocol (Illumina, CA, USA).</p>
</sec>
<sec id="S2.SS3">
<title>Bioinformatic analysis: Pre-processing, <italic>de novo</italic> assembly, and annotation</title>
<p>Quality control of the sequencing raw data was performed with FastQC 0.11.3 (<xref ref-type="bibr" rid="B4">Andrews, 2010</xref>). The pre-processing tool Trimmomatic v3 (<xref ref-type="bibr" rid="B7">Bolger et al., 2014</xref>) was used to remove adapters, trimmed bases by position, clean artifacts and keep a minimum quality threshold Q22. <italic>De novo</italic> assembly of mycobacterial genomes was performed with SPAdes v3.8.2 (<xref ref-type="bibr" rid="B6">Bankevich et al., 2012</xref>) using different odd <italic>k</italic>-mer sizes in the range <italic>k</italic> = 21 to <italic>k</italic> = 87. Assembly stats such as statistics such as N50, largest contig, GC-content, genome fraction covered, and genes number were computed with Quast v3.1 (<xref ref-type="bibr" rid="B39">Gurevich et al., 2013</xref>) to assess the quality and compare it to the reference genome H37Rv (NC_000962.3). In addition, correction of assembly errors was carried out using Pilon v1.22 (<xref ref-type="bibr" rid="B72">Walker et al., 2014</xref>). To call features within contigs, genome annotation was performed using Prokka v1.11 (<xref ref-type="bibr" rid="B62">Seemann, 2014</xref>), with protein-coding genes predicted with Prodigal (<xref ref-type="bibr" rid="B41">Hyatt et al., 2010</xref>) and homology searches done against UniProt (Swiss-prot) database (<xref ref-type="bibr" rid="B5">Apweiler, 2004</xref>) using BLASTP (<xref ref-type="bibr" rid="B3">Altschul et al., 1990</xref>) with an <italic>E</italic>-value of 1e-6.</p>
</sec>
<sec id="S2.SS4">
<title>Pan-genome construction, core and accessory genome evolution</title>
<p>Identification of clusters of homologous genes within the 47 genomes annotated was performed with GET_HOMOLOGUES V09062017 (<xref ref-type="bibr" rid="B18">Contreras-Moreira and Vinuesa, 2013</xref>), using BLASTP with a sequence identity of 90% and minimum default query coverage of 75% for paired alignments. Once the local alignments were sorted and indexed, three algorithms, Bidirectional best-hit (BDBH) (<xref ref-type="bibr" rid="B18">Contreras-Moreira and Vinuesa, 2013</xref>), OrthoMCL (OMCL) (<xref ref-type="bibr" rid="B31">Fischer et al., 2011</xref>), COGtriangles (COG) (<xref ref-type="bibr" rid="B45">Kristensen et al., 2010</xref>) also included in GET_HOMOLOGUES software, were used to produce sequence clusters (both nucleotide and peptides) according to protein sequence similarity. Single-copy orthologous clusters were defined as those containing only one sequence from each input genome; extra copies were considered paralogous.</p>
<p>To estimate the core genome size, a genome composition analysis was done using the <italic>get_homologues.pl &#x2013;c</italic> script. Syntenic groups of genes were identified at the intersection of the three clustering algorithms to calculate the overlap of orthologous protein sets. Accordingly, we estimated the size of the intersected core genome, defined as the subset of clusters with genes from all genomes. Similarly, the intersection of the COG and OMLC algorithms generated a binary matrix in tabular format that summarized genes present (1) and absent (0) in each of the 47 genomes. This matrix represents the pan-genome, the complete repertoire of genes of the 47 <italic>Mtb</italic> genomes lineage-4.</p>
<p>In order to measure how much the core and the accessory genomes of the <italic>Mtb</italic> Lineage-4 change as new isolates are added, a genome composition analysis was performed. This is a simulation that estimates how many core and novel sequences are added by genomes sampled in random order. This was done after taking 10 replicates using random permutations of the 47 strains using the <italic>get_homologues.pl</italic> and <italic>plot_pancore_matrix.pl</italic> scripts, producing plots with the fitted functions proposed by Tettelin (<xref ref-type="bibr" rid="B66">Tettelin et al., 2005</xref>).</p>
</sec>
<sec id="S2.SS5">
<title>Functional pan-genome analysis and pan-genome wide association study of <italic>Mtb</italic> lineage-4</title>
<p>The data of the pan-genome matrix (PGM) and the GeneBank annotation files were used as input in BPGA v1.3 software (<xref ref-type="bibr" rid="B13">Chaudhari et al., 2016</xref>). Identifiers with the best hits were assigned from the KEGG (Kyoto Encyclopedia of Genes and Genomes) and COG (Clusters of Orthologous Groups of proteins) reference databases (<xref ref-type="bibr" rid="B43">Kanehisa and Goto, 2000</xref>; <xref ref-type="bibr" rid="B65">Tatusov et al., 2000</xref>), for which USEARCH&#x2019;s ublast (<xref ref-type="bibr" rid="B30">Edgar, 2010</xref>) function was used, with an <italic>E</italic>-value cutoff of 10E-5, databases and tools were used by default in BPGA V1.3. This analysis allowed us to annotate metabolic pathways and the functional categories of COGs within the dispensable genome, the core genome, and the unique genes that make up the pan-genome.</p>
<p>An association study was conducted based on patterns of gene presence/absence in the pan-genome. The polymorphisms considered for analysis included changes caused by the presence of variants such as those containing insertions or deletions, correlating them with the characteristics of high or low prevalence of <italic>Mtb</italic> strains. The PGM containing the variants and a matrix with the prevalence trait were used such as input for Scoary v1.6.16 (<xref ref-type="bibr" rid="B8">Brynildsrud et al., 2016</xref>). The observed presence/absence was correlated with the prevalence by evaluating its significance through Fischer&#x2019;s exact test. A list of variants with a <italic>P</italic>-value &#x003C; 0.05 was generated. To avoid false results as significant and avoid the probability of the family wise error rate (FWER), an adjustment of the <italic>P</italic>-value with tests of Bonferroni&#x2019;s and Benjamini&#x2013;Hochberg correction was performed.</p>
<p>The insertions, deletions or genes previously identified and significantly associated with high and low prevalence of <italic>Mtb</italic> strains were individually verified by multiple genome alignment with Mauve v2.4.0 (<xref ref-type="bibr" rid="B25">Darling et al., 2004</xref>). This software can also generate a file in tabular format with the SNPs and its coordinates in each genome. This can be used to identify differential SNPs between high and low prevalence strains. Gaps in the alignments were removed, and minor nucleotide frequency in each position computed with a script available at <ext-link ext-link-type="uri" xlink:href="https://github.com/ualonso85/Thesis-pipeline/blob/master/freqs.sh">https://github.com/ualonso85/Thesis-pipeline/blob/master/freqs.sh</ext-link>.</p>
<p>To assess the accuracy of our sequencing and bioinformatic approaches, we amplified the sequence region of the variants associated with the high or low prevalence in 18 <italic>Mtb</italic> strains randomly selected. The PCR products were sequenced on an ABI 3730 sequencer (Applied Biosystems). There was 100% identity between all sequenced regions both by Illumina HiSeq 2500 and ABI 3730.</p>
</sec>
<sec id="S2.SS6">
<title>Phylogenomic analysis</title>
<p>The PGM generated from the consensus of orthologs groups with the COG and OMCL algorithms was used for the phylogenomic reconstruction of the 47 genomes. The phylogenetic trees were built with the parsimony method from discrete characters (present/absence genes) using the PARS v3.69 software include in PHYLIP suite. In addition, a maximum likelihood method was used for phylogenomic analysis with the GTR substitution model and ultrafast bootstrap of 1000 replicates (&#x2013;bb 1000 -alrt 1000) using IQ-TREE v1.6.12 (<xref ref-type="bibr" rid="B52">Nguyen et al., 2015</xref>). The signal of each genetic tree was calculated based on the mean values of branch support with the SH-aLRT and UFBoot tests. The grouping of clades in the phylogenomic tree was verified with 26 <italic>Mtb</italic> genomes of different lineages (L2, L3, L4, L6) with known tree topology downloaded from the NCBI database.</p>
<p>Phylogenomic reconstruction with SNPs was performed by converting the eXtended Multi-Fasta (XMFA) alignment file from Mauve to fasta format with a script available at <ext-link ext-link-type="uri" xlink:href="https://github.com/eead-csic-compbio/eead-csic-compbio.github.io/blob/master/scripts/xmfa2fasta.pl">https://github.com/eead-csic-compbio/eead-csic-compbio.github.io/blob/master/scripts/xmfa2fasta.pl</ext-link>. Trimal software v1.2rev59 (<xref ref-type="bibr" rid="B11">Capella-Gutierrez et al., 2009</xref>) was used to removed gaps leaving only the nucleotide positions that were present in all <italic>Mtb</italic> genomes (core). The concatenated sequences from each genome that contained the SNPs were used to build the phylogeny using the maximum likelihood method with the same parameters in IQ-TREE described above but using TVM (transversional model) such as nucleotide substitution model. A strain of <italic>Mycobacterium canettii</italic> 140010059 Accession <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="NC_015848">NC_015848</ext-link> was included as root.</p>
</sec>
<sec id="S2.SS7">
<title>Data availability</title>
<p>The data generated during the current study are available online at National Center for Biotechnology Information under Bioproject accession <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="PRJNA838941">PRJNA838941</ext-link>.</p>
</sec>
<sec id="S2.SS8">
<title>Ethics statement</title>
<p>This study was approved and carried out in accordance with guidelines and regulations by the Ethics Review Committee of Coporaci&#x00F3;n para Investigaciones Biol&#x00F3;gicas.</p>
</sec>
</sec>
<sec id="S3" sec-type="results">
<title>Results</title>
<sec id="S3.SS1">
<title>Prevalence of <italic>Mtb</italic> lineage- 4 in NE area of Medell&#x00ED;n</title>
<p>In the Medell&#x00ED;n city, 135 <italic>Mtb</italic> isolates were recovered from Popular, Santa Cruz, Manrique, and Aranjuez communes, from 2005 to 2006. The spoligotype of these isolates were represented by patterns of presence or absence of some of the 43 spacers in the Direct Repeat (DR) locus. Sublineage classification was performed by assessing the SIT code retrieved from the SPOLDB4 database, as well as with 24 MIRU-VNTR analyzed online with MIRU-VNTR <italic>plus</italic> tool (<xref ref-type="supplementary-material" rid="TS1">Supplementary Table 1</xref>). In total 99 (73.3%) of the 135 isolates were classified as high prevalence due to the high frequency of the Haarlem1 SIT62 and LAM9 SIT42 sublineages, with a relative frequency of 0.37 and 0.36, respectively. The remaining 26.7% corresponded to 34 isolates of different sublineages with low frequency (&#x003C;0.10) in the same geographical area and during the same period. Simple random sampling allowed the selection of 27 high-prevalence isolates and 20 low-prevalence isolates that represented the largest number of genotyped sublineages (<xref ref-type="supplementary-material" rid="TS2">Supplementary Table 2</xref>).</p>
</sec>
<sec id="S3.SS2">
<title><italic>M. tuberculosis</italic> isolates and general genomic features</title>
<p>Once the 47 <italic>Mtb</italic> isolates were defined according to the high or low prevalence, each isolate was sequenced. A total of 156,746,390 reads of the isolates were obtained after filtering by quality parameters, with an average of 3,335,030 reads per sample. The average genome size of the 47 isolates was 4.348 Mb, G + C 65.5%, genome fraction 98.6%, with coverage depth of 102X (ranging from 30X in UT331 to 198X in UT86), and N50 was 114,855 (ranging from 78,270 in UT414 to 166,979 in UT86) with 105 contigs on average (ranging from 66 in UT105 to 140 in UT487) (<xref ref-type="supplementary-material" rid="TS3">Supplementary Table 3</xref>). Where possible, assembly errors were corrected by remapping reads, which considers alignment differences and reads coverage for each genome. Isolates UT487 and UT30 were taken as an example to show the most identified types of assembly errors in the data set (<xref ref-type="supplementary-material" rid="TS4">Supplementary Table 4</xref>).</p>
<p>Regarding gene model annotation, the UT53 genome presented the lowest number with 4,024 CDS, and UT08 presented the highest number with 4,087 CDS. An average of 2,880 (71%) CDS corresponded to proteins where function can potentially be assigned by homology; furthermore, 1,122 hypothetical proteins (28%), and 52.9 tRNAs (1%) were annotated in the 47 genomes (<xref ref-type="supplementary-material" rid="FS1">Supplementary Figure 1</xref>).</p>
</sec>
<sec id="S3.SS3">
<title>Pan-genome and core-genome of <italic>Mycobacterium tuberculosis</italic> lineage-4</title>
<p>Comparative analysis allowed us to determine the complete repertoire of genes of the 47 <italic>Mtb</italic> lineage-4 genomes, that together make up the pan-genome. The core-genome was composed of gene clusters shared among all isolates representing 73.5% of the pan-genome resulting from the intersection of the COG, BDBH and OMCL algorithms (see section &#x201C;Materials and methods&#x201D; and <xref ref-type="supplementary-material" rid="FS2">Supplementary Figure 2</xref>). In addition to the core subset, the accessory genome containing the dispensable genes and the strain-specific genes together added a total of 4,846 gene clusters (<xref ref-type="supplementary-material" rid="FS3">Supplementary Figure 3</xref>).</p>
<p>The complete repertoire of 4,846 gene clusters of the pan-genome was classified into four occupancy classes, as summarized in <xref ref-type="fig" rid="F1">Figure 1</xref> (<xref ref-type="bibr" rid="B18">Contreras-Moreira and Vinuesa, 2013</xref>):</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption><p><italic>Mycobacterium tuberculosis</italic> (<italic>Mtb</italic>) pan-genome area of lineage-4. Global composition of gene clusters divided into four compartments.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-13-1076797-g001.tif"/>
</fig>
<list list-type="simple">
<list-item><label>1)</label><p>Core: genes conserved in all isolates studied.</p></list-item>
<list-item><label>2)</label><p>Soft core: genes found in 45 (95%) or more of the isolates, thus including core genes.</p></list-item>
<list-item><label>3)</label><p>Shell: Gene clusters conserved in a variable number (3 to 44) isolates.</p></list-item>
<list-item><label>4)</label><p>Cloud: rare or unique genes present in two or fewer isolates.</p></list-item>
</list>
<p>Accordingly, the core compartment has the highest occupancy, and includes 3,566 gene clusters. The soft-core class amounts to a total of 3,650 (75%) gene clusters, which includes the strict core. The shell class contained 537 gene clusters, representing 11% of the total. Finally, cloud clusters represented 13% of the pan-genome (<xref ref-type="fig" rid="F2">Figure 2</xref>).</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption><p>Occupancy of genomes in gene clusters. Every compartment or class is depicted in a different color. The <italic>X</italic>-axis shows the occupancy, which is the number of genomes that are contained in a given number of clusters of genes.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-13-1076797-g002.tif"/>
</fig>
<p>On average, the accessory genome consisted of 477 genes with a standard deviation (SD) of 16, with 12 strain-specific genes (SD = 5.5). The average number of CDS in each isolate annotated was 4,055 with an SD of 15 (<xref ref-type="fig" rid="F3">Figure 3</xref>). Moreover, to assess whether the number of sequenced genomes was sufficient to describe the content of the core and accessory genome of the <italic>Mtb</italic> lineage-4, a genome composition analysis was performed. Both in the core and in the accessory genome, a change was observed as a function of its size each time a genome was added in random order until the 47 genomes were completed. The evolution of core and accessory genome size was analyzed in terms of exponential decay and growth models. The fitted exponential decay suggested that the number of orthologous gene clusters in the core tends to plateau near 3,700 gene clusters (<xref ref-type="fig" rid="F4">Figure 4A</xref>). Conversely, the exponential growth model fails to plateau in our simulation and continued to grow above 4,900 gene clusters (<xref ref-type="fig" rid="F4">Figure 4B</xref>). The number of non-redundant genes found in <italic>Mtb</italic> isolates would seem to increase by approximately 15 genes each time a new genome is added.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption><p>Pan-genome flower. It shows all <italic>Mtb</italic> lineage-4 isolates that make up the pan-genome. In the center the number of core genes is observed, the second clear circle shows the accessory genes, and the petals show the number of specific genes of each isolate in the 47 genomes. The numbers below each isolate denote the total number of related CDSs.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-13-1076797-g003.tif"/>
</fig>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption><p>Composition analysis of the core and the pan-genome of <italic>Mtb</italic> L4 from 47 genomes. <bold>(A)</bold> Each point on the <italic>Y</italic>-axis indicates the number of gene clusters after adding a new genome in randomized simulations. The line red indicates the exponential decay as a function of the average values of the clusters each time a genome was added to the analysis. <bold>(B)</bold> Pan-genome growth simulation by counting new genes added by the last genome sampled. Note that sequences matching a previously seen gene coverage &#x2265;20% will be considered homologous and thus won&#x2019;t be considered new. An open pan genome model is observed.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-13-1076797-g004.tif"/>
</fig>
</sec>
<sec id="S3.SS4">
<title>Functional pan-genome analysis</title>
<p>A functional profile of the core-genome was obtained by annotating COG categories. The [R] general function was the most frequently observed (16%), followed by [I] Lipid transport and metabolism (8%), [S] Function unknown (8%), [Q] Secondary metabolites biosynthesis, transport and catabolism (7%), [K] Transcription (7%), [C] Energy production and conversion (7%), [E] Amino acid transport and metabolism (7%), and [L] Replication, recombination and repair (5%). Other functional categories were found related to processes of the central metabolism of mycobacteria such as translation, transport biogenesis and metabolism of carbohydrates and coenzymes, among others (<xref ref-type="supplementary-material" rid="FS4">Supplementary Figure 4</xref>).</p>
<p>Metabolic pathways were identified in <italic>Mtb</italic> pan-genome using the KEGG databases. The highest hierarchical levels found for core gene clusters were related to metabolism (75%); many of them with the metabolism of carbohydrates (18%), lipids (8%), amino acids (16%), cofactors and vitamins (8%), and the remaining 25% associated with replication, repair, translation and signal transduction among others (<xref ref-type="fig" rid="F5">Figures 5</xref>, <xref ref-type="fig" rid="F6">6</xref>). The general functions of the dispensable genome in the KEGG pathways were related to environmental information processing (30%), primary paths of cellular community (19%), signal transduction (19%) and signaling molecules and interaction (18%). Among general pathways, genes were found to be associated with organismal systems (29%) and human diseases (19%). The majority of those were assigned to primary functions such as the immune system (16%) and infectious diseases (19%) (<xref ref-type="fig" rid="F5">Figures 5</xref>, <xref ref-type="fig" rid="F6">6</xref>). For the case of strain-specific genes, cellular processes were the general function with the highest assignment (38%), followed by organismal systems (25%), environmental information processing (25%), and human diseases (13%) (<xref ref-type="fig" rid="F5">Figure 5</xref>).</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption><p>General KEGG pathways. It shows the differences in percentage of the functional annotations at the highest hierarchical level between the genes highly conserved of the core (red), genes moderately conserved of the dispensable genome (green) and genes exclusive to each isolate or unique (gray).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-13-1076797-g005.tif"/>
</fig>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption><p>Primary KEGG pathways. It shows the differences in percentage of the functional annotations among the genes highly conserved of the core (red), where the highest number of assignments for each category is observed. Genes moderately conserved of the dispensable genome (green) are mainly related to adaptation to the environment-host and genes exclusive to each isolate or unique (gray).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-13-1076797-g006.tif"/>
</fig>
</sec>
<sec id="S3.SS5">
<title>Pan-genome association study of <italic>Mtb</italic> lineage-4</title>
<p>The prevalence phenotype was correlated with the presence or absence of genes in the pan-genome. Variants such as insertions and deletions were identified in gene regions related to their presence or absence. Each variant was initially assigned the null hypothesis of no association with prevalence. A total of 115 variants in the 47 genomes of <italic>Mtb</italic> lineage-4 were found to be associated to the trait of high or low prevalence (<italic>p</italic>-value &#x003C; 0.05). Nevertheless, due to the high number of null hypotheses evaluated and the multiple factors analyzed simultaneously in the pan-GWAS, <italic>p</italic>-values were adjusted with the Bonferroni&#x2019;s method, as implemented in Scoary. With this correction in place, four genes were significantly associated (<italic>mmpl12</italic>, <italic>PPE29</italic>, <italic>Rv1419</italic>, <italic>Rv1762c</italic>). Due to the conservative nature of the Bonferroni&#x2019;s multi-testing correction, the Benjamini-Hochberg correction (FDR) was also considered. In this case, the list of associated genes comprised seven additional genes: <italic>Rv3371</italic>, <italic>Rv2735c</italic>, <italic>scoA</italic>, <italic>mhpE</italic>, <italic>PE-PGRS52</italic>, <italic>lppB</italic>, and <italic>gabD2</italic> (<xref ref-type="table" rid="T2">Table 2</xref>). The name of each isolate, its genotype, as well as the number of isolates associated with high or low prevalence are shown in <xref ref-type="supplementary-material" rid="TS5">Supplementary Table 5</xref>.</p>
<table-wrap position="float" id="T2">
<label>TABLE 2</label>
<caption><p>List of <italic>Mtb</italic> lineage-4 genes identified in pan-GWAS analysis.</p></caption>
<table cellspacing="5" cellpadding="5" frame="box" rules="all">
<thead>
<tr>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Gene</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Uniprot code</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Functional annotation</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;"><italic>P</italic>-value</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Bonferroni&#x2019;s-adjusted <italic>P</italic>-value</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Benjamini-H. adjusted <italic>P</italic>-value</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Variant association</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><italic>mmpL12</italic></td>
<td valign="top" align="left">A0A0T9BXN4</td>
<td valign="top" align="left">Probable transport of membrane. Responds to host immune response.</td>
<td valign="top" align="center">1.13e-07</td>
<td valign="top" align="center">0.00015</td>
<td valign="top" align="center">2.90e-05</td>
<td valign="top" align="left">Low prevalence</td>
</tr>
<tr>
<td valign="top" align="left"><italic>PPE29</italic></td>
<td valign="top" align="left">P9WI09</td>
<td valign="top" align="left">May be required for invasion of host endothelial cells</td>
<td valign="top" align="center">1.13e-07</td>
<td valign="top" align="center">0.00015</td>
<td valign="top" align="center">2.90e-05</td>
<td valign="top" align="left">Low prevalence</td>
</tr>
<tr>
<td valign="top" align="left"><italic>Rv1419</italic></td>
<td valign="top" align="left">P9WLX8</td>
<td valign="top" align="left">Uncharacterized protein/GO carbohydrate binding</td>
<td valign="top" align="center">5.51e-07</td>
<td valign="top" align="center">0.00071</td>
<td valign="top" align="center">7.84e-05</td>
<td valign="top" align="left">Low prevalence</td>
</tr>
<tr>
<td valign="top" align="left"><italic>Rv1762c</italic></td>
<td valign="top" align="left">A0A1K6RHD7</td>
<td valign="top" align="left">Domain of uncharacterized function (DUF74)/GO transmembrane</td>
<td valign="top" align="center">5.51e-07</td>
<td valign="top" align="center">0.00071</td>
<td valign="top" align="center">7.84e-05</td>
<td valign="top" align="left">Low prevalence</td>
</tr>
<tr>
<td valign="top" align="left"><italic>Rv3371</italic></td>
<td valign="top" align="left">P9WKA9</td>
<td valign="top" align="left">Putative diacylglycerol O-acyltransferase</td>
<td valign="top" align="center">0.00014</td>
<td valign="top" align="center">0.18124</td>
<td valign="top" align="center">0.0062</td>
<td valign="top" align="left">High prevalence</td>
</tr>
<tr>
<td valign="top" align="left"><italic>Rv2735c</italic></td>
<td valign="top" align="left">R4ML16</td>
<td valign="top" align="left">hypothetical protein uncharacterized</td>
<td valign="top" align="center">0.00014</td>
<td valign="top" align="center">0.18124</td>
<td valign="top" align="center">0.0062</td>
<td valign="top" align="left">High prevalence</td>
</tr>
<tr>
<td valign="top" align="left"><italic>scoA</italic></td>
<td valign="top" align="left">P9WPW5</td>
<td valign="top" align="left">Probable succinyl-CoA:3-ketoacid coenzyme A transferase subunit A</td>
<td valign="top" align="center">0.00014</td>
<td valign="top" align="center">0.18124</td>
<td valign="top" align="center">0.0062</td>
<td valign="top" align="left">High prevalence</td>
</tr>
<tr>
<td valign="top" align="left"><italic>mhpE</italic></td>
<td valign="top" align="left">R4MB83</td>
<td valign="top" align="left">4-hydroxy-2-oxovalerate/4-hydroxy-2-oxopentanoic acid aldolase</td>
<td valign="top" align="center">0.00014</td>
<td valign="top" align="center">0.18124</td>
<td valign="top" align="center">0.0062</td>
<td valign="top" align="left">High prevalence</td>
</tr>
<tr>
<td valign="top" align="left"><italic>PE-PGRS42</italic></td>
<td valign="top" align="left">I6XEF1</td>
<td valign="top" align="left">Family PE-PGRS proteins</td>
<td valign="top" align="center">0.00014</td>
<td valign="top" align="center">0.18124</td>
<td valign="top" align="center">0.0062</td>
<td valign="top" align="left">High prevalence</td>
</tr>
<tr>
<td valign="top" align="left"><italic>lppB</italic></td>
<td valign="top" align="left">P9WK79</td>
<td valign="top" align="left">Putative lipoprotein LppB, extracellular region/GO cell membrane</td>
<td valign="top" align="center">0.00014</td>
<td valign="top" align="center">0.18124</td>
<td valign="top" align="center">0.0062</td>
<td valign="top" align="left">High prevalence</td>
</tr>
<tr>
<td valign="top" align="left"><italic>gabD2</italic></td>
<td valign="top" align="left">P9WNX7</td>
<td valign="top" align="left">Putative succinate-semialdehyde dehydrogenase [NADP(+)] 2</td>
<td valign="top" align="center">0.00014</td>
<td valign="top" align="center">0.18124</td>
<td valign="top" align="center">0.0062</td>
<td valign="top" align="left">High prevalence</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="S3.SS6">
<title>Identification of indels and SNPs associated with the prevalence</title>
<p>Multiple alignments of the 47 genomic isolates plus the <italic>Mtb</italic> reference genome H37Rv allowed us to verify the differential genetic variants between the high and low prevalence isolates. For instance, we discovered that the <italic>mmpL12</italic> gene locus, a single CDS associated with high-prevalence, turned out to be split into two neighbor CDS by the IS6110 insertion in low-prevalence genomes. This type of insertions generally results in inactivation corresponding gene (see <xref ref-type="supplementary-material" rid="FS5">Supplementary Figure 5</xref>; <xref ref-type="bibr" rid="B59">Reyes et al., 2012</xref>). Overall, 5 variants significantly associated to prevalence traits corresponded to insertions: in the <italic>mmpL12</italic> gene between position 1642_3000ins, <italic>PPE29</italic> between position 639_641insG, <italic>Rv2735c</italic> position 363_364insGT, <italic>ScoA</italic> position 476_478insG, <italic>gabD2</italic> between position 610_644ins. Moreover, 3 deletions were also identified: in the <italic>Rv1419</italic> gene position 199delA, the <italic>Rv3371</italic> gene position 376delA and the <italic>PEPGRS42</italic> gene between positions 1382_1417del (see <xref ref-type="table" rid="T3">Table 3</xref>).</p>
<table-wrap position="float" id="T3">
<label>TABLE 3</label>
<caption><p>Genetic variants associated with high or low prevalence of clinical isolates of <italic>Mtb</italic> lineage-4 in the North-Eastern zone of Medell&#x00ED;n.</p></caption>
<table cellspacing="5" cellpadding="5" frame="box" rules="all">
<thead>
<tr>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">No.</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Gene</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Size (pb)</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Locus (pb)</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Variant</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Genomes number high prev. <italic>N</italic> = 27</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Genomes number low prev. <italic>N</italic> = 20</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Benjamini-H. adjusted <italic>P</italic>-value</td>
<td valign="top" align="center" style="color:#ffffff;background-color: #7f8080;">Variant association</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1</td>
<td valign="top" align="center"><italic>mmpL12</italic></td>
<td valign="top" align="center">3.341</td>
<td valign="top" align="center">1.714.172</td>
<td valign="top" align="center">1642_3000ins</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">14</td>
<td valign="top" align="center">2.90e-05</td>
<td valign="top" align="center">Low prevalence</td>
</tr>
<tr>
<td valign="top" align="left">2</td>
<td valign="top" align="center"><italic>PPE29</italic></td>
<td valign="top" align="center">1.272</td>
<td valign="top" align="center">2.042.001</td>
<td valign="top" align="center">639_641insG</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">14</td>
<td valign="top" align="center">2.90e-05</td>
<td valign="top" align="center">Low prevalence</td>
</tr>
<tr>
<td valign="top" align="left">3</td>
<td valign="top" align="center"><italic>Rv1419</italic></td>
<td valign="top" align="center">474</td>
<td valign="top" align="center">1.593.505</td>
<td valign="top" align="center">199delA</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">14</td>
<td valign="top" align="center">7.84e-05</td>
<td valign="top" align="center">Low prevalence</td>
</tr>
<tr>
<td valign="top" align="left">4</td>
<td valign="top" align="center"><italic>Rv1762c</italic></td>
<td valign="top" align="center">789</td>
<td valign="top" align="center">1.995.054</td>
<td valign="top" align="center">538C &#x003E; T</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">13</td>
<td valign="top" align="center">7.84e-05</td>
<td valign="top" align="center">Low prevalence</td>
</tr>
<tr>
<td valign="top" align="left">5</td>
<td valign="top" align="center"><italic>Rv3371</italic></td>
<td valign="top" align="center">1.341</td>
<td valign="top" align="center">3.784.932</td>
<td valign="top" align="center">374T &#x003E; A, 376delA</td>
<td valign="top" align="center">16</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">0.0062</td>
<td valign="top" align="center">High prevalence</td>
</tr>
<tr>
<td valign="top" align="left">6</td>
<td valign="top" align="center"><italic>Rv2735c</italic></td>
<td valign="top" align="center">993</td>
<td valign="top" align="center">3.047.560</td>
<td valign="top" align="center">363_364insGT</td>
<td valign="top" align="center">16</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">0.0062</td>
<td valign="top" align="center">High prevalence</td>
</tr>
<tr>
<td valign="top" align="left">7</td>
<td valign="top" align="center"><italic>scoA</italic></td>
<td valign="top" align="center">747</td>
<td valign="top" align="center">2.819.124</td>
<td valign="top" align="center">476_478insG</td>
<td valign="top" align="center">16</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">0.0062</td>
<td valign="top" align="center">High prevalence</td>
</tr>
<tr>
<td valign="top" align="left">8</td>
<td valign="top" align="center"><italic>mhpE</italic></td>
<td valign="top" align="center">1.011</td>
<td valign="top" align="center">3.886.073</td>
<td valign="top" align="center">169G &#x003E; T</td>
<td valign="top" align="center">16</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">0.0062</td>
<td valign="top" align="center">High prevalence</td>
</tr>
<tr>
<td valign="top" align="left">9</td>
<td valign="top" align="center"><italic>PE-PGRS42</italic></td>
<td valign="top" align="center">2.085</td>
<td valign="top" align="center">2.795.301</td>
<td valign="top" align="center">1382_1417del</td>
<td valign="top" align="center">16</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">0.0062</td>
<td valign="top" align="center">High prevalence</td>
</tr>
<tr>
<td valign="top" align="left">10</td>
<td valign="top" align="center"><italic>lppB</italic></td>
<td valign="top" align="center">561</td>
<td valign="top" align="center">2.867.124</td>
<td valign="top" align="center">362C &#x003E; A</td>
<td valign="top" align="center">16</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">0.0062</td>
<td valign="top" align="center">High prevalence</td>
</tr>
<tr>
<td valign="top" align="left">11</td>
<td valign="top" align="center"><italic>gabD2</italic></td>
<td valign="top" align="center">1.557</td>
<td valign="top" align="center">1.957.577</td>
<td valign="top" align="center">610_644ins</td>
<td valign="top" align="center">16</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">0.0062</td>
<td valign="top" align="center">High prevalence</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Single nucleotide polymorphisms (SNP) identified in the multiple alignment of the 47 <italic>Mtb</italic> genomes were obtained taking as reference the H37Rv genome. This was done after obtaining the coordinates of the SNPs in each of the genome alignments. Significantly associated SNPs (<italic>p</italic>-value &#x003C; 0.05 Benjamini-H correction) were identified in several genes: <italic>Rv1419</italic> positions 122G &#x003E; A, 362C &#x003E; A, <italic>Rv1762</italic> position 538C &#x003E; T, <italic>Rv3371</italic> positions; 374T &#x003E; A, <italic>mhpE</italic> position 169G &#x003E; T, and in the <italic>lppB</italic> gene position 362C &#x003E; A (see <xref ref-type="table" rid="T3">Table 3</xref>).</p>
<p>All in all, a total of 12 genomic variants were identified combining indels and SNPs. These are distributed in 11 CDS. Each of these genetic loci were amplified by PCR to confirm our findings. Eleven of the 12 genetic loci were successfully amplified with sizes ranging from 278 bp for the <italic>RV2735c</italic> gene to 1650 bp for the <italic>mmpL12</italic> gene (<xref ref-type="supplementary-material" rid="TS6">Supplementary Table 6</xref>); the <italic>PE-PGRS42</italic> gene did not amplify despite using two different pairs of primers. PCR products sequenced on an ABI 3730 had 100% identity among all sequenced regions when compared to the Illumina HiSeq 2500 sequence.</p>
</sec>
<sec id="S3.SS7">
<title>Phylogenomic analysis of the <italic>Mtb</italic> lineage-4</title>
<p>The analysis of the discrete characters of the PGM allowed the phylogenomic reconstruction of the isolates of <italic>Mtb</italic>, capturing the phylogeny implicit in the matrix. The tree showed high branch support values. However, some branches in each of the three main clades showed support values lower than 70% (see <xref ref-type="fig" rid="F7">Figure 7</xref>). The tree was rooted with the <italic>Mycobacterium canetti</italic> species that was used as an &#x201C;outgroup&#x201D; genome. The topologies suggest three main clades that represent the evolutionary relationships of the 47 <italic>Mtb</italic> genomes based on their gene content (<xref ref-type="fig" rid="F7">Figure 7</xref>).</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption><p>Phylogeny of the <italic>Mtb</italic> L4 pan-genome by maximum likelihood. IQ-TREE was used to estimate the phylogenetic tree from the consensus of the 4,846 gene clusters produced by both the COG and OMCL algorithms. The node supports after 1000 bootstraps is shown on the branches. The outgroup is <italic>M. canetti</italic>. Three major clades were observed. Lilac; all the branches of the isolates, except UT277, coincide with the Haarlem1-SIT62 sublineage considered to be of high prevalence. Purple; the branches correspond mainly to Haarlem1-SIT45 and Haarlem3-SIT50 isolates, and less frequently to Haarlem1 and Haarlem3 with variable SIT. Green; a clade was observed mixed with branches of isolates considered to be of high prevalence, mostly LAM9 SIT42, and four branches of low prevalence belonging to the LAM sublineages. The red branches correspond to isolates with high prevalence and the pink branches correspond to all isolates with low prevalence in the North-Eastern zone of Medell&#x00ED;n.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-13-1076797-g007.tif"/>
</fig>
<p>As the phylogenomic reconstruction was conducted only with genomes of isolates that belonged to the <italic>Mtb</italic> lineage-4, we wanted to confirm whether the topology of the branches was conserved when adding to the analysis 26 complete <italic>Mtb</italic> genomes from different lineage [11 lineage-4 (Euro-American), 8 lineage-2 (East-Asian), 4 Lineage-3 (East-African-Indian) y 3 Lineage-6 (West-Africa)] previously downloaded from NCBI database. A total of 6,160 clusters of protein group were obtained and computed from the PGM. The topology obtained was consistent with the trees obtained by maximum likelihood of lineage-4, in addition to coinciding with the topology of the trees established as a reference for the grouping of these lineages (<xref ref-type="bibr" rid="B51">Ngabonziza et al., 2020</xref>), coinciding with the most accepted phylogenetic history for this species. This allowed us to observe that the topology obtained based on the genetic content was adequate and can be used as a reference in future comparisons (<xref ref-type="supplementary-material" rid="FS6">Supplementary Figure 6</xref>).</p>
<p>From 34,999 SNPs present in the core genome, the reconstruction of phylogenetic trees was carried out with CIPT <italic>Mycobacterium canetti</italic> strain, which was used as root. The support values were greater than 95% for most of the branches of the tree, however, some of the branches presented support values lower than 70% (<xref ref-type="supplementary-material" rid="FS7">Supplementary Figure 7</xref>). The topologies showed the same three clades that represented the evolutionary relationships of the 47 <italic>Mtb</italic> genomes in the pan-genome phylogeny described above, retaining a similar distribution of isolates in each branch of the tree (<xref ref-type="fig" rid="F8">Figure 8</xref>). Both approaches also match the classification of the sublineages reported by molecular genotyping methods (Spoligotyping and 24-loci MIRU-VNTR).</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption><p>Topology comparison trees of Mtb L4 were constructed by maximum likelihood using different molecular markers. IQ-TREE was used to estimate the phylogeny with node supports after a bootstrap of 1000 replicas. <bold>(A)</bold> Consensus of gene clusters of the pan-genome (4,846 genes). <bold>(B)</bold> Concatenation of 34,999 SNPs of the core genome. In both, three main clades were observed, with very similar topologies and distribution of the isolates in each of the branches, despite having used different markers as an approximation.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-13-1076797-g008.tif"/>
</fig>
</sec>
</sec>
<sec id="S4" sec-type="discussion">
<title>Discussion</title>
<sec id="S4.SS1">
<title><italic>Mtb</italic> pan-genome and their accessory genes</title>
<p>This is the first pan-genomic analysis of <italic>Mtb</italic> lineage-4 (Euro-American) carried out in Colombia, characterized by being the most dominant lineage in the country. The pan-genome estimation showed that the number of genes with new characteristics continues to increase regardless of the number of <italic>Mtb</italic> genomes added to the analysis. Experimental results and predictions with mathematical models have shown that in some species new genes are discovered even after the sequencing of hundreds of genomes for a particular species (<xref ref-type="bibr" rid="B47">Medini et al., 2005</xref>; <xref ref-type="bibr" rid="B60">Rouli et al., 2015</xref>). The size of the pan-genome is related to the events of loss or gain of genes. If the niche changes, some functions might be used less and eventually lost. Unlike bacteria that are found in diverse environments, which frequently undergo gene gain events, bacteria that have a lifestyle restricted to a specific niche possess a smaller and specialized genome, therefore the pan-genome tends to be closed or finite (<xref ref-type="bibr" rid="B47">Medini et al., 2005</xref>; <xref ref-type="bibr" rid="B60">Rouli et al., 2015</xref>). However, our results showed that <italic>Mtb</italic> lineage-4 in the NE area of Medell&#x00ED;n has an open pan-genome, this may be partly related to the low number of genomes used in this study. We found a great genetic variability despite this species being restricted to a specific niche with a more conserved genome (<xref ref-type="bibr" rid="B47">Medini et al., 2005</xref>; <xref ref-type="bibr" rid="B76">Yang et al., 2018</xref>; <xref ref-type="bibr" rid="B73">Woodman et al., 2019</xref>). Therefore, intra-species diversity at the sublineage level was broader than expected. Recent composition analysis studies of the <italic>Mtb</italic> pan-genome have shown that by increasing the number of genomes above 120, the curve tends to flatten considering the pan-genome of this species is almost closed. Nevertheless, it continues to present a significant number of variations (<xref ref-type="bibr" rid="B24">Dar et al., 2020</xref>; <xref ref-type="bibr" rid="B77">Zakham et al., 2021</xref>).</p>
<p>The pan-genome of <italic>Mtb</italic> lineage-4 was rather associated with the presence of smaller indels and SNPs that cause frameshifts or premature stop codons in genes increasing the genetic variability. Nevertheless, this variability is not as abundant as compared to other bacterial species where gene gain events are common due to recombination mechanisms (<xref ref-type="bibr" rid="B60">Rouli et al., 2015</xref>). These results suggest that despite the fact that <italic>Mtb</italic> does not have horizontal gene transfer, it has mechanisms for genetic variability such as transposable elements and a high frequency of mutations in a latency state (microevolution) (<xref ref-type="bibr" rid="B19">Coros et al., 2008</xref>; <xref ref-type="bibr" rid="B71">Vernikos et al., 2015</xref>).</p>
<p>Out of the identified 4,846 genes in the pan-genome 1,071 (26.5%) corresponded to hypothetical proteins representing similar proportion to the one reported previously in <italic>Mtb</italic> (<xref ref-type="bibr" rid="B17">Cole et al., 1998</xref>). A total of 1,196 genes were identified in the accessory genome; composed of dispensable genes located in the shell compartment and unique genes located in the cloud. Of the 537 genes in the shell, 99 coded for PE family genes including 22 PE-PGRS genes and 17 for PPE family genes. The PE/PPE family represents 10% of the genes encoding <italic>Mtb</italic> and its proteins present the greatest source of antigenic variability between different isolates, it is assumed that they are of critical importance in the general pathogenesis of TB (<xref ref-type="bibr" rid="B50">Mukhopadhyay and Balaji, 2011</xref>).</p>
<p>In the shell compartment, 237 genes encoded hypothetical proteins representing 44% of the genes in this compartment, and are responsible for much of the genetic variability that remains unknown (<xref ref-type="bibr" rid="B50">Mukhopadhyay and Balaji, 2011</xref>; <xref ref-type="bibr" rid="B71">Vernikos et al., 2015</xref>). These dispensable genes shared between certain groups of isolates contribute to diversity, and presumably play roles in complementary biochemical pathways not essential for growth. However, they may confer selective advantages and adaptation to different niches, new host colonization, and antibiotic resistance (<xref ref-type="bibr" rid="B71">Vernikos et al., 2015</xref>). For instance, the Beijing genotype of lineage-2 has been characterized by having a higher transmission frequency than other lineages, possibly due to shell genes, in addition to being associated with a higher drug resistance, which suggest a relatively high transmission fitness (<xref ref-type="bibr" rid="B40">Holt et al., 2018</xref>; <xref ref-type="bibr" rid="B54">Nieto Ramirez et al., 2020</xref>).</p>
<p>Of the 659 strain-specific genes located in the cloud compartment, 274 coded for hypothetical proteins, 224 of them belonging to the PE family of which 53 coded for proteins of the PE-PGRS family, representing together 83.6% of the genes of the cloud. Genes in the PE/PE-PGRS family are known to have multiple copies of polymorphic repetitive sequences; however, this high variability added to the lower quality contigs and scaffolds assembled for these regions, could cause this overestimation (<xref ref-type="bibr" rid="B50">Mukhopadhyay and Balaji, 2011</xref>; <xref ref-type="bibr" rid="B56">Periwal et al., 2015</xref>).</p>
</sec>
<sec id="S4.SS2">
<title>Functional analysis of the accessory <italic>Mtb</italic> genome</title>
<p>Among the COG categories assigned to the 537 genes that were part of the dispensable genome, four had the highest proportion of genes related: Q] Secondary metabolites biosynthesis, transport and catabolism, followed by [D] Cell cycle control, cell division, chromosome partitioning, [U] Intracellular trafficking, secretion and vesicular transport and [S] Function unknown. Many of these genes are partially shared between <italic>Mtb</italic> isolates and may represent particular metabolic characteristics associated to adaptation to one human host or another. Most of the dispensable genes were found to be related to general KEGG pathways for Environmental Information Processing and Human Diseases that causes in the host (<xref ref-type="fig" rid="F5">Figure 5</xref> and <xref ref-type="supplementary-material" rid="FS4">Supplementary Figure 4</xref>). The presence or absence of these genes could be the result of adaptation processes by <italic>Mtb</italic> to the host or selection pressure exerted by the environment (<xref ref-type="bibr" rid="B47">Medini et al., 2005</xref>; <xref ref-type="bibr" rid="B71">Vernikos et al., 2015</xref>). The dispensable genes, despite being secondary genes, are important for adaptation, they code for important functions related to virulence and pathogenicity (<xref ref-type="bibr" rid="B18">Contreras-Moreira and Vinuesa, 2013</xref>). Differences in these genes make it possible to distinguish between lineages. Modern lineages such as lineages 2 and 4, have been characterized by being more virulent and transmitted more successfully within the global population than other lineages (<xref ref-type="bibr" rid="B20">Coscolla and Gagneux, 2014</xref>). Our data confirm that the high resolution of whole genome sequencing has not only allowed the differentiation between different lineages, but also between sublineages present in the same geographic population.</p>
<p>Among strain-specific genes, they were mostly enriched in COG categories related to [R] general function prediction and [U] Intracellular trafficking, secretion and vesicular transport, follow by [M] Cell wall/membrane/envelope biogenesis, [T] Signal transduction mechanisms and by [Q] Secondary metabolites biosynthesis, transport and catabolism. Among these functions, the transport and secretion of vesicles by <italic>Mtb</italic> through their membrane and cell wall has been described to have an important role in the host interaction (<xref ref-type="bibr" rid="B38">Gupta and Rodriguez, 2018</xref>). The release of extracellular vesicles allows the interaction of the cell with the environment. These vesicles in pathogenic bacteria play roles in cell-cell communication, immunomodulation, virulence, and cell survival (<xref ref-type="bibr" rid="B38">Gupta and Rodriguez, 2018</xref>). It is possible that some of these strain-specific genes in each isolate give it unique characteristics to interact individually with its host, allowing it to adapt to changing external conditions.</p>
</sec>
<sec id="S4.SS3">
<title>The <italic>Mtb</italic> core genome: Analysis and functional classification</title>
<p>The core represented the largest fraction of the pan-genome, including more than 70% of the genes identified. As expected, every time a new genome was added to the analysis, the size of the core genome decreased. Many of these genes are constitutive and are part of the central metabolism, therefore many of these genes can be considered essential for the growth, development, and reproduction of <italic>Mtb</italic> (<xref ref-type="bibr" rid="B47">Medini et al., 2005</xref>; <xref ref-type="bibr" rid="B66">Tettelin et al., 2005</xref>).</p>
<p>In the functional analysis, the most abundant category within the genomic core after the [R] general function prediction category was the [I] Lipid transport and metabolism category. The <italic>Mtb</italic> cell membrane is known to be rich in lipids and approximately 40% of the dry weight of its cell wall is composed of these (<xref ref-type="bibr" rid="B14">Chiaradia et al., 2017</xref>). A large part of its genome encodes for lipid biosynthesis and degradation. Many of these genes are involved in the cell membrane, which allows understanding the importance not only at the structural level, but it is also known that they are implicated in functions related to cell invasion, host immune system evasion, virulence, and slow growth (<xref ref-type="bibr" rid="B17">Cole et al., 1998</xref>; <xref ref-type="bibr" rid="B34">Forrellad et al., 2013</xref>; <xref ref-type="bibr" rid="B76">Yang et al., 2018</xref>).</p>
<p>The data confirmed that the core genes are primary genes that determine the generalities of <italic>Mtb</italic> isolates and their lineage-4 and contain the majority of genes essential for their survival (<xref ref-type="bibr" rid="B17">Cole et al., 1998</xref>; <xref ref-type="bibr" rid="B65">Tatusov et al., 2000</xref>). Like in <italic>Mtb</italic>, many of the genes identified in the core of bacterial species such as <italic>S. agalactiae</italic>, <italic>S. pyogenes</italic>, and <italic>E. coli</italic> are housekeeping genes, most of the genes participate in basic biology aspects and are responsible for its main phenotypic characteristics (<xref ref-type="bibr" rid="B47">Medini et al., 2005</xref>; <xref ref-type="bibr" rid="B71">Vernikos et al., 2015</xref>).</p>
</sec>
<sec id="S4.SS4">
<title>Phylogenetic analysis of the pan-genome and SNPs</title>
<p>Comparing the topology of the phylogenomic tree from PGM with the genotyping data generated by spoligotyping and 24-loci MIRU-VNTR, the 16 isolates previously genotyped as Haarlem1 SIT62 were part of the same clade, together with a Haarlem1 SIT3022, 14 isolates between Haarlem1 and Haarlem3 with different SIT were part of the second clade and 16 LAM isolates with different SIT made up the third clade (<xref ref-type="fig" rid="F7">Figure 7</xref>). The only isolate that did not coincide with genotyping was UT277 Haarlem1 SIT3022, because it was located in the same clade of Haarlem1 SIT62. As spoligotyping and 24-loci MIRU-VNTR tests have a lower level of resolution than whole-genome analysis, the information they provide at the clade and family level is less reliable (<xref ref-type="bibr" rid="B65">Tatusov et al., 2000</xref>). When confirming the topology of the branches by adding 26 genomes downloaded from the NCBI belonging to the lineages (2, 3, 4, and 6), a clear grouping was observed for each lineage that was consistent with previously reported studies of the evolutionary relationships of the lineages (2, 3, and 4) considered modern and that differ from the ancient lineages (1, 5, and 6) by the deletion of the TbD1 genomic region, which in our case was observed only in the isolates belonging to the ancient lineage 6 (<xref ref-type="supplementary-material" rid="FS6">Supplementary Figure 6</xref>; <xref ref-type="bibr" rid="B28">Donoghue, 2011</xref>; <xref ref-type="bibr" rid="B20">Coscolla and Gagneux, 2014</xref>; <xref ref-type="bibr" rid="B53">Niemann and Supply, 2014</xref>). The above allowed us to infer that the topology and branch lengths obtained as a function of the genetic content (presence-absence) was adequate and can be used as a reference in future comparisons.</p>
<p>Single nucleotide polymorphisms are individual positions of orthologous nucleotides that vary across genomes (<xref ref-type="bibr" rid="B46">Leach&#x00E9; and Oaks, 2017</xref>). The SNPs present in the core genome were used to perform phylogenetic reconstruction of <italic>Mtb</italic> lineage-4 isolates. Although phylogenetic trees were reconstructed with different molecular markers for both genes and SNPs, the tree topology was preserved in the major clades and for most of the branches with each isolation (<xref ref-type="supplementary-material" rid="FS7">Supplementary Figure 7</xref>). The topology of the trees was consistent for the phylogenetic inference of both the pan-genome genes and the SNPs present in the genomic core of the isolates, directly correlated with the sublineages of <italic>Mtb</italic> lineage-4 genotyped by spoligotyping and 24 MIRU- VNTR.</p>
<p>Although it was expected that isolates would be grouped according to sublineage, the resolution scale for phylogenetic inference is much higher when using genes or SNPs, as compared to the DR locus and spacers analyzed from the spoligotyping technique or the tandem repeat regions present in the MIRU-VNTR that are found much less frequently in the genome. The phylogenetic trees performed with both pan-genome complete gene sequences and the concatenation of single nucleotide polymorphisms present in the genomic core were highly congruent and statistically robust. When compared with the spoligotyping and 24-loci MIRU-VNTR, a greater resolution power was observed by the genes and SNPs for being more diverse and abundant markers throughout the genome. This latter allowed us the identification of the UT277 H1 3022 isolate initially classified as a low prevalence sublineage to a high prevalence UT277 H1 SIT62 isolate.</p>
</sec>
<sec id="S4.SS5">
<title>Pan-genome-wide association study</title>
<p>The pan-genome allowed the study of the association of genetic variants with the high or low prevalence trait of the isolates of <italic>Mtb</italic> lineage-4. In 11 genes 12 variants among insertions, deletions and SNPs were identified. Four genes with the highest statistical association with low prevalence were <italic>mmpL12</italic>, <italic>PPE29</italic>, <italic>Rv1419</italic>, <italic>Rv1762c</italic>. These encode proteins are part of the membrane or the cell wall of the mycobacterium. The remaining seven genes were significantly associated with high prevalence; <italic>Rv3371</italic>, <italic>Rv2735c</italic>, <italic>scoA</italic>, <italic>mhpE</italic>, PE-PGRS42, <italic>lppB</italic>, and <italic>gabD2</italic> which have different biological and functions in the mycobacteria.</p>
</sec>
<sec id="S4.SS6">
<title>Low prevalence genes</title>
<p>We found that the <italic>mmpL12</italic> gene (14 of 20 isolates) had an insertion at position 1642_3000ins. This insertion of 1,358 bp generates a change in the open reading frame giving rise to two CDS of 1,740 and 1,788 bp. The insertion corresponded to a mobile genetic element IS6110, it is known that the insertion of IS6110 in coding regions of the genome generally produces an inactive gene (<xref ref-type="bibr" rid="B37">Gonzalo-Asensio et al., 2018</xref>). Despite having used WGS, the repetitive nature of IS6110 represented a technical challenge for both precise identification in the <italic>mmpL12</italic> gene as for its experimental validation, this difficulty has been reported previously to define its precise location in the chromosome due to these repetitive regions (<xref ref-type="bibr" rid="B59">Reyes et al., 2012</xref>; <xref ref-type="bibr" rid="B37">Gonzalo-Asensio et al., 2018</xref>). The presence of the IS6110 insertion in the <italic>mmpL12</italic> gene had already been previously identified but its effects on the gene are unknown (<xref ref-type="bibr" rid="B59">Reyes et al., 2012</xref>). Nevertheless, we hypothesize that the two proteins predicted as a result of the insertion in <italic>mmpL12</italic>, in low prevalence isolates, have most likely lost their original function, due to the loss of 5 transmembrane domains and one periplasmic domain (<xref ref-type="supplementary-material" rid="FS8">Supplementary Figure 8</xref>).</p>
<p>The proteins of the MmpL family are located in the cell membrane and their main role is the transport of lipids and siderophores directly related to the survival, virulence, and pathogenicity through the plasma membrane toward the periplasmic space (<xref ref-type="bibr" rid="B27">Domenech et al., 2005</xref>; <xref ref-type="bibr" rid="B48">Melly and Purdy, 2019</xref>). The substrate carried by this protein (MMPL12) is not yet known with certainty. However, some studies have revealed that the <italic>mmpl12</italic> gene is close to the <italic>pks</italic> and <italic>fad</italic> genes, which may indicate that the substrates for this protein are glyco or polyketides. An ortholog of <italic>mmpL12</italic> in <italic>M. abscessus</italic> (MAB_0855) was recently identified as a transporter for glycosyl diacylated nonadecyl diol (GDND), suggesting that the substrate for MMPL12 in <italic>Mtb</italic> is an undescribed cell envelope glycolipid which could exert similar functions as GDND in <italic>M. abscessus</italic> such as protective functions allowing survival within the host immune cells (<xref ref-type="bibr" rid="B29">Dubois et al., 2018</xref>).</p>
<p>We hypothesize that, if the mycobacteria does not have a functional MMPL12 protein that efficiently transports GDND, the cell wall will not have the basic requirements for this glycolipid to confront the host&#x2019;s immune system and may be more easily cleared. This phenomenon would help to explain the low prevalence in the population of the isolates that presented insertion 1642_300ins in <italic>mmpL12</italic>.</p>
<p>The <italic>PPE29</italic> gene presented an insertion at position 639_641insG shifting the open reading frame in low prevalence isolates. In the wild type, the PPE29 protein is required for endothelial cell invasion and intracellular survival of <italic>Mtb</italic> (<xref ref-type="bibr" rid="B36">Gey van Pittius et al., 2006</xref>). It is likely that this insertion is affecting PPE29 protein function, reducing the ability of the bacteria to survive the action of the host immune system. As its fitness is affected, the ability to survive within cells will not be able to efficiently establish the infectious process, since it is an obligate intracellular pathogen.</p>
<p>The <italic>Rv1419</italic> gene encodes a lectin, the deletion identified at position 199delA generated a smaller protein in 14 of the 20 low prevalence genomes, which possibly inactivates its function. Lectins are a family of secretory proteins with the ability to specifically bind to carbohydrates. Several pathogens have been shown to express these molecules and to be highly involved in recognition and invasion processes (<xref ref-type="bibr" rid="B55">Nogueira et al., 2010</xref>). There is evidence to suggest that lectin-host interactions in <italic>Mtb</italic> are a potential mechanism to facilitate the establishment of infection. In this sense, it has been shown that lectins derived from <italic>Mtb</italic> could play an important role in infection <italic>in vivo</italic> (<xref ref-type="bibr" rid="B55">Nogueira et al., 2010</xref>; <xref ref-type="bibr" rid="B44">Kolbe et al., 2019</xref>). It is likely that the deletion found in the <italic>Rv1419</italic> gene in low prevalence isolates altered this host-pathogen interaction and therefore the ability of the mycobacterium to establish an infectious process.</p>
<p>Unlike the high-prevalence genomes, in 13 of the 20 low-prevalence genomes the <italic>Rv1762c</italic> gene contained one SNP at position 538C &#x003E; T, producing a premature stop codon. Sequence analysis suggests it encodes for a non-essential protein that has two putative heavy metal-binding domains, located in a fraction of the cell membrane. It is likely to be related to the regulation or transport of metal ion concentrations across the membrane as a physiological mechanism between the mycobacteria-host (<xref ref-type="supplementary-material" rid="FS9">Supplementary Figure 9</xref>; <xref ref-type="bibr" rid="B1">Agranoff and Krishna, 2004</xref>).</p>
</sec>
<sec id="S4.SS7">
<title>High prevalence genes</title>
<p>We found that the <italic>Rv3371</italic> gene had a SNP at position 374T &#x003E; A and a deletion at position 376delA in 16 of the 27 high-prevalence genomes, which generate an open reading frame shift and result in a smaller protein that is probably not functional. These results are consistent with previous studies, as this deletion was found only in high-prevalence isolates (<xref ref-type="bibr" rid="B23">Daniel et al., 2004</xref>; <xref ref-type="bibr" rid="B63">Sirakova, 2006</xref>). The <italic>Rv3371</italic> gene encodes for the enzyme diacylglycerol acyltransferase, involved in the synthesis of triacylglycerols, these are important in the survival of the bacterium when it is in a latent state. The mycobacteria store lipids in intracellular inclusion bodies in the form of triacylglycerols which will be a source of carbon and fatty acids during the latent state. The <italic>Rv3371</italic> promoter has been shown to be under-expressed when mycobacteria are in a state of active replication <italic>in vitro</italic> and is overexpressed when the bacilli enter conditions of hypoxia and low metabolic activity (<xref ref-type="bibr" rid="B23">Daniel et al., 2004</xref>; <xref ref-type="bibr" rid="B63">Sirakova, 2006</xref>). Studies with mutants of the <italic>Rv3371</italic> gene found that mycobacteria could not enter the non-replicative phase under conditions of hypoxia, nitrosative stress, or iron depletion; that is, they did not have the ability to go into latency. This inability was due to the low formation of vesicles that stored triacylglycerols. The deletion of this gene did not affect the <italic>in vitro</italic> growth of the mycobacteria or the morphology of the colonies (<xref ref-type="bibr" rid="B23">Daniel et al., 2004</xref>; <xref ref-type="bibr" rid="B63">Sirakova, 2006</xref>). It is likely that these isolates present an inability to enter a state of latency and this contributes to more cases of active TB, associated with an increase in the transmission of these isolates.</p>
<p>The <italic>scoA</italic> gene was found over-expressed in a study carried out with a persistence model, which indicates that this gene is probably involved in the metabolism of mycobacteria which are in a state of latency (<xref ref-type="bibr" rid="B32">Fleischmann et al., 2002</xref>). In this work, the scoA gene showed the 476insG insertion, which generated a change in the codon (TAT) for a stop codon (TAG) at position 159, truncating the protein. The effect of this mutation is most likely inactivation of the protein. This gene encodes the enzyme succinyl-CoA:3-ketoacid-coenzyme transferase subunit A, which is used by mycobacteria for the utilization of ketones resulting from B-oxidation. Studies in mycobacteria suggest that dormant bacilli use fatty acids as their predominant source of energy through &#x03B2;-oxidation pathways (<xref ref-type="bibr" rid="B32">Fleischmann et al., 2002</xref>). It is possible that this insertion in <italic>scoA</italic> is affecting the ability to go into the state of latency of the mycobacteria and this favors it to remain in a state of active replication, facilitating its transmission and high prevalence.</p>
<p>The <italic>Rv2735c</italic> gene encodes a hypothetical conserved 330 amino acid protein with unknown function, 16 of the 27 high prevalence isolates present an insert at position 363_364insGT displacing the open reading frame. This generates a truncated protein of 159 amino acids, which is possibly not functional. The <italic>lppB</italic> gene encodes a lipoprotein which has not yet have a known function (<xref ref-type="bibr" rid="B64">Sutcliffe and Harrington, 2004</xref>). Lipoproteins are found in the cell envelope and help support this structure. Several of them have been characterized as virulence factors (<xref ref-type="bibr" rid="B64">Sutcliffe and Harrington, 2004</xref>). In this work, this gene presented a SNP at position 362C &#x003E; A, this polymorphism produces a non-synonymous mutation changing the codon (TCG) of a serine (S) for a stop codon (TAG), truncating the protein at position 120 of the open reading frame. It is difficult to hypothesize why these genes are possibly truncated in high prevalence isolates. Functional studies would be needed to determine their role and the possible effect in high prevalence isolates.</p>
<p>The <italic>mhpE</italic> and <italic>gabD2</italic> genes encode the enzymes 4-hydroxy-2-oxopentanoic acid aldolase and succinate-semialdehyde dehydrogenase, respectively. These enzymes are involved with intermediates of the tricarboxylic acid cycle such as pyruvate (<italic>mhpE</italic>) and succinate (<italic>gabD2</italic>) (<xref ref-type="bibr" rid="B68">Tian et al., 2005</xref>; <xref ref-type="bibr" rid="B12">Carere et al., 2013</xref>). When mycobacteria are found under hypoxic or intracellular conditions within macrophages, these enzymes participate in alternative pathways to the tricarboxylic acid cycle for obtaining energy (<xref ref-type="bibr" rid="B68">Tian et al., 2005</xref>; <xref ref-type="bibr" rid="B12">Carere et al., 2013</xref>). These alternate pathways appear to be most active when the bacteria enter in a reduction of metabolic activity. In this work, the <italic>mhpE</italic> and <italic>gabD2</italic> genes presented a SNP and insertion, respectively. These changes were observed in the high prevalence isolates, which could be indicating that these isolates, being in a phase of active replication and circulating within a population, it may not be necessary to activate alternative routes of the tricarboxylic acid cycle.</p>
<p>The genetic variants identified in this association study to the prevalence were part of the <italic>Mtb</italic> dispensable genome. The importance of these gene variants in transmission, pathogenicity and virulence of <italic>Mtb</italic> has not yet been experimentally proven. Experimental trials in animal models infected with high and low prevalence <italic>Mtb</italic> isolates, in addition to the use of CRISPR technology for editing the genes identified with association to prevalence, should be carried out to demonstrate their role in pathogenicity and transmission. Similar trials of infection and transmission of <italic>Mtb</italic> have been carried out in mini pigs, proving to be a suitable model for the study of infection and natural transmission of TB (<xref ref-type="bibr" rid="B16">Clark et al., 2014</xref>; <xref ref-type="bibr" rid="B15">Choudhary et al., 2015</xref>; <xref ref-type="bibr" rid="B57">Ramos et al., 2017</xref>).</p>
<p>Despite the fact that the pan-genome association study was performed independent of the sublineage, the phylogenomic analysis together with the pan-GWAS showed that most of the differential variants identified in the high or low prevalence of <italic>Mtb</italic> also showed an association with the sublineage. When comparing the topology of the <italic>Mtb</italic> isolates in the clades of the trees grouped by their close evolutionary relationships with the differential variants between high and low prevalence, it was observed that the presence or absence of the variant was related to certain sublineages. Suggesting that particular dominant sublineages such as LAM9 SIT42 and H1 SIT62 have evolutionary advantages that could be due to the genetic variants associated with the high prevalence of these sublineages over the other sublineages present in the NE area of Medell&#x00ED;n. The above allows us to infer that the high prevalence trait is not given by a single polymorphism or variant, but by multiple variants in the genome that probably contribute to the success of <italic>Mtb</italic> sublineages LAM9 SIT42 and H1 SIT62.</p>
<p>Several studies have shown that lineages and sublineages are associated with virulence and transmission traits. For example, the outbreak of the most prevalent <italic>Mtb</italic> lineage in Scandinavia, the specific strain 2/1112-15 (C2), between the years 1992 to 2014 in Denmark the rapid transmission of the C2 genotype was evidenced, suggesting that specific virulence factors added poor TB control favored the success of this genotype (<xref ref-type="bibr" rid="B33">Folkvardsen et al., 2018</xref>). In another study, the association of the LAM RD<sup>Rio</sup> sublineage with drug resistance has been demonstrated; 33% of a total of 857 <italic>Mtb</italic> isolates analyzed in Portugal were due to the strain RD<sup>Rio</sup> and represented more than 60% of the MDR strains, the most predominant sublineage was the RD<sup>Rio</sup> LAM1 SIT20. Analysis with 12 loci MIRU-VNTR of the RD<sup>Rio</sup> strain revealed that 96.3% (129/134) of the MDR and XDR clusters belonged to RD<sup>Rio</sup> strains (<xref ref-type="bibr" rid="B26">David et al., 2012</xref>). Strains of the Beijing lineage demonstrated an increase in their fitness of transmission when they were resistant to streptomycin when compared with strains of the EAI lineage. Inferring that streptomycin resistance, contrary to popular belief, gives Beijing strains an advantage in fitness compared to other genotypes (<xref ref-type="bibr" rid="B9">Buu et al., 2012</xref>).</p>
<p>It is likely that the adaptive advantages of certain lineages or sublineages improve their fitness, generating compensatory mutations allowing them to be more successful. In contrast, in <italic>Mtb</italic> lineages with low prevalence their fitness cost is higher, making the process of transmission and infection difficult in a new host. Variants have been identified in genes under positive selection, with association to the virulence of <italic>Mtb</italic>, these have shown the expansion of clones (Beijing) in different branches, making them more successful than others (<xref ref-type="bibr" rid="B49">Merker et al., 2015</xref>). Strains of this lineage have been suggested to have selective advantages compared to other strains of MTBC lineages, such as increasing their ability to acquire drug resistance, linked to the high frequency of mutations, many of them compensatory mutations that mitigate the fitness cost created by resistance mutations. This increases their transmissibility, virulence, and favors a rapid progression to disease after infection (<xref ref-type="bibr" rid="B10">Caminero et al., 2001</xref>; <xref ref-type="bibr" rid="B42">Jou et al., 2005</xref>; <xref ref-type="bibr" rid="B22">Cowley et al., 2008</xref>; <xref ref-type="bibr" rid="B9">Buu et al., 2012</xref>). This heterogeneity suggests the existence of sublineages with biogeographic diversity, which present particular pathogenic properties associated with their biology (<xref ref-type="bibr" rid="B10">Caminero et al., 2001</xref>; <xref ref-type="bibr" rid="B42">Jou et al., 2005</xref>; <xref ref-type="bibr" rid="B22">Cowley et al., 2008</xref>; <xref ref-type="bibr" rid="B9">Buu et al., 2012</xref>).</p>
<p>Our results suggest that the variants identified in the LAM9 SIT42 and H1 SIT62 high prevalence sublineages possibly contribute to their virulence, pathogenicity, or the ability to transmission in a community, conferring genetic advantages that favor the transmission success of these isolates compared to the less prevalent isolates. In these latter sublineages, the variants that are affecting the functionality of the <italic>mmpL12</italic>, <italic>PPE29</italic>, <italic>Rv1419</italic>, and <italic>Rv1762c</italic> genes may have a fitness cost, which is reflected in a possible decrease in their ability to generate an active infection. Interestingly, the <italic>mmpL12</italic>, <italic>PPE29</italic>, <italic>Rv1419</italic> genes are found related to intracellular recognition, invasion, and survival processes directly involved in host-pathogen interaction (<xref ref-type="bibr" rid="B27">Domenech et al., 2005</xref>; <xref ref-type="bibr" rid="B36">Gey van Pittius et al., 2006</xref>; <xref ref-type="bibr" rid="B55">Nogueira et al., 2010</xref>). It is possible that the polymorphisms found in these genes have a high fitness cost in the low prevalence isolates, making their transmission difficult. It is necessary to continue increasing the number of isolates in this type of study to determine the generality of these variants in different lineages and sublineages that behave with phenotypic traits of high and low prevalence in other geographic areas of Colombia and the world and thus verify the universality of the results.</p>
</sec>
</sec>
<sec id="S5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The data presented in this study are deposited in the <ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/">https://www.ncbi.nlm.nih.gov/</ext-link> repository, accession number: <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="PRJNA838941">PRJNA838941</ext-link>.</p>
</sec>
<sec id="S6" sec-type="author-contributions">
<title>Author contributions</title>
<p>JR, FR, UH-P, and RA contributed to the conception and design of the study. N&#x00C1; performed the DNA isolation and PCR validations. UH-P and BC-M carried out the bioinformatic and statistical analysis. UH-P wrote the first draft of the manuscript. All authors contributed to the manuscript revision, read, and approved the submitted version.</p>
</sec>
</body>
<back>
<sec id="S7" sec-type="funding-information">
<title>Funding</title>
<p>This work was funded by the Ministerio de Ciencia Tecnolog&#x00ED;a e Innovaci&#x00F3;n&#x2013;Minciencias (Project code: 657057636375).</p>
</sec>
<ack>
<p>We thank Teresa Realpe for supplying the MIRUS-VNTR and SIT data from the clinical isolates and Carlos Cantalapiedra for Bioinformatic Technical support.</p>
</ack>
<sec id="S8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="S9" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="S10" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fmicb.2022.1076797/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fmicb.2022.1076797/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Image_1.TIFF" id="FS1" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image_2.TIFF" id="FS2" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image_3.TIFF" id="FS3" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image_4.tiff" id="FS4" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image_5.tiff" id="FS5" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image_6.tiff" id="FS6" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image_7.tiff" id="FS7" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image_8.tiff" id="FS8" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image_9.tiff" id="FS9" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_1.PDF" id="TS1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_2.PDF" id="TS2" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_3.PDF" id="TS3" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_4.PDF" id="TS4" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_5.PDF" id="TS5" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_6.PDF" id="TS6" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Agranoff</surname> <given-names>D.</given-names></name> <name><surname>Krishna</surname> <given-names>S.</given-names></name></person-group> (<year>2004</year>). <article-title>Metal ion transport and regulation in <italic>Mycobacterium tuberculosis</italic>.</article-title> <source><italic>Front. Biosci.</italic></source> <volume>9</volume> <fpage>2996</fpage>&#x2013;<lpage>3006</lpage>. <pub-id pub-id-type="doi">10.2741/1454</pub-id> <pub-id pub-id-type="pmid">15353332</pub-id></citation></ref>
<ref id="B2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Almanza</surname> <given-names>R.</given-names></name> <name><surname>Montes</surname> <given-names>F.</given-names></name> <name><surname>Gonz&#x00E1;lez</surname> <given-names>D.</given-names></name> <name><surname>Zapata</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). <source><italic>Situaci&#x00F3;n de la tuberculosis en medell&#x00ED;n 2018.</italic></source> <publisher-loc>Secretar&#x00ED;a de Salud de Medell&#x00ED;n</publisher-loc>: <publisher-name>Boletin epidemiol&#x00F3;gico</publisher-name>.</citation></ref>
<ref id="B3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Altschul</surname> <given-names>S.</given-names></name> <name><surname>Gish</surname> <given-names>W.</given-names></name> <name><surname>Miller</surname> <given-names>W.</given-names></name> <name><surname>Myers</surname> <given-names>E.</given-names></name> <name><surname>Lipman</surname> <given-names>D.</given-names></name></person-group> (<year>1990</year>). <article-title>Basic local alignment search tool.</article-title> <source><italic>J. Mol. Biol.</italic></source> <volume>215</volume> <fpage>403</fpage>&#x2013;<lpage>410</lpage>. <pub-id pub-id-type="doi">10.1016/S0022-2836(05)80360-2</pub-id></citation></ref>
<ref id="B4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Andrews</surname> <given-names>S.</given-names></name></person-group> (<year>2010</year>). <source><italic>FastQC a quality control tool for high throughput sequence data.</italic></source> <comment>Available online at:</comment> <ext-link ext-link-type="uri" xlink:href="http://www.bioinformatics.babraham.ac.uk/projects/fastqc/">http://www.bioinformatics.babraham.ac.uk/projects/fastqc/</ext-link> <comment>(accessed July, 2017)</comment>.</citation></ref>
<ref id="B5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Apweiler</surname> <given-names>R.</given-names></name></person-group> (<year>2004</year>). <article-title>UniProt: The universal protein knowledgebase.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>32</volume>:<fpage>115D</fpage>&#x2013;<lpage>119D</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkh131</pub-id> <pub-id pub-id-type="pmid">14681372</pub-id></citation></ref>
<ref id="B6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bankevich</surname> <given-names>A.</given-names></name> <name><surname>Nurk</surname> <given-names>S.</given-names></name> <name><surname>Antipov</surname> <given-names>D.</given-names></name> <name><surname>Gurevich</surname> <given-names>A.</given-names></name> <name><surname>Dvorkin</surname> <given-names>M.</given-names></name> <name><surname>Kulikov</surname> <given-names>A.</given-names></name><etal/></person-group> (<year>2012</year>). <article-title>SPAdes: A new genome assembly algorithm and its applications to single-cell sequencing.</article-title> <source><italic>J. Comput. Biol.</italic></source> <volume>19</volume> <fpage>455</fpage>&#x2013;<lpage>477</lpage>. <pub-id pub-id-type="doi">10.1089/cmb.2012.0021</pub-id> <pub-id pub-id-type="pmid">22506599</pub-id></citation></ref>
<ref id="B7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bolger</surname> <given-names>A.</given-names></name> <name><surname>Lohse</surname> <given-names>M.</given-names></name> <name><surname>Usadel</surname> <given-names>B.</given-names></name></person-group> (<year>2014</year>). <article-title>Trimmomatic: A flexible trimmer for illumina sequence data.</article-title> <source><italic>Bioinformatics</italic></source> <volume>30</volume> <fpage>2114</fpage>&#x2013;<lpage>2120</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btu170</pub-id> <pub-id pub-id-type="pmid">24695404</pub-id></citation></ref>
<ref id="B8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brynildsrud</surname> <given-names>O.</given-names></name> <name><surname>Bohlin</surname> <given-names>J.</given-names></name> <name><surname>Scheffer</surname> <given-names>L.</given-names></name> <name><surname>Eldholm</surname> <given-names>V.</given-names></name></person-group> (<year>2016</year>). <article-title>Rapid scoring of genes in microbial pan-genome-wide association studies with scoary.</article-title> <source><italic>Genome. Biol.</italic></source> <volume>17</volume>:<issue>238</issue>. <pub-id pub-id-type="doi">10.1186/s13059-016-1108-8</pub-id> <pub-id pub-id-type="pmid">27887642</pub-id></citation></ref>
<ref id="B9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Buu</surname> <given-names>T.</given-names></name> <name><surname>van Soolingen</surname> <given-names>D.</given-names></name> <name><surname>Huyen</surname> <given-names>M.</given-names></name> <name><surname>Lan</surname> <given-names>N.</given-names></name> <name><surname>Quy</surname> <given-names>H.</given-names></name> <name><surname>Tiemersma</surname> <given-names>E.</given-names></name><etal/></person-group> (<year>2012</year>). <article-title>Increased transmission of <italic>Mycobacterium tuberculosis</italic> Beijing genotype strains associated with resistance to streptomycin: A population-based study.</article-title> <source><italic>PLoS One.</italic></source> <volume>7</volume>:<issue>e42323</issue>. <pub-id pub-id-type="doi">10.1371/journal.pone.0042323</pub-id> <pub-id pub-id-type="pmid">22912700</pub-id></citation></ref>
<ref id="B10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Caminero</surname> <given-names>J.</given-names></name> <name><surname>Pena</surname> <given-names>M.</given-names></name> <name><surname>Campos-Herrero</surname> <given-names>M.</given-names></name> <name><surname>Rodr&#x00ED;guez</surname> <given-names>J.</given-names></name> <name><surname>Garc&#x00ED;a</surname> <given-names>I.</given-names></name> <name><surname>Cabrera</surname> <given-names>P.</given-names></name><etal/></person-group> (<year>2001</year>). <article-title>Epidemiological evidence of the spread of a <italic>Mycobacterium tuberculosis</italic> strain of the Beijing genotype on Gran Canaria Island.</article-title> <source><italic>Am. J. Respir. Crit. Care Med.</italic></source> <volume>164</volume> <fpage>1165</fpage>&#x2013;<lpage>1170</lpage>. <pub-id pub-id-type="doi">10.1164/ajrccm.164.7.2101031</pub-id> <pub-id pub-id-type="pmid">11673204</pub-id></citation></ref>
<ref id="B11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Capella-Gutierrez</surname> <given-names>S.</given-names></name> <name><surname>Silla-Martinez</surname> <given-names>J.</given-names></name> <name><surname>Gabaldon</surname> <given-names>T.</given-names></name></person-group> (<year>2009</year>). <article-title>trimAl: A tool for automated alignment trimming in large-scale phylogenetic analyses.</article-title> <source><italic>Bioinformatics</italic></source> <volume>25</volume> <fpage>1972</fpage>&#x2013;<lpage>1973</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp348</pub-id> <pub-id pub-id-type="pmid">19505945</pub-id></citation></ref>
<ref id="B12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Carere</surname> <given-names>J.</given-names></name> <name><surname>McKenna</surname> <given-names>S.</given-names></name> <name><surname>Kimber</surname> <given-names>M.</given-names></name> <name><surname>Seah</surname> <given-names>S.</given-names></name></person-group> (<year>2013</year>). <article-title>Characterization of an aldolase&#x2013;dehydrogenase complex from the cholesterol degradation pathway of <italic>Mycobacterium tuberculosis</italic>.</article-title> <source><italic>Biochemistry</italic></source> <volume>52</volume> <fpage>3502</fpage>&#x2013;<lpage>3511</lpage>. <pub-id pub-id-type="doi">10.1021/bi400351h</pub-id> <pub-id pub-id-type="pmid">23614353</pub-id></citation></ref>
<ref id="B13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chaudhari</surname> <given-names>N.</given-names></name> <name><surname>Gupta</surname> <given-names>V.</given-names></name> <name><surname>Dutta</surname> <given-names>C.</given-names></name></person-group> (<year>2016</year>). <article-title>BPGA- an ultra-fast pan-genome analysis pipeline.</article-title> <source><italic>Sci. Rep.</italic></source> <volume>6</volume>:<issue>24373</issue>. <pub-id pub-id-type="doi">10.1038/srep24373</pub-id> <pub-id pub-id-type="pmid">27071527</pub-id></citation></ref>
<ref id="B14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chiaradia</surname> <given-names>L.</given-names></name> <name><surname>Lefebvre</surname> <given-names>C.</given-names></name> <name><surname>Parra</surname> <given-names>J.</given-names></name> <name><surname>Marcoux</surname> <given-names>J.</given-names></name> <name><surname>Burlet-Schiltz</surname> <given-names>O.</given-names></name> <name><surname>Etienne</surname> <given-names>G.</given-names></name><etal/></person-group> (<year>2017</year>). <article-title>Dissecting the mycobacterial cell envelope and defining the composition of the native mycomembrane.</article-title> <source><italic>Sci. Rep.</italic></source> <volume>7</volume>:<issue>12807</issue>. <pub-id pub-id-type="doi">10.1038/s41598-017-12718-4</pub-id> <pub-id pub-id-type="pmid">28993692</pub-id></citation></ref>
<ref id="B15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Choudhary</surname> <given-names>E.</given-names></name> <name><surname>Thakur</surname> <given-names>P.</given-names></name> <name><surname>Pareek</surname> <given-names>M.</given-names></name> <name><surname>Agarwal</surname> <given-names>N.</given-names></name></person-group> (<year>2015</year>). <article-title>Gene silencing by CRISPR interference in mycobacteria.</article-title> <source><italic>Nat. Commun.</italic></source> <volume>6</volume>:<issue>6267</issue>. <pub-id pub-id-type="doi">10.1038/ncomms7267</pub-id> <pub-id pub-id-type="pmid">25711368</pub-id></citation></ref>
<ref id="B16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Clark</surname> <given-names>S.</given-names></name> <name><surname>Hall</surname> <given-names>Y.</given-names></name> <name><surname>Williams</surname> <given-names>A.</given-names></name></person-group> (<year>2014</year>). <article-title>Animal models of tuberculosis: Guinea pigs.</article-title> <source><italic>Cold Spring Harb. Perspect. Med.</italic></source> <volume>5</volume>:<issue>a018572</issue>. <pub-id pub-id-type="doi">10.1101/cshperspect.a018572</pub-id> <pub-id pub-id-type="pmid">25524720</pub-id></citation></ref>
<ref id="B17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cole</surname> <given-names>S.</given-names></name> <name><surname>Brosch</surname> <given-names>R.</given-names></name> <name><surname>Parkhill</surname> <given-names>J.</given-names></name> <name><surname>Garnier</surname> <given-names>T.</given-names></name> <name><surname>Churcher</surname> <given-names>C.</given-names></name> <name><surname>Harris</surname> <given-names>D.</given-names></name><etal/></person-group> (<year>1998</year>). <article-title>Deciphering the biology of <italic>Mycobacterium tuberculosis</italic> from the complete genome sequence.</article-title> <source><italic>Nature</italic></source> <volume>396</volume>:<issue>27</issue>. <pub-id pub-id-type="doi">10.1038/24206</pub-id></citation></ref>
<ref id="B18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Contreras-Moreira</surname> <given-names>B.</given-names></name> <name><surname>Vinuesa</surname> <given-names>P.</given-names></name></person-group> (<year>2013</year>). <article-title>GET_HOMOLOGUES, a versatile software package for scalable and robust microbial pangenome analysis.</article-title> <source><italic>Appl. Environ. Microbiol.</italic></source> <volume>79</volume> <fpage>7696</fpage>&#x2013;<lpage>7701</lpage>. <pub-id pub-id-type="doi">10.1128/AEM.02411-13</pub-id> <pub-id pub-id-type="pmid">24096415</pub-id></citation></ref>
<ref id="B19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Coros</surname> <given-names>A.</given-names></name> <name><surname>DeConno</surname> <given-names>E.</given-names></name> <name><surname>Derbyshire</surname> <given-names>K.</given-names></name></person-group> (<year>2008</year>). <article-title>IS6110, a <italic>Mycobacterium tuberculosis</italic> complex-specific insertion sequence, is also present in the genome of <italic>Mycobacterium smegmatis</italic>, suggestive of lateral gene transfer among mycobacterial species.</article-title> <source><italic>J. Bacteriol.</italic></source> <volume>190</volume> <fpage>3408</fpage>&#x2013;<lpage>3410</lpage>. <pub-id pub-id-type="doi">10.1128/JB.00009-08</pub-id> <pub-id pub-id-type="pmid">18326566</pub-id></citation></ref>
<ref id="B20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Coscolla</surname> <given-names>M.</given-names></name> <name><surname>Gagneux</surname> <given-names>S.</given-names></name></person-group> (<year>2014</year>). <article-title>Consequences of genomic diversity in <italic>Mycobacterium tuberculosis</italic>.</article-title> <source><italic>Semin. Immunol.</italic></source> <volume>26</volume> <fpage>431</fpage>&#x2013;<lpage>444</lpage>. <pub-id pub-id-type="doi">10.1016/j.smim.2014.09.012</pub-id> <pub-id pub-id-type="pmid">25453224</pub-id></citation></ref>
<ref id="B21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Couvin</surname> <given-names>D.</given-names></name> <name><surname>Rastogi</surname> <given-names>N.</given-names></name></person-group> (<year>2015</year>). <article-title>Tuberculosis &#x2013; A global emergency: Tools and methods to monitor, understand, and control the epidemic with specific example of the Beijing lineage.</article-title> <source><italic>Tuberculosis</italic></source> <volume>95</volume>:<fpage>S177</fpage>&#x2013;<lpage>S189</lpage>. <pub-id pub-id-type="doi">10.1016/j.tube.2015.02.023</pub-id> <pub-id pub-id-type="pmid">25797613</pub-id></citation></ref>
<ref id="B22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cowley</surname> <given-names>D.</given-names></name> <name><surname>Govender</surname> <given-names>D.</given-names></name> <name><surname>February</surname> <given-names>B.</given-names></name> <name><surname>Wolfe</surname> <given-names>M.</given-names></name> <name><surname>Steyn</surname> <given-names>L.</given-names></name> <name><surname>Evans</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2008</year>). <article-title>Recent and rapid emergence of W-Beijing strains of <italic>Mycobacterium tuberculosis</italic> in Cape Town, South Africa.</article-title> <source><italic>Clin. Infect. Dis.</italic></source> <volume>47</volume> <fpage>1252</fpage>&#x2013;<lpage>1259</lpage>. <pub-id pub-id-type="doi">10.1086/592575</pub-id> <pub-id pub-id-type="pmid">18834315</pub-id></citation></ref>
<ref id="B23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Daniel</surname> <given-names>J.</given-names></name> <name><surname>Deb</surname> <given-names>C.</given-names></name> <name><surname>Dubey</surname> <given-names>V.</given-names></name> <name><surname>Sirakova</surname> <given-names>T.</given-names></name> <name><surname>Abomoelak</surname> <given-names>B.</given-names></name> <name><surname>Morbidoni</surname> <given-names>H.</given-names></name><etal/></person-group> (<year>2004</year>). <article-title>Induction of a novel class of diacylglycerol acyltransferases and triacylglycerol accumulation in <italic>Mycobacterium tuberculosis</italic> as it goes into a dormancy-like state in culture.</article-title> <source><italic>J. Bacteriol.</italic></source> <volume>186</volume> <fpage>5017</fpage>&#x2013;<lpage>5030</lpage>. <pub-id pub-id-type="doi">10.1128/JB.186.15.5017-5030.2004</pub-id> <pub-id pub-id-type="pmid">15262939</pub-id></citation></ref>
<ref id="B24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dar</surname> <given-names>H.</given-names></name> <name><surname>Zaheer</surname> <given-names>T.</given-names></name> <name><surname>Ullah</surname> <given-names>N.</given-names></name> <name><surname>Bakhtiar</surname> <given-names>S.</given-names></name> <name><surname>Zhang</surname> <given-names>T.</given-names></name> <name><surname>Yasir</surname> <given-names>M.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>Pangenome analysis of <italic>Mycobacterium tuberculosis</italic> reveals core-drug targets and screening of promising lead compounds for drug discovery.</article-title> <source><italic>Antibiotics</italic></source> <volume>9</volume>:<issue>819</issue>. <pub-id pub-id-type="doi">10.3390/antibiotics9110819</pub-id> <pub-id pub-id-type="pmid">33213029</pub-id></citation></ref>
<ref id="B25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Darling</surname> <given-names>A.</given-names></name> <name><surname>Mau</surname> <given-names>B.</given-names></name> <name><surname>Blattner</surname> <given-names>F.</given-names></name> <name><surname>Perna</surname> <given-names>N.</given-names></name></person-group> (<year>2004</year>). <article-title>Mauve: Multiple alignment of conserved genomic sequence with rearrangements.</article-title> <source><italic>Genome Res.</italic></source> <volume>14</volume> <fpage>1394</fpage>&#x2013;<lpage>1403</lpage>. <pub-id pub-id-type="doi">10.1101/gr.2289704</pub-id> <pub-id pub-id-type="pmid">15231754</pub-id></citation></ref>
<ref id="B26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>David</surname> <given-names>S.</given-names></name> <name><surname>Duarte</surname> <given-names>E.</given-names></name> <name><surname>Leite</surname> <given-names>C.</given-names></name> <name><surname>Ribeiro</surname> <given-names>J.</given-names></name> <name><surname>Maio</surname> <given-names>J.</given-names></name> <name><surname>Paix&#x00E3;o</surname> <given-names>E.</given-names></name><etal/></person-group> (<year>2012</year>). <article-title>Implication of the RDRio <italic>Mycobacterium tuberculosis</italic> sublineage in multidrug resistant tuberculosis in Portugal.</article-title> <source><italic>Infect. Genet. Evol.</italic></source> <volume>12</volume> <fpage>1362</fpage>&#x2013;<lpage>1367</lpage>. <pub-id pub-id-type="doi">10.1016/j.meegid.2012.04.021</pub-id> <pub-id pub-id-type="pmid">22569099</pub-id></citation></ref>
<ref id="B27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Domenech</surname> <given-names>P.</given-names></name> <name><surname>Reed</surname> <given-names>M.</given-names></name> <name><surname>Barry</surname> <given-names>C.</given-names></name></person-group> (<year>2005</year>). <article-title>Contribution of the <italic>Mycobacterium tuberculosis</italic> MmpL protein family to virulence and drug resistance.</article-title> <source><italic>Infect. Immun.</italic></source> <volume>73</volume> <fpage>3492</fpage>&#x2013;<lpage>3501</lpage>. <pub-id pub-id-type="doi">10.1128/IAI.73.6.3492-3501.2005</pub-id> <pub-id pub-id-type="pmid">15908378</pub-id></citation></ref>
<ref id="B28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Donoghue</surname> <given-names>H.</given-names></name></person-group> (<year>2011</year>). <article-title>Insights gained from palaeomicrobiology into ancient and modern tuberculosis.</article-title> <source><italic>Clin. Microbiol. Infect.</italic></source> <volume>17</volume> <fpage>821</fpage>&#x2013;<lpage>829</lpage>. <pub-id pub-id-type="doi">10.1111/j.1469-0691.2011.03554.x</pub-id> <pub-id pub-id-type="pmid">21682803</pub-id></citation></ref>
<ref id="B29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dubois</surname> <given-names>V.</given-names></name> <name><surname>Viljoen</surname> <given-names>A.</given-names></name> <name><surname>Laencina</surname> <given-names>L.</given-names></name> <name><surname>Le Moigne</surname> <given-names>V.</given-names></name> <name><surname>Bernut</surname> <given-names>A.</given-names></name> <name><surname>Dubar</surname> <given-names>F.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>MmpL8 MAB controls <italic>Mycobacterium abscessus</italic> virulence and production of a previously unknown glycolipid family.</article-title> <source><italic>Proc. Natl. Acad. Sci. U.S.A.</italic></source> <volume>115</volume>:<fpage>E10147</fpage>&#x2013;<lpage>E10156</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1812984115</pub-id> <pub-id pub-id-type="pmid">30301802</pub-id></citation></ref>
<ref id="B30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Edgar</surname> <given-names>R.</given-names></name></person-group> (<year>2010</year>). <article-title>Search and clustering orders of magnitude faster than BLAST.</article-title> <source><italic>Bioinformatics</italic></source> <volume>26</volume> <fpage>2460</fpage>&#x2013;<lpage>2461</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btq461</pub-id> <pub-id pub-id-type="pmid">20709691</pub-id></citation></ref>
<ref id="B31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fischer</surname> <given-names>S.</given-names></name> <name><surname>Brunk</surname> <given-names>B.</given-names></name> <name><surname>Chen</surname> <given-names>F.</given-names></name> <name><surname>Gao</surname> <given-names>X.</given-names></name> <name><surname>Harb</surname> <given-names>O.</given-names></name> <name><surname>Iodice</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2011</year>). <article-title>Using OrthoMCL to assign proteins to OrthoMCL-DB groups or to cluster proteomes into new ortholog groups.</article-title> <source><italic>Curr. Protoc. Bioinformatics</italic></source> <volume>6</volume>:<fpage>6.12.1</fpage>&#x2013;<lpage>19</lpage>. <pub-id pub-id-type="doi">10.1002/0471250953.bi0612s35</pub-id> <pub-id pub-id-type="pmid">21901743</pub-id></citation></ref>
<ref id="B32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fleischmann</surname> <given-names>R.</given-names></name> <name><surname>Alland</surname> <given-names>D.</given-names></name> <name><surname>Eisen</surname> <given-names>J.</given-names></name> <name><surname>Carpenter</surname> <given-names>L.</given-names></name> <name><surname>White</surname> <given-names>O.</given-names></name> <name><surname>Peterson</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2002</year>). <article-title>Whole-genome comparison of <italic>Mycobacterium tuberculosis</italic> clinical and laboratory strains.</article-title> <source><italic>J. Bacteriol.</italic></source> <volume>184</volume> <fpage>5479</fpage>&#x2013;<lpage>5490</lpage>. <pub-id pub-id-type="doi">10.1128/JB.184.19.5479-5490.2002</pub-id> <pub-id pub-id-type="pmid">12218036</pub-id></citation></ref>
<ref id="B33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Folkvardsen</surname> <given-names>D.</given-names></name> <name><surname>Norman</surname> <given-names>A.</given-names></name> <name><surname>Andersen</surname> <given-names>&#x00C5;</given-names></name> <name><surname>Rasmussen</surname> <given-names>E.</given-names></name> <name><surname>Lillebaek</surname> <given-names>T.</given-names></name> <name><surname>Jelsbak</surname> <given-names>L.</given-names></name></person-group> (<year>2018</year>). <article-title>A major mycobacterium tuberculosis outbreak caused by one specific genotype in a low-incidence country: Exploring gene profile virulence explanations.</article-title> <source><italic>Sci. Rep.</italic></source> <volume>8</volume>:<issue>11869</issue>. <pub-id pub-id-type="doi">10.1038/s41598-018-30363-3</pub-id> <pub-id pub-id-type="pmid">30089859</pub-id></citation></ref>
<ref id="B34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Forrellad</surname> <given-names>M.</given-names></name> <name><surname>Klepp</surname> <given-names>L.</given-names></name> <name><surname>Gioffr&#x00E9;</surname> <given-names>A.</given-names></name> <name><surname>Sabio y Garc&#x00ED;a</surname> <given-names>J.</given-names></name> <name><surname>Morbidoni</surname> <given-names>H.</given-names></name> <name><surname>de la Paz Santangelo</surname> <given-names>M.</given-names></name><etal/></person-group> (<year>2013</year>). <article-title>Virulence factors of the <italic>Mycobacterium tuberculosis</italic> complex.</article-title> <source><italic>Virulence</italic></source> <volume>4</volume> <fpage>3</fpage>&#x2013;<lpage>66</lpage>. <pub-id pub-id-type="doi">10.4161/viru.22329</pub-id> <pub-id pub-id-type="pmid">23076359</pub-id></citation></ref>
<ref id="B35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Galagan</surname> <given-names>J.</given-names></name></person-group> (<year>2014</year>). <article-title>Genomic insights into <italic>tuberculosis</italic>.</article-title> <source><italic>Nat. Rev. Genet.</italic></source> <volume>15</volume> <fpage>307</fpage>&#x2013;<lpage>320</lpage>. <pub-id pub-id-type="doi">10.1038/nrg3664</pub-id> <pub-id pub-id-type="pmid">24662221</pub-id></citation></ref>
<ref id="B36"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gey van Pittius</surname> <given-names>N.</given-names></name> <name><surname>Sampson</surname> <given-names>S.</given-names></name> <name><surname>Lee</surname> <given-names>H.</given-names></name> <name><surname>Kim</surname> <given-names>Y.</given-names></name> <name><surname>van Helden</surname> <given-names>P.</given-names></name> <name><surname>Warren</surname> <given-names>R.</given-names></name></person-group> (<year>2006</year>). <article-title>Evolution and expansion of the <italic>Mycobacterium tuberculosis</italic> PE and PPE multigene families and their association with the duplication of the ESAT-6 (esx) gene cluster regions.</article-title> <source><italic>BMC Evol. Biol.</italic></source> <volume>6</volume>:<issue>95</issue>. <pub-id pub-id-type="doi">10.1186/1471-2148-6-95</pub-id> <pub-id pub-id-type="pmid">17105670</pub-id></citation></ref>
<ref id="B37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gonzalo-Asensio</surname> <given-names>J.</given-names></name> <name><surname>P&#x00E9;rez</surname> <given-names>I.</given-names></name> <name><surname>Aguil&#x00F3;</surname> <given-names>N.</given-names></name> <name><surname>Uranga</surname> <given-names>S.</given-names></name> <name><surname>Pic&#x00F3;</surname> <given-names>A.</given-names></name> <name><surname>Lampreave</surname> <given-names>C.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>New insights into the transposition mechanisms of IS6110 and its dynamic distribution between <italic>Mycobacterium tuberculosis</italic> complex lineages.</article-title> <source><italic>PLoS Genet.</italic></source> <volume>14</volume>:<issue>e1007282</issue>. <pub-id pub-id-type="doi">10.1371/journal.pgen.1007282</pub-id> <pub-id pub-id-type="pmid">29649213</pub-id></citation></ref>
<ref id="B38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gupta</surname> <given-names>S.</given-names></name> <name><surname>Rodriguez</surname> <given-names>G.</given-names></name></person-group> (<year>2018</year>). <article-title>Mycobacterial extracellular vesicles and host pathogen interactions.</article-title> <source><italic>Pathog. Dis.</italic></source> <volume>76</volume>:<issue>fty031</issue>. <pub-id pub-id-type="doi">10.1093/femspd/fty031</pub-id> <pub-id pub-id-type="pmid">29722822</pub-id></citation></ref>
<ref id="B39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gurevich</surname> <given-names>A.</given-names></name> <name><surname>Saveliev</surname> <given-names>V.</given-names></name> <name><surname>Vyahhi</surname> <given-names>N.</given-names></name> <name><surname>Tesler</surname> <given-names>G.</given-names></name></person-group> (<year>2013</year>). <article-title>QUAST: Quality assessment tool for genome assemblies.</article-title> <source><italic>Bioinformatics</italic></source> <volume>29</volume> <fpage>1072</fpage>&#x2013;<lpage>1075</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btt086</pub-id> <pub-id pub-id-type="pmid">23422339</pub-id></citation></ref>
<ref id="B40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Holt</surname> <given-names>K.</given-names></name> <name><surname>McAdam</surname> <given-names>P.</given-names></name> <name><surname>Thai</surname> <given-names>P.</given-names></name> <name><surname>Thuong</surname> <given-names>N.</given-names></name> <name><surname>Ha</surname> <given-names>D.</given-names></name> <name><surname>Lan</surname> <given-names>N.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>Frequent transmission of the <italic>Mycobacterium tuberculosis</italic> Beijing lineage and positive selection for the EsxW Beijing variant in Vietnam.</article-title> <source><italic>Nat. Genet.</italic></source> <volume>50</volume> <fpage>849</fpage>&#x2013;<lpage>856</lpage>. <pub-id pub-id-type="doi">10.1038/s41588-018-0117-9</pub-id> <pub-id pub-id-type="pmid">29785015</pub-id></citation></ref>
<ref id="B41"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hyatt</surname> <given-names>D.</given-names></name> <name><surname>Chen</surname> <given-names>G.</given-names></name> <name><surname>LoCascio</surname> <given-names>P.</given-names></name> <name><surname>Land</surname> <given-names>M.</given-names></name> <name><surname>Larimer</surname> <given-names>F.</given-names></name> <name><surname>Hauser</surname> <given-names>L.</given-names></name></person-group> (<year>2010</year>). <article-title>Prodigal: Prokaryotic gene recognition and translation initiation site identification.</article-title> <source><italic>BMC Bioinformatics</italic></source> <volume>11</volume>:<issue>119</issue>. <pub-id pub-id-type="doi">10.1186/1471-2105-11-119</pub-id> <pub-id pub-id-type="pmid">20211023</pub-id></citation></ref>
<ref id="B42"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jou</surname> <given-names>R.</given-names></name> <name><surname>Chiang</surname> <given-names>C.</given-names></name> <name><surname>Huang</surname> <given-names>W.</given-names></name></person-group> (<year>2005</year>). <article-title>Distribution of the Beijing family genotypes of <italic>Mycobacterium tuberculosis</italic> in Taiwan.</article-title> <source><italic>J Clin Microbiol.</italic></source> <volume>43</volume> <fpage>95</fpage>&#x2013;<lpage>100</lpage>. <pub-id pub-id-type="doi">10.1128/JCM.43.1.95-100.2005</pub-id> <pub-id pub-id-type="pmid">15634956</pub-id></citation></ref>
<ref id="B43"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kanehisa</surname> <given-names>M.</given-names></name> <name><surname>Goto</surname> <given-names>S.</given-names></name></person-group> (<year>2000</year>). <article-title>KEGG: Kyoto encyclopedia of genes and genomes.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>28</volume> <fpage>27</fpage>&#x2013;<lpage>30</lpage>. <pub-id pub-id-type="doi">10.1093/nar/28.1.27</pub-id> <pub-id pub-id-type="pmid">10592173</pub-id></citation></ref>
<ref id="B44"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kolbe</surname> <given-names>K.</given-names></name> <name><surname>Veleti</surname> <given-names>S.</given-names></name> <name><surname>Reiling</surname> <given-names>N.</given-names></name> <name><surname>Lindhorst</surname> <given-names>T.</given-names></name></person-group> (<year>2019</year>). <article-title>Lectins of <italic>Mycobacterium tuberculosis</italic> &#x2013; rarely studied proteins.</article-title> <source><italic>Beilstein J. Org. Chem.</italic></source> <volume>15</volume> <fpage>1</fpage>&#x2013;<lpage>15</lpage>. <pub-id pub-id-type="doi">10.3762/bjoc.15.1</pub-id> <pub-id pub-id-type="pmid">30680034</pub-id></citation></ref>
<ref id="B45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kristensen</surname> <given-names>D.</given-names></name> <name><surname>Kannan</surname> <given-names>L.</given-names></name> <name><surname>Coleman</surname> <given-names>M.</given-names></name> <name><surname>Wolf</surname> <given-names>Y.</given-names></name> <name><surname>Sorokin</surname> <given-names>A.</given-names></name> <name><surname>Koonin</surname> <given-names>E.</given-names></name><etal/></person-group> (<year>2010</year>). <article-title>A low-polynomial algorithm for assembling clusters of orthologous groups from intergenomic symmetric best matches.</article-title> <source><italic>Bioinformatics</italic></source> <volume>26</volume> <fpage>1481</fpage>&#x2013;<lpage>1487</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btq229</pub-id> <pub-id pub-id-type="pmid">20439257</pub-id></citation></ref>
<ref id="B46"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Leach&#x00E9;</surname> <given-names>A. D.</given-names></name> <name><surname>Oaks</surname> <given-names>J. R.</given-names></name></person-group> (<year>2017</year>). <article-title>The utility of single nucleotide polymorphism (SNP) data in phylogenetics</article-title>. <source><italic>Ann. Rev. Ecol. Evol. Syst.</italic></source> <volume>48</volume>, <fpage>69</fpage>&#x2013;<lpage>84</lpage>. <pub-id pub-id-type="doi">10.1146/annurev-ecolsys-110316-022645</pub-id></citation></ref>
<ref id="B47"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Medini</surname> <given-names>D.</given-names></name> <name><surname>Donati</surname> <given-names>C.</given-names></name> <name><surname>Tettelin</surname> <given-names>H.</given-names></name> <name><surname>Masignani</surname> <given-names>V.</given-names></name> <name><surname>Rappuoli</surname> <given-names>R.</given-names></name></person-group> (<year>2005</year>). <article-title>The microbial pan-genome.</article-title> <source><italic>Curr. Opin. Genet. Dev.</italic></source> <volume>15</volume> <fpage>589</fpage>&#x2013;<lpage>594</lpage>. <pub-id pub-id-type="doi">10.1016/j.gde.2005.09.006</pub-id> <pub-id pub-id-type="pmid">16185861</pub-id></citation></ref>
<ref id="B48"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Melly</surname> <given-names>G.</given-names></name> <name><surname>Purdy</surname> <given-names>G.</given-names></name></person-group> (<year>2019</year>). <article-title>MmpL proteins in physiology and pathogenesis of <italic>M. tuberculosis</italic>.</article-title> <source><italic>Microorganisms</italic></source> <volume>7</volume>:<issue>70</issue>. <pub-id pub-id-type="doi">10.3390/microorganisms7030070</pub-id> <pub-id pub-id-type="pmid">30841535</pub-id></citation></ref>
<ref id="B49"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Merker</surname> <given-names>M.</given-names></name> <name><surname>Blin</surname> <given-names>C.</given-names></name> <name><surname>Mona</surname> <given-names>S.</given-names></name> <name><surname>Duforet-Frebourg</surname> <given-names>N.</given-names></name> <name><surname>Lecher</surname> <given-names>S.</given-names></name> <name><surname>Willery</surname> <given-names>E.</given-names></name><etal/></person-group> (<year>2015</year>). <article-title>Evolutionary history and global spread of the <italic>Mycobacterium tuberculosis</italic> Beijing lineage.</article-title> <source><italic>Nat. Genet.</italic></source> <volume>47</volume> <fpage>242</fpage>&#x2013;<lpage>249</lpage>. <pub-id pub-id-type="doi">10.1038/ng.3195</pub-id> <pub-id pub-id-type="pmid">25599400</pub-id></citation></ref>
<ref id="B50"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mukhopadhyay</surname> <given-names>S.</given-names></name> <name><surname>Balaji</surname> <given-names>K.</given-names></name></person-group> (<year>2011</year>). <article-title>The PE and PPE proteins of <italic>Mycobacterium tuberculosis</italic>.</article-title> <source><italic>Tuberculosis.</italic></source> <volume>91</volume> <fpage>441</fpage>&#x2013;<lpage>447</lpage>. <pub-id pub-id-type="doi">10.1016/j.tube.2011.04.004</pub-id> <pub-id pub-id-type="pmid">21527209</pub-id></citation></ref>
<ref id="B51"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ngabonziza</surname> <given-names>J.</given-names></name> <name><surname>Loiseau</surname> <given-names>C.</given-names></name> <name><surname>Marceau</surname> <given-names>M.</given-names></name> <name><surname>Jouet</surname> <given-names>A.</given-names></name> <name><surname>Menardo</surname> <given-names>F.</given-names></name> <name><surname>Tzfadia</surname> <given-names>O.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>A sister lineage of the <italic>Mycobacterium tuberculosis</italic> complex discovered in the African Great Lakes region.</article-title> <source><italic>Nat. Commun.</italic></source> <volume>11</volume>:<issue>2917</issue>. <pub-id pub-id-type="doi">10.1038/s41467-020-16626-6</pub-id> <pub-id pub-id-type="pmid">32518235</pub-id></citation></ref>
<ref id="B52"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nguyen</surname> <given-names>L.</given-names></name> <name><surname>Schmidt</surname> <given-names>H.</given-names></name> <name><surname>von Haeseler</surname> <given-names>A.</given-names></name> <name><surname>Minh</surname> <given-names>B.</given-names></name></person-group> (<year>2015</year>). <article-title>IQ-TREE: A fast and effective stochastic algorithm for estimating maximum-likelihood phylogenies.</article-title> <source><italic>Mol. Biol. Evol.</italic></source> <volume>32</volume> <fpage>268</fpage>&#x2013;<lpage>274</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msu300</pub-id> <pub-id pub-id-type="pmid">25371430</pub-id></citation></ref>
<ref id="B53"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Niemann</surname> <given-names>S.</given-names></name> <name><surname>Supply</surname> <given-names>P.</given-names></name></person-group> (<year>2014</year>). <article-title>Diversity and evolution of <italic>Mycobacterium tuberculosis</italic>: Moving to whole-genome-based approaches.</article-title> <source><italic>Cold Spring Harb. Perspect. Med.</italic></source> <volume>4</volume>:<fpage>a021188</fpage>&#x2013;<lpage>a021188</lpage>. <pub-id pub-id-type="doi">10.1101/cshperspect.a021188</pub-id> <pub-id pub-id-type="pmid">25190252</pub-id></citation></ref>
<ref id="B54"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nieto Ramirez</surname> <given-names>L.</given-names></name> <name><surname>Ferro</surname> <given-names>B.</given-names></name> <name><surname>Diaz</surname> <given-names>G.</given-names></name> <name><surname>Anthony</surname> <given-names>R.</given-names></name> <name><surname>de Beer</surname> <given-names>J.</given-names></name> <name><surname>van Soolingen</surname> <given-names>D.</given-names></name></person-group> (<year>2020</year>). <article-title>Genetic profiling of <italic>Mycobacterium tuberculosis</italic> revealed &#x201C;modern&#x201D; Beijing strains linked to MDR-TB from Southwestern Colombia.</article-title> <source><italic>PLoS One.</italic></source> <volume>15</volume>:<issue>e0224908</issue>. <pub-id pub-id-type="doi">10.1371/journal.pone.0224908</pub-id> <pub-id pub-id-type="pmid">32330146</pub-id></citation></ref>
<ref id="B55"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nogueira</surname> <given-names>L.</given-names></name> <name><surname>Cardoso</surname> <given-names>F.</given-names></name> <name><surname>Mattos</surname> <given-names>A.</given-names></name> <name><surname>Bordignon</surname> <given-names>J.</given-names></name> <name><surname>Figueiredo</surname> <given-names>C.</given-names></name> <name><surname>Dahlstrom</surname> <given-names>P.</given-names></name><etal/></person-group> (<year>2010</year>). <article-title><italic>Mycobacterium tuberculosis</italic> Rv1419 encodes a secreted 13 kDa lectin with immunological reactivity during human tuberculosis.</article-title> <source><italic>Eur. J. Immunol.</italic></source> <volume>40</volume> <fpage>744</fpage>&#x2013;<lpage>753</lpage>. <pub-id pub-id-type="doi">10.1002/eji.200939747</pub-id> <pub-id pub-id-type="pmid">20017196</pub-id></citation></ref>
<ref id="B56"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Periwal</surname> <given-names>V.</given-names></name> <name><surname>Patowary</surname> <given-names>A.</given-names></name> <name><surname>Vellarikkal</surname> <given-names>S.</given-names></name> <name><surname>Gupta</surname> <given-names>A.</given-names></name> <name><surname>Singh</surname> <given-names>M.</given-names></name> <name><surname>Mittal</surname> <given-names>A.</given-names></name><etal/></person-group> (<year>2015</year>). <article-title>Comparative whole-genome analysis of clinical isolates reveals characteristic architecture of <italic>Mycobacterium tuberculosis</italic> pangenome.</article-title> <source><italic>PLoS One.</italic></source> <volume>10</volume>:<issue>e0122979</issue>. <pub-id pub-id-type="doi">10.1371/journal.pone.0122979</pub-id> <pub-id pub-id-type="pmid">25853708</pub-id></citation></ref>
<ref id="B57"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ramos</surname> <given-names>L.</given-names></name> <name><surname>Obregon-Henao</surname> <given-names>A.</given-names></name> <name><surname>Henao-Tamayo</surname> <given-names>M.</given-names></name> <name><surname>Bowen</surname> <given-names>R.</given-names></name> <name><surname>Lunney</surname> <given-names>J.</given-names></name> <name><surname>Gonzalez-Juarrero</surname> <given-names>M.</given-names></name></person-group> (<year>2017</year>). <article-title>The minipig as an animal model to study <italic>Mycobacterium tuberculosis</italic> infection and natural transmission.</article-title> <source><italic>Tuberculosis.</italic></source> <volume>106</volume> <fpage>91</fpage>&#x2013;<lpage>98</lpage>. <pub-id pub-id-type="doi">10.1016/j.tube.2017.07.003</pub-id> <pub-id pub-id-type="pmid">28802411</pub-id></citation></ref>
<ref id="B58"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Realpe</surname> <given-names>T.</given-names></name> <name><surname>Correa</surname> <given-names>N.</given-names></name> <name><surname>Rozo</surname> <given-names>J.</given-names></name> <name><surname>Ferro</surname> <given-names>B.</given-names></name> <name><surname>Gomez</surname> <given-names>V.</given-names></name> <name><surname>Zapata</surname> <given-names>E.</given-names></name><etal/></person-group> (<year>2014</year>). <article-title>Population structure among <italic>Mycobacterium tuberculosis</italic> isolates from <italic>Pulmonary Tuberculosis</italic> patients in Colombia.</article-title> <source><italic>PLoS One.</italic></source> <volume>9</volume>:<issue>e93848</issue>. <pub-id pub-id-type="doi">10.1371/journal.pone.0093848</pub-id> <pub-id pub-id-type="pmid">24747767</pub-id></citation></ref>
<ref id="B59"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Reyes</surname> <given-names>A.</given-names></name> <name><surname>Sandoval</surname> <given-names>A.</given-names></name> <name><surname>Cubillos-Ruiz</surname> <given-names>A.</given-names></name> <name><surname>Varley</surname> <given-names>K.</given-names></name> <name><surname>Hern&#x00E1;ndez-Neuta</surname> <given-names>I.</given-names></name> <name><surname>Samper</surname> <given-names>S.</given-names></name><etal/></person-group> (<year>2012</year>). <article-title>IS-seq: A novel high throughput survey of <italic>in vivo</italic> IS6110 transposition in multiple <italic>Mycobacterium tuberculosis</italic> genomes.</article-title> <source><italic>BMC Genom.</italic></source> <volume>13</volume>:<issue>249</issue>. <pub-id pub-id-type="doi">10.1186/1471-2164-13-249</pub-id> <pub-id pub-id-type="pmid">22703188</pub-id></citation></ref>
<ref id="B60"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rouli</surname> <given-names>L.</given-names></name> <name><surname>Merhej</surname> <given-names>V.</given-names></name> <name><surname>Fournier</surname> <given-names>P.</given-names></name> <name><surname>Raoult</surname> <given-names>D.</given-names></name></person-group> (<year>2015</year>). <article-title>The bacterial pangenome as a new tool for analysing pathogenic bacteria.</article-title> <source><italic>New Microbes New Infect.</italic></source> <volume>7</volume> <fpage>72</fpage>&#x2013;<lpage>85</lpage>. <pub-id pub-id-type="doi">10.1016/j.nmni.2015.06.005</pub-id> <pub-id pub-id-type="pmid">26442149</pub-id></citation></ref>
<ref id="B61"><citation citation-type="journal"><collab>Secretar&#x00ED;a Seccional de Salud y Protecci&#x00F3;n Social de Antioquia</collab>. (<year>2017</year>). <source><italic>Situaci&#x00F3;n de la tuberculosis en el departamento de Antioquia 2015-2016.</italic></source> <publisher-loc>Medell&#x00ED;n</publisher-loc>: <publisher-name>Secretar&#x00ED;a Seccional de Salud y Protecci&#x00F3;n Social de Antioquia</publisher-name>.</citation></ref>
<ref id="B62"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Seemann</surname> <given-names>T.</given-names></name></person-group> (<year>2014</year>). <article-title>Prokka: Rapid prokaryotic genome annotation.</article-title> <source><italic>Bioinformatics</italic></source> <volume>30</volume> <fpage>2068</fpage>&#x2013;<lpage>2069</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btu153</pub-id> <pub-id pub-id-type="pmid">24642063</pub-id></citation></ref>
<ref id="B63"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sirakova</surname> <given-names>T.</given-names></name></person-group> (<year>2006</year>). <article-title>Identification of a diacylglycerol acyltransferase gene involved in accumulation of triacylglycerol in <italic>Mycobacterium tuberculosis</italic> under stress.</article-title> <source><italic>Microbiology</italic></source> <volume>152</volume> <fpage>2717</fpage>&#x2013;<lpage>2725</lpage>. <pub-id pub-id-type="doi">10.1099/mic.0.28993-0</pub-id> <pub-id pub-id-type="pmid">16946266</pub-id></citation></ref>
<ref id="B64"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sutcliffe</surname> <given-names>I.</given-names></name> <name><surname>Harrington</surname> <given-names>D.</given-names></name></person-group> (<year>2004</year>). <article-title>Lipoproteins of <italic>Mycobacterium tuberculosis</italic>?: An abundant and functionally diverse class of cell envelope components.</article-title> <source><italic>FEMS Microbiol Rev.</italic></source> <volume>28</volume> <fpage>645</fpage>&#x2013;<lpage>659</lpage>. <pub-id pub-id-type="doi">10.1016/j.femsre.2004.06.002</pub-id> <pub-id pub-id-type="pmid">15539077</pub-id></citation></ref>
<ref id="B65"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tatusov</surname> <given-names>R.</given-names></name> <name><surname>Galperin</surname> <given-names>M.</given-names></name> <name><surname>Natale</surname> <given-names>D.</given-names></name> <name><surname>Koonin</surname> <given-names>E.</given-names></name></person-group> (<year>2000</year>). <article-title>The COG database: A tool for genome-scale analysis of protein functions and evolution.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>28</volume> <fpage>33</fpage>&#x2013;<lpage>36</lpage>. <pub-id pub-id-type="doi">10.1093/nar/28.1.33</pub-id> <pub-id pub-id-type="pmid">10592175</pub-id></citation></ref>
<ref id="B66"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tettelin</surname> <given-names>H.</given-names></name> <name><surname>Masignani</surname> <given-names>V.</given-names></name> <name><surname>Cieslewicz</surname> <given-names>M.</given-names></name> <name><surname>Donati</surname> <given-names>C.</given-names></name> <name><surname>Medini</surname> <given-names>D.</given-names></name> <name><surname>Ward</surname> <given-names>N.</given-names></name><etal/></person-group> (<year>2005</year>). <article-title>Genome analysis of multiple pathogenic isolates of <italic>Streptococcus</italic> agalactiae: Implications for the microbial &#x201C;pan-genome.&#x201D;.</article-title> <source><italic>Proc. Natl. Acad. Sci. U.S.A.</italic></source> <volume>102</volume> <fpage>13950</fpage>&#x2013;<lpage>13955</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.0506758102</pub-id> <pub-id pub-id-type="pmid">16172379</pub-id></citation></ref>
<ref id="B67"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tettelin</surname> <given-names>H.</given-names></name> <name><surname>Riley</surname> <given-names>D.</given-names></name> <name><surname>Cattuto</surname> <given-names>C.</given-names></name> <name><surname>Medini</surname> <given-names>D.</given-names></name></person-group> (<year>2008</year>). <article-title>Comparative genomics: The bacterial pan-genome.</article-title> <source><italic>Curr. Opin. Microbiol.</italic></source> <volume>11</volume> <fpage>472</fpage>&#x2013;<lpage>477</lpage>. <pub-id pub-id-type="doi">10.1016/j.mib.2008.09.006</pub-id> <pub-id pub-id-type="pmid">19086349</pub-id></citation></ref>
<ref id="B68"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tian</surname> <given-names>J.</given-names></name> <name><surname>Bryk</surname> <given-names>R.</given-names></name> <name><surname>Itoh</surname> <given-names>M.</given-names></name> <name><surname>Suematsu</surname> <given-names>M.</given-names></name> <name><surname>Nathan</surname> <given-names>C.</given-names></name></person-group> (<year>2005</year>). <article-title>Variant tricarboxylic acid cycle in <italic>Mycobacterium tuberculosis</italic>: Identification of -ketoglutarate decarboxylase.</article-title> <source><italic>Proc. Natl. Acad. Sci. U.S.A.</italic></source> <volume>102</volume> <fpage>10670</fpage>&#x2013;<lpage>10675</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.0501605102</pub-id> <pub-id pub-id-type="pmid">16027371</pub-id></citation></ref>
<ref id="B69"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Uchiya</surname> <given-names>K.</given-names></name> <name><surname>Tomida</surname> <given-names>S.</given-names></name> <name><surname>Nakagawa</surname> <given-names>T.</given-names></name> <name><surname>Asahi</surname> <given-names>S.</given-names></name> <name><surname>Nikai</surname> <given-names>T.</given-names></name> <name><surname>Ogawa</surname> <given-names>K.</given-names></name></person-group> (<year>2017</year>). <article-title>Comparative genome analyses of <italic>Mycobacterium avium</italic> reveal genomic features of its subspecies and strains that cause progression of pulmonary disease.</article-title> <source><italic>Sci. Rep.</italic></source> <volume>7</volume>:<issue>39750</issue>. <pub-id pub-id-type="doi">10.1038/srep39750</pub-id> <pub-id pub-id-type="pmid">28045086</pub-id></citation></ref>
<ref id="B70"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>van Soolingen</surname> <given-names>D.</given-names></name> <name><surname>de Haas</surname> <given-names>P.</given-names></name> <name><surname>Hermans</surname> <given-names>P.</given-names></name> <name><surname>van Embden</surname> <given-names>J.</given-names></name></person-group> (<year>1994</year>). <article-title>DNA fingerprinting of <italic>Mycobacterium tuberculosis</italic>.</article-title> <source><italic>Meth Enzymol.</italic></source> <volume>235</volume> <fpage>196</fpage>&#x2013;<lpage>205</lpage>. <pub-id pub-id-type="doi">10.1016/0076-6879(94)35141-4</pub-id></citation></ref>
<ref id="B71"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vernikos</surname> <given-names>G.</given-names></name> <name><surname>Medini</surname> <given-names>D.</given-names></name> <name><surname>Riley</surname> <given-names>D.</given-names></name> <name><surname>Tettelin</surname> <given-names>H.</given-names></name></person-group> (<year>2015</year>). <article-title>Ten years of pan-genome analyses.</article-title> <source><italic>Curr. Opin. Microbiol.</italic></source> <volume>23</volume> <fpage>148</fpage>&#x2013;<lpage>154</lpage>. <pub-id pub-id-type="doi">10.1016/j.mib.2014.11.016</pub-id> <pub-id pub-id-type="pmid">25483351</pub-id></citation></ref>
<ref id="B72"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Walker</surname> <given-names>B.</given-names></name> <name><surname>Abeel</surname> <given-names>T.</given-names></name> <name><surname>Shea</surname> <given-names>T.</given-names></name> <name><surname>Priest</surname> <given-names>M.</given-names></name> <name><surname>Abouelliel</surname> <given-names>A.</given-names></name> <name><surname>Sakthikumar</surname> <given-names>S.</given-names></name><etal/></person-group> (<year>2014</year>). <article-title>Pilon: An integrated tool for comprehensive microbial variant detection and genome assembly improvement.</article-title> <source><italic>PLoS One.</italic></source> <volume>9</volume>:<issue>e112963</issue>. <pub-id pub-id-type="doi">10.1371/journal.pone.0112963</pub-id> <pub-id pub-id-type="pmid">25409509</pub-id></citation></ref>
<ref id="B73"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Woodman</surname> <given-names>M.</given-names></name> <name><surname>Haeusler</surname> <given-names>I.</given-names></name> <name><surname>Grandjean</surname> <given-names>L.</given-names></name></person-group> (<year>2019</year>). <article-title>Tuberculosis genetic epidemiology: A latin American perspective.</article-title> <source><italic>Genes</italic></source> <volume>10</volume>:<issue>53</issue>. <pub-id pub-id-type="doi">10.3390/genes10010053</pub-id> <pub-id pub-id-type="pmid">30654542</pub-id></citation></ref>
<ref id="B74"><citation citation-type="journal"><collab>World Health Organization [WHO]</collab> (<year>2018</year>). <source><italic>Licence CC BY-NC-SA 3.0 IGO. Global tuberculosis report 2018.</italic></source> <publisher-loc>Geneva</publisher-loc>: <publisher-name>World Health Organization</publisher-name>.</citation></ref>
<ref id="B75"><citation citation-type="journal"><collab>World Health Organization [WHO]</collab> (<year>2022</year>). <source><italic>Licence CC BY-NC-SA 3.0 IGO. Global tuberculosis report 2022.</italic></source> <publisher-loc>Geneva</publisher-loc>: <publisher-name>World Health Organization</publisher-name>.</citation></ref>
<ref id="B76"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>T.</given-names></name> <name><surname>Zhong</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>C.</given-names></name> <name><surname>Yu</surname> <given-names>X.</given-names></name> <name><surname>Xiao</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>Pan-genomic study of <italic>Mycobacterium tuberculosis</italic> reflecting the primary/secondary genes, generality/individuality, and the interconversion through copy number variations.</article-title> <source><italic>Front. Microbiol.</italic></source> <volume>9</volume>:<issue>1886</issue>. <pub-id pub-id-type="doi">10.3389/fmicb.2018.01886</pub-id> <pub-id pub-id-type="pmid">30177918</pub-id></citation></ref>
<ref id="B77"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zakham</surname> <given-names>F.</given-names></name> <name><surname>Sironen</surname> <given-names>T.</given-names></name> <name><surname>Vapalahti</surname> <given-names>O.</given-names></name> <name><surname>Kant</surname> <given-names>R.</given-names></name></person-group> (<year>2021</year>). <article-title>Pan and core genome analysis of 183 <italic>Mycobacterium tuberculosis</italic> strains revealed a high inter-species diversity among the human adapted strains.</article-title> <source><italic>Antibiotics</italic></source> <volume>10</volume>:<issue>500</issue>. <pub-id pub-id-type="doi">10.3390/antibiotics10050500</pub-id> <pub-id pub-id-type="pmid">33924811</pub-id></citation></ref>
</ref-list>
</back>
</article>