<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Microbiol.</journal-id>
<journal-title>Frontiers in Microbiology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Microbiol.</abbrev-journal-title>
<issn pub-type="epub">1664-302X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmicb.2022.787856</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Microbiology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Selective sweep sites and SNP dense regions differentiate <italic>Mycobacterium bovis</italic> isolates across scales</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author"><name><surname>Legall</surname><given-names>Noah</given-names></name><xref rid="aff1" ref-type="aff"><sup>1</sup></xref><xref rid="aff2" ref-type="aff"><sup>2</sup></xref><xref rid="aff3" ref-type="aff"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author" corresp="yes"><name><surname>Salvador</surname><given-names>Liliana C. M.</given-names></name><xref rid="aff2" ref-type="aff"><sup>2</sup></xref><xref rid="aff3" ref-type="aff"><sup>3</sup></xref><xref rid="aff4" ref-type="aff"><sup>4</sup></xref><xref rid="c001" ref-type="corresp"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/847579/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Interdisciplinary Disease Ecology Across Scales Research Traineeship Program, University of Georgia</institution>, <addr-line>Athens, GA</addr-line>, <country>United States</country></aff>
<aff id="aff2"><sup>2</sup><institution>Institute of Bioinformatics, University of Georgia</institution>, <addr-line>Athens, GA</addr-line>, <country>United States</country></aff>
<aff id="aff3"><sup>3</sup><institution>Center for the Ecology of Infectious Diseases, University of Georgia</institution>, <addr-line>Athens, GA</addr-line>, <country>United States</country></aff>
<aff id="aff4"><sup>4</sup><institution>Department of Infectious Diseases, College of Veterinary Medicine, University of Georgia</institution>, <addr-line>Athens, GA</addr-line>, <country>United States</country></aff>
<author-notes>
<fn id="fn0001" fn-type="edited-by">
<p>Edited by: Marian Louise Price-Carter, AgResearch Ltd., New Zealand</p>
</fn>
<fn id="fn0002" fn-type="edited-by">
<p>Reviewed by: Bo Zhu, Shanghai Jiao Tong University, China; Maxime Godfroid, Coll&#x00E8;ge de France, France; I&#x00F1;aki Comas, Institute of Biomedicine of Valencia (CSIC), Spain</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Liliana C. M. Salvador, <email>salvador@uga.edu</email>
</corresp>
<fn id="fn0003" fn-type="other">
<p>This article was submitted to Evolutionary and Genomic Microbiology, a section of the journal Frontiers in Microbiology</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>07</day>
<month>09</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>787856</elocation-id>
<history>
<date date-type="received">
<day>01</day>
<month>10</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>08</day>
<month>08</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2022 Legall and Salvador.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Legall and Salvador</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p><italic>Mycobacterium bovis</italic>, a bacterial zoonotic pathogen responsible for the economically and agriculturally important livestock disease bovine tuberculosis (bTB), infects a broad mammalian host range worldwide. This characteristic has led to bidirectional transmission events between livestock and wildlife species as well as the formation of wildlife reservoirs, impacting the success of bTB control measures. Next Generation Sequencing (NGS) has transformed our ability to understand disease transmission events by tracking variant sites, however the genomic signatures related to host adaptation following spillover, alongside the role of other genomic factors in the <italic>M. bovis</italic> transmission process are understudied problems. We analyzed publicly available <italic>M. bovis</italic> datasets collected from 700 hosts across three countries with bTB endemic regions (United Kingdom, United States, and New Zealand) to investigate if genomic regions with high SNP density and/or selective sweep sites play a role in <italic>Mycobacterium bovis</italic> adaptation to new environments (e.g., at the host-species, geographical, and/or sub-population levels). A simulated <italic>M. bovis</italic> alignment was created to generate null distributions for defining genomic regions with high SNP counts and regions with selective sweeps evidence. Random Forest (RF) models were used to investigate evolutionary metrics within the genomic regions of interest to determine which genomic processes were the best for classifying <italic>M. bovis</italic> across ecological scales. We identified in the <italic>M.</italic> bovis genomes 14 and 132 high SNP density and selective sweep regions, respectively. Selective sweep regions were ranked as the most important in classifying <italic>M. bovis</italic> across the different scales in all RF models. SNP dense regions were found to have high importance in the badger and cattle specific RF models in classifying badger derived isolates from livestock derived ones. Additionally, the genes detected within these genomic regions harbor various pathogenic functions such as virulence and immunogenicity, membrane structure, host survival, and mycobactin production. The results of this study demonstrate how comparative genomics alongside machine learning approaches are useful to investigate further the nature of <italic>M. bovis</italic> host-pathogen interactions.</p>
</abstract>
<kwd-group>
<kwd>comparative genomics</kwd>
<kwd><italic>Mycobacterium bovis</italic></kwd>
<kwd>host range</kwd>
<kwd>geographic location</kwd>
<kwd>population clusters</kwd>
<kwd>selective sweeps</kwd>
<kwd>SNP dense regions</kwd>
<kwd>ecological scales</kwd>
</kwd-group>
<contract-num rid="cn1">DGE-1545433 501</contract-num>
<contract-sponsor id="cn1">National Science Foundation<named-content content-type="fundref-id">10.13039/501100008982</named-content>
</contract-sponsor>
<contract-sponsor id="cn2">University of Georgia Office of Research</contract-sponsor>
<counts>
<fig-count count="8"/>
<table-count count="1"/>
<equation-count count="0"/>
<ref-count count="61"/>
<page-count count="14"/>
<word-count count="9963"/>
</counts>
</article-meta>
</front>
<body>
<sec id="sec1" sec-type="intro">
<title>Introduction</title>
<p>Bovine tuberculosis (bTB) is a livestock disease caused by the transmission of the bacterial pathogen <italic>Mycobacterium bovis</italic> (<italic>M. bovis</italic>), and has wide ranging impacts on agriculture, economics, and human health (<xref ref-type="bibr" rid="ref40">Palmer et al., 2012</xref>; <xref ref-type="bibr" rid="ref39">Palmer, 2013</xref>). <italic>M. bovis</italic> is a member of the <italic>Mycobacterium tuberculosis</italic> complex (MTBC), which is a genetically related group (99% similarity) of Mycobacterium species that can cause tuberculosis within vertebrate host-species, including humans (<xref ref-type="bibr" rid="ref8">Brosch et al., 2002</xref>). <italic>Mycobacterium tuberculosis</italic> is the main causative agent of human associated tuberculosis, while other lineages are defined in vertebrate hosts such as <italic>M. africanum</italic>, <italic>M. canettii</italic>, <italic>M. microti</italic>, <italic>M. bovis</italic>, <italic>M. caprae</italic>, <italic>M. pinnipedii</italic>, <italic>M. mungi</italic>, <italic>M. orygis</italic>, and <italic>M. suricattae</italic> (<xref ref-type="bibr" rid="ref41">Patan&#x00E9; et al., 2017</xref>; <xref ref-type="bibr" rid="ref28">Lekko et al., 2020</xref>). The wide range of MTBC members is hypothesized to be mainly due to the divergence of certain lineages from an ancestral species <italic>Mycobacterium canettii</italic> through multiple insertion or deletion mutations (<xref ref-type="bibr" rid="ref20">Gutierrez et al., 2005</xref>; <xref ref-type="bibr" rid="ref41">Patan&#x00E9; et al., 2017</xref>).</p>
<p><italic>Mycobacterium bovis</italic> has been identified as having the widest mammalian host range to date compared to other members of the MTBC (<xref ref-type="bibr" rid="ref39">Palmer, 2013</xref>). This allows <italic>M. bovis</italic> to transmit between species much more easily, such that if close spatial proximity between wild and domestic animals can occur through direct contact (infected individuals and/or carcasses) or indirect contact (contaminated soil or water resources), frequent spillover events to and/or from wildlife can occur (<xref ref-type="bibr" rid="ref39">Palmer, 2013</xref>; <xref ref-type="bibr" rid="ref19">Gormley and Corner, 2018</xref>). Certain regions in the United States of America (USA), New Zealand (NZ), and the United Kingdom (UK) are endemic for bTB, in which certain wildlife species have transitioned from novel spillover hosts to reservoirs of infection, defined as populations that can maintain a pathogen and transmit it to a target population (<xref ref-type="bibr" rid="ref22">Haydon et al., 2002</xref>; <xref ref-type="bibr" rid="ref58">Viana et al., 2014</xref>; <xref ref-type="bibr" rid="ref21">Hallmaier-Wacker et al., 2017</xref>). The consistent contact between a wildlife reservoir and livestock can lead to frequent spillback of <italic>M. bovis</italic> into livestock populations, jeopardizing existing control measures to mitigate the disease. These dynamics exacerbate the agricultural and economic impacts bTB has on society, with contemporary estimates of 50 million cattle worldwide infected and costing farmers $3 billion annually (<xref ref-type="bibr" rid="ref39">Palmer, 2013</xref>). However, only a few species have successfully transitioned from novel spillover host to reservoir of infection, suggesting that there exist certain factors that limit the ability of <italic>M. bovis</italic> to adapt to specific hosts. Understanding how these fundamental processes occur in order for hosts to transition from spillover to adapted host populations would contribute to our understanding of pathogen-host interactions and inform disease control measures in order to mitigate the impacts of <italic>M. bovis</italic> transmission.</p>
<p>The use of next-generation sequencing (NGS) to study pathogen genomics has revolutionized the field of <italic>M. bovis</italic> molecular epidemiology. Detection of genomic variations among multiple samples have allowed researchers to characterize <italic>M. bovis</italic> population structure (<xref ref-type="bibr" rid="ref36">M&#x00FC;ller et al., 2009</xref>; <xref ref-type="bibr" rid="ref4">Berg et al., 2011</xref>; <xref ref-type="bibr" rid="ref51">Smith et al., 2011</xref>; <xref ref-type="bibr" rid="ref46">Rodriguez-Campos et al., 2012</xref>; <xref ref-type="bibr" rid="ref44">Reis and Cunha, 2021</xref>; <xref ref-type="bibr" rid="ref45">Rodrigues et al., 2021</xref>), sources and chains of transmission (<xref ref-type="bibr" rid="ref32">McCluskey et al., 2014</xref>; <xref ref-type="bibr" rid="ref34">Milian-Suazo et al., 2016</xref>; <xref ref-type="bibr" rid="ref37">Orloski et al., 2018</xref>; <xref ref-type="bibr" rid="ref30">Loiseau et al., 2020</xref>; <xref ref-type="bibr" rid="ref59">Zimpel et al., 2020</xref>; <xref ref-type="bibr" rid="ref55">Tonder et al., 2021</xref>; <xref ref-type="bibr" rid="ref47">Rossi et al., 2022</xref>), and the role of host-species in the transmission process (<xref ref-type="bibr" rid="ref5">Biek et al., 2012</xref>; <xref ref-type="bibr" rid="ref12">Crispell et al., 2017</xref>, <xref ref-type="bibr" rid="ref10">2019</xref>, <xref ref-type="bibr" rid="ref11">2020</xref>; <xref ref-type="bibr" rid="ref48">Salvador et al., 2019</xref>; <xref ref-type="bibr" rid="ref1">Akhmetova et al., 2021</xref>). Authors from <xref ref-type="bibr" rid="ref17">Fuente et al. (2015)</xref> studied how single nucleotide polymorphisms (SNPs) and gene presence/absence contributed to lesion scores on three <italic>M. bovis</italic> isolates in an attempt to associate genomic changes with phenotype differences in <italic>M. bovis</italic>. The results highlighted that there were key differences between the presence and absence of genes that were associated with host&#x2013;pathogen interactions and resulting virulence among hosts. However, it is still unclear if there are <italic>M. bovis</italic> genomic variations across different environments (e.g., host-associated populations of <italic>M. bovis</italic> or geographical locations) and if they play a specific role in the adaptation process to those environments. Genomic analyses like these would unravel specific evolutionary forces responsible for <italic>M. bovis</italic> adaptation to particular environments (<xref ref-type="bibr" rid="ref3">Allen, 2017</xref>; <xref ref-type="bibr" rid="ref41">Patan&#x00E9; et al., 2017</xref>; <xref ref-type="bibr" rid="ref50">Sheppard et al., 2018</xref>) and improve our understanding of <italic>M. bovis</italic> evolution across different ecological scales.</p>
<p>One avenue to explore the genomic factors that impact <italic>M. bovis</italic> adaptation is to investigate genomic regions of interest that potentially harbor signatures of <italic>M. bovis</italic> evolution. For instance, within different biological populations of organisms, a highly beneficial mutation that is introduced into the population can quickly become the most common allele, leading to a lack of variation at that location (<xref ref-type="bibr" rid="ref52">Stephan, 2019</xref>). The site of this beneficial mutation can be in linkage disequilibrium with neutral flanking SNPs, meaning that the variation at these neighboring SNP sites would be non-random and correlated to the variation up and downstream of a beneficial site (<xref ref-type="bibr" rid="ref2">Alachiotis et al., 2012</xref>). This signature, defined as a &#x2018;selective sweep&#x2019; can highlight important genes linked with <italic>M. bovis</italic> adaptation to new environments. Additionally, bacterial genomes can possess regions with a higher number of mutations than expected if analyzed genome wide. Regions of high SNP density might indicate sites of elevated mutation rates that provide short term fitness advantages when adapting to a new ecological niche. This phenomena, often defined as hypermutation, is common in pathogenic bacteria that must adapt to newly encountered stressors such as antibiotic resistance and novel hosts (<xref ref-type="bibr" rid="ref53">Swings et al., 2017</xref>; <xref ref-type="bibr" rid="ref33">Mehta et al., 2019</xref>). A deeper investigation into <italic>M. bovis</italic> genomic regions from isolates extracted from multiple hosts and geographic locations can provide an opportunity to determine <italic>M. bovis</italic> evolutionary processes that might confer adaptive advantages. In this study, we conduct a comparative genomic study of <italic>M. bovis</italic> isolates collected from livestock and wildlife host-species from three bTB endemic regions around the world. We use WGS data to create a high-resolution genomic dataset with spatial and host phenotypic information. Our objectives for this study are to: (a) identify <italic>M. bovis</italic> genomic signatures; (b) investigate the relationship of these genomic signatures across different scales (geographical, host-species and sub-populations scales); and (c) determine sets of genes associated with the <italic>M. bovis</italic> genomic regions of interest and/or the different scales.</p>
</sec>
<sec id="sec2" sec-type="materials|methods">
<title>Materials and methods</title>
<sec id="sec3">
<title>Data description</title>
<p>In this study, we downloaded a total of 700 <italic>Mycobacterium bovis</italic> whole-genomes from the National Center for Biotechnology Information (NCBI) Sequence Read Archive (SRA). These isolates originated from three previous studies focusing on <italic>M. bovis</italic> evolutionary dynamics and cross-species transmission in NZ (PRJNA363037; <xref ref-type="bibr" rid="ref12">Crispell et al., 2017</xref>), UK&#x2014;Woodchester Park (PRJNA523164; <xref ref-type="bibr" rid="ref10">Crispell et al., 2019</xref>), and in the USA&#x2014;Michigan (PRJNA251692; <xref ref-type="bibr" rid="ref48">Salvador et al., 2019</xref>), respectively. Metadata associated with these isolates included geographic location and host species from which <italic>M. bovis</italic> samples were extracted (11 in total): six in NZ [stoat, <italic>Mustela erminea</italic>; porcine, <italic>Sus scrofa</italic>; cervine (various NZ cervine species unknown); cattle, <italic>Bos taurus</italic>; brushtail possum, <italic>Trichosurus vulpecula</italic>; and ferret, <italic>Mustela furo</italic>], two in the UK (cattle, <italic>Bos taurus</italic> and the Eurasian badger, <italic>Meles meles</italic>), and three in the USA (white-tailed deer, <italic>Odocoileus virginianus</italic>; elk, <italic>Cervus canadensis</italic>; and cattle, <italic>Bos taurus</italic>). Although there was only one <italic>M. bovis</italic> isolate from a stoat host, it was kept in the dataset to assist in further analyses of geographic and subpopulation structure. Bioinformatic statistics associated with the isolates are shown in <xref ref-type="supplementary-material" rid="SM1">Supplementary Table 1</xref>.</p>
</sec>
<sec id="sec4">
<title>Data processing</title>
<p>To improve the quality from the paired-end sequences, we used fastp v0.22 (<xref ref-type="bibr" rid="ref9">Chen et al., 2018</xref>) to remove adapter sequences and low-quality ends. We used a sliding window approach (size 4&#x2009;bp) to trim reads with window average quality below 30. Additionally, we filtered out all reads less than 15&#x2009;bp. Once all the reads were processed, we mapped them against the <italic>M. bovis</italic> reference genome AF2122/97 (NC_002945.4, Genbank accession code PRJNA89) using the Burrows-Wheeler Aligner (BWA) software (<xref ref-type="bibr" rid="ref500">Li and Durbin, 2009</xref>). We removed duplicated reads using Picard v2.0.1 (<xref ref-type="bibr" rid="ref7">Broadinstitute/Picard, 2021</xref>/2014) to limit the impact of non-unique or erroneous reads on the SNP calling process. We performed Variant calling using Freebayes v1.3.5 (<xref ref-type="bibr" rid="ref18">Garrison and Marth, 2012</xref>), and we kept detected SNPs for the downstream analysis if they possessed (i) a phred-quality score (QUAL) above 20, (ii) a mapping quality (MQ) above 59, (iii) did not fall within gene regions that coded for <italic>pe, pe/PGRS and ppe</italic> genes (a family of redundant sequence genes that are difficult to work with <italic>in-silico</italic>; <xref ref-type="bibr" rid="ref49">Sampson, 2011</xref>; <xref ref-type="bibr" rid="ref15">Delogu et al., 2017</xref>), and (iv) were more than 500&#x2009;bp from insertion or deletion mutation (INDEL) regions (this is an extra measure to reduce impacts in the alignment that can lead to false positive detections). Furthermore, the isolates used in this study were sequenced either on a HiSeq or MiseqIllumina platform, and we compared specific bioinformatic metrics (read depth, number of SNPs, SNP depth, and bp depth) between isolates to check if the choice of sequencing platform played a role in SNP detection and quality. A SNP alignment was created using BCFtools v1.13 (<xref ref-type="bibr" rid="ref29">Li, 2011</xref>) &#x2018;consensus&#x2019; command and snp-sites v2.5.1 (<xref ref-type="bibr" rid="ref38">Page et al., 2016</xref>) to integrate the detected SNPs into a copy of the <italic>M. bovis</italic> reference genome.</p>
</sec>
<sec id="sec5">
<title>Phylogenetic inference</title>
<p>Evolutionary relationships among <italic>M. bovis</italic> isolates were investigated using the Maximum Likelihood software IQtree v2.1.4 (<xref ref-type="bibr" rid="ref35">Minh et al., 2020</xref>) with the <italic>M. bovis</italic> SNP alignment used as input. IQTree uses the ModelFinder approach to determine the nucleotide substitution model that best fits the data based on the Bayesian Information Criteria (BIC) comparison between substitution models (<xref ref-type="bibr" rid="ref24">Kalyaanamoorthy et al., 2017</xref>). After inferring the substitution model, IQtree ran 1,000 bootstrap iterations in order to provide a measure of internal node support ranging from 0 to 100 (<xref ref-type="bibr" rid="ref23">Hoang et al., 2018</xref>).</p>
<p>To determine if the existence of potential regions with elevated densities of base substitutions in the <italic>M. bovis</italic> genomes has any effect on the resolution of their phylogeny and nodal support, we compared a phylogeny produced with the original SNP alignment data with the one generated by the Gubbins software (<xref ref-type="bibr" rid="ref13">Croucher et al., 2015</xref>). Each phylogeny was assessed based on the proportion of highly supported internal nodes (nodal support). After choosing the phylogeny with the best nodal support, additional comparisons between phylogenies were made based on different rooting strategies such as using the <italic>M. bovis</italic> reference (AF2122/97, NC_002945.4, Genbank accession code PRJNA89) as an outgroup, Minimal Ancestor Deviation (<xref ref-type="bibr" rid="ref57">Tria et al., 2017</xref>), and Midpoint rooting (<xref ref-type="bibr" rid="ref25">Kinene et al., 2016</xref>). The best rooting strategy was based on comparisons of Robinson Foulds distance between the different rooted phylogenies, which is a popular metric to ascertain the number of operations required to convert the topology of one tree to another (<xref ref-type="bibr" rid="ref42">Pattengale et al., 2007</xref>). If this value was large, then the two trees being compared are highly dissimilar. After assessing which phylogeny was more representative of the evolutionary history between isolates, that phylogeny was used to simulate an <italic>M. bovis</italic> alignment that will be used to generate null distributions of SNP density and selective sweep regions (sections below).</p>
</sec>
<sec id="sec6">
<title>Population structure</title>
<p>In order to determine <italic>M. bovis</italic> population structure, we used fastBAPS v1.0.4 (<xref ref-type="bibr" rid="ref56">Tonkin-Hill et al., 2019</xref>), a model-based clustering approach, to determine clusters of related sequences present in the genetic alignment. This software utilizes a Bayesian hierarchical clustering approach that handles large alignments and quickly discerns sub-populations. As input, fastBAPS received the SNP only alignment and utilized the optimized symmetric prior with the assumption that all allele frequencies at a SNP site are equivalent (making it analogous to a uniform-like prior). The sub-population clusters were extracted directly from the output of fastBAPS.</p>
</sec>
<sec id="sec7">
<title>Simulated <italic>Mycobacterium bovis</italic> alignment</title>
<p>To investigate if certain regions in the <italic>M. bovis</italic> genome have significantly higher values of either SNP density or selective sweep sites, a probabilistic hypothesis test was used to find highly significant genomic regions. For that, we produced a simulated alignment from the original alignment with the tool Alisim using the same alignment length, nucleotide substitution model, and phylogenetic tree topology (created from the previously computed Maximum Likelihood tree; <xref ref-type="bibr" rid="ref31">Ly-Trong et al., 2022</xref>). Alisim next created a random sequence based on the previous specifications and then simulated nucleotide substitutions that were independently added while also conforming to the phylogenetic topology, resulting in a simulated alignment output. From the simulated alignment, we performed two separate sliding window analyses for calculating SNP dense regions and Selective Sweep regions, respectively. We captured the number of SNPs for the SNP density analysis and the support for selective sweep events in the selective sweep analysis. The data generated from the two separate sliding window analyses were then used to create two null distributions that were used to perform hypothesis testing for the SNP dense and selective sweep regions in the original data. Significance was determined by performing a hypothesis test on each window, where the probability of calculating a certain metric in the sliding window from the original data was compared to the probability of seeing the same metric in the null distribution (<xref rid="fig1" ref-type="fig">Figure 1A</xref>).</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>Workflow to detect SNP Dense Regions and Selective Sweep Regions in the <italic>Mycobacterium bovis</italic> genome. <bold>(A)</bold> The inferred phylogeny created using the original data is used as the input for Alisim to simulate an alignment based on the topology of the <italic>M. bovis</italic> maximum likelihood trees. This simulated alignment will differ from the original alignment since the SNP positions (symbolized by red &#x2018;X&#x2019;s) will be included randomly and independently. <bold>(B)</bold> After obtaining the simulated alignment, a sliding window approach is used to identify SNPs present in each window. The window starts at the beginning of the alignment, and moves across the entire genome by a defined step size. <bold>(C)</bold> A window approach was also used to investigate regions of selective sweeps. The window was centered on a predetermined nucleotide coordinate, and the amount of linkage disequilibrium was compared upstream and downstream of the coordinate.</p>
</caption>
<graphic xlink:href="fmicb-13-787856-g001.tif"/>
</fig>
</sec>
<sec id="sec8">
<title>SNP dense regions</title>
<p>To find SNP dense regions in the genome alignment, we implemented a SNP counting approach that used a sliding window to find regions with significantly higher SNP counts than what would be expected by chance (<xref rid="fig1" ref-type="fig">Figure 1B</xref>). A combination of sliding window size (w) and step size (s) values, where <italic>w</italic>&#x2009;=&#x2009;{100, 500, 1,000, 2,500, 5,000}, <italic>s</italic>&#x2009;=&#x2009;{50, 100, 300, 500, 1,000}, and <italic>w</italic>&#x2009;&#x003E;&#x2009;<italic>s</italic>, was used to determine the best combination of parameters to the highest number of SNPs in windows that were found to be significant (<xref ref-type="bibr" rid="ref54">Tajima, 1991</xref>). Once the optimal window and step lengths were inferred, we used those lengths for each region in the genome to identify high SNP counts. Significance was determined by performing a hypothesis test on each window, where the probability of finding a certain number of SNPs in a window was based on the simulated alignment null distribution. The higher the number of SNPs present in a window, the lower the probability that this window evolved due to conventional means. Our implemented approach was simpler than other tools that are meant to find regions with high SNP density. For example, pre-existing software tools, such as Gubbins, detect regions with high SNP counts by first allowing a sliding window to take lengths between 0.1 to 10&#x2009;kb to identify regions with at least 10 SNPs. Gubbins methodology also incorporates additional hypothesis tests to find both regions with high SNP counts and putative regions of homologous recombination. Our implementation of high SNP regions ensured that we were extracting all of the genomic regions with high SNP counts and not only regions that had high SNP counts and putative support for homologous recombination (which would make us miss some of the high SNP counts regions).</p>
<p>The null distribution should have had characteristics of a Poisson distribution since we were counting the number of SNPs in a certain window size. Therefore, Poisson distribution parameters for this null distribution were estimated using the R package <italic>fitdistrplus</italic> based on a maximum likelihood method (<xref ref-type="bibr" rid="ref600">R Core Team, 2022</xref>; <xref ref-type="bibr" rid="ref14">Delignette-Muller and Dutang, 2015</xref>). For each window that was analyzed, we performed a statistical hypothesis test to determine the probability that the number of SNPs in a window came from the null distribution. Since multiple tests were required for the alignment, to reduce the number of false-positive regions, we used a Bonferroni corrected <italic>p</italic>-value (<xref ref-type="bibr" rid="ref16">Dunn, 1961</xref>). If the probability of the number of SNPs based on the null distribution was lower than the Bonferroni corrected <italic>p</italic>-value {<italic>p</italic>-value/(the number of tests)}, then that window was recorded as having a significantly higher number of SNPs than expected by chance. The combination of <italic>w</italic> and <italic>s</italic> that produced the highest number of significant results was used for subsequent analyses. We merged the overlapping windows to highlight regions that had high numbers of SNPs throughout using BEDtools (<xref ref-type="bibr" rid="ref43">Quinlan and Hall, 2010</xref>). These discrete regions were then labelled as SNP dense regions (SDRs).</p>
</sec>
<sec id="sec9">
<title>Selective sweep regions</title>
<p>To identify the sites that were the center of the selective sweep regions, we used the tool OmegaPlus v3.03 (<xref ref-type="bibr" rid="ref2">Alachiotis et al., 2012</xref>) to determine the value omega, which summarizes the extent of linkage disequilibrium (LD) upstream and downstream of a pre-determined genomic position in the <italic>M. bovis</italic> genome alignment (<xref rid="fig1" ref-type="fig">Figure 1C</xref>). The higher the omega value, the higher the possibility that the genomic position is within the proximity of a selective sweep. We computed the omega values for 5,000 equidistant positions within the simulated alignment. Similar to the protocol employed for detecting SNP dense regions, we first characterized the null distribution of omega values as a gamma distribution since the values started at zero, were continuous, and theoretically could have been unbounded. After filtering out the positions that had the value of zero, we again used the <italic>fitdistrplus</italic> package to estimate the gamma distribution parameters based on the maximum likelihood method. For each window that was analyzed, we performed a statistical hypothesis test with Bonferroni corrected <italic>p</italic>-values to determine the probability that the evidence for a selective sweep coming from that region was from the null distribution. To determine the appropriate maximum window size, we computed the number of significant results (<italic>n</italic>) using multiple window sizes <italic>w</italic>&#x2009;=&#x2009;{150, 300, 500, 750, 1,000, 1,500, 3,000, 5,000, and 10,000}, and divided the number of significant results by <italic>w</italic>. Since the calculation of the omega statistic consistently increases as <italic>w</italic> increases, dividing the number of successful regions by <italic>w</italic> provided a way to find window sizes that did not add extra significant results to the analysis. The maximum value of <italic>w</italic> was determined by finding the window size that maximized the calculation (number of significant windows)/(window size length). We merged the overlapping windows to highlight genomic regions that possessed high evidence for selective sweep events using BEDtools. These discrete regions were then labelled as selective sweep regions (SSRs).</p>
</sec>
<sec id="sec10">
<title>Random forest modeling</title>
<p>After finding the genomic regions with statistical support for increased SNP occurrence and selective sweeps, we measured the impact that evolution in these regions has on being able to differentiate isolates based on their ecological grouping. Random Forest models supported by the random Forest package, were used to perform classification of isolates based on evolutionary metrics to elucidate if SDRs and/or SSRs were important individual predictors in correctly classifying the <italic>M. bovis</italic> data (<xref rid="fig2" ref-type="fig">Figure 2</xref>; <xref ref-type="bibr" rid="ref6">Breiman, 2001</xref>). Specifically, we used the following data as predictors in our models: (i) SDR number of SNPs, (ii) SDR number of INDELs, and (iii) SSR number of SNPs. To measure how the genomic regions we found compare to general characteristics of the <italic>M. bovis</italic> genomes, we also included other genomic metrics of the isolates in our analysis. GC content was calculated directly from the full <italic>M. bovis</italic> alignment, whereas number of non-coding SNPs, number of coding SNPs, number of non-coding INDELs, and number of coding INDELs were extracted directly from each isolate&#x2019;s variant calling format (VCF) file. Prior to model fitting, predictors with a high amount of correlation with other predictors were removed in order to limit redundant data using the R <italic>caret</italic> package (see <xref ref-type="supplementary-material" rid="SM1">Supplementary File 3</xref>; <xref ref-type="bibr" rid="ref26">Kuhn, 2008</xref>). Model performance was calculated based on the overall accuracy in predicting the correct <italic>M. bovis</italic> isolate membership for particular groupings. A predictor&#x2019;s importance was measured through its mean decrease in accuracy (MDA), with higher values indicating that models which did not incorporate the predictor suffered by the indicated number of accuracy points. For each random forest analysis, we ordered predictors based on their MDA and recorded the top 20 most important ones. Certain analyses called for predictors to be compared across different models, such as comparing predictor importance across ecological scales, global versus local cattle hosts, and wildlife reservoir versus livestock hosts. For these analyses, Venn diagrams were constructed to observe which predictors were shared or unique amongst the models. Additionally, predictors that saw sharp increases or decreases in importance when compared across models were recorded.</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>Workflow to determine evolutionary predictors for <italic>M. bovis</italic> across scales. <bold>(A)</bold> Here the dataset is divided based on a binary trait of &#x2018;yes&#x2019; or &#x2018;no&#x2019; with accompanying predictor variables that may provide support in making classifications on the trait. In this study, these predictors are as follows: the number of SNPs in a SNP dense region, the number of INDELs in a SNP dense region, the number of SNPs in a selective sweep region, the number of SNPs in coding regions, the number of SNPs in non-coding regions, the number of INDELs in coding regions, the number of INDELs in non-coding regions, and the genome wide GC%. <bold>(B)</bold> Random Forest models use multiple, uncorrelated decision trees created from randomly subset predictors of the original data. Due to the individual trees being unrelated to each other, their combined predictive ability leads to excellent classification ability. <bold>(C)</bold> Since each individual decision tree is created from random subsets of the predictors, the accuracy of individual trees will vary greatly (red and blue represent low and high accuracy, respectively). The amount of increase and/or decrease in model accuracy when a specific predictor is included is determined by tracking which predictors are included in the simple decision trees.</p>
</caption>
<graphic xlink:href="fmicb-13-787856-g002.tif"/>
</fig>
</sec>
<sec id="sec11">
<title>Gene functions associated with models</title>
<p>After key genomic regions were identified from each random forest model, we used BEDtools to identify sets of genes that were present within important SDRs and SSRs. In particular, we highlight genes that have been catalogued in previous research as playing a role in host pathogen interactions in mycobacterial pathogens.</p>
</sec>
</sec>
<sec id="sec12" sec-type="results">
<title>Results</title>
<sec id="sec13">
<title>Phylogenetic reconstruction of <italic>Mycobacterium bovis</italic> isolates</title>
<p>Whole-genome sequencing of the 700 <italic>M. bovis</italic> isolates from studies implemented in NZ, UK, and the USA identified a total of 6,847 SNPs (<xref ref-type="supplementary-material" rid="SM1">Supplementary Table 1</xref>). Even though the genomes were sequenced on different platforms, in the 692 samples that were sequenced on Illumina HiSeq 2500 (197) and Illumina MiSeq (495), the number of SNPs, depth per base pair, and average depth per genome did not deviate much across platforms (<xref ref-type="supplementary-material" rid="SM1">Supplementary Figure 1</xref>). Additionally, the number of SNPs that were detected in coding versus non-coding regions of the genome shared similar distributions between all isolates sequenced on the differing platforms (<xref ref-type="supplementary-material" rid="SM1">Supplementary Figure 2</xref>), suggesting that the choice of sequencing technology did not influence the downstream results. Isolates sequenced on a NextSeq platform (8) were not included in the violin plots constructed for <xref ref-type="supplementary-material" rid="SM1">Supplementary Figures S1, S2</xref> due to an insufficient number of isolates characterized with this platform.</p>
<p>When comparing the phylogenetic trees created from the original alignment and the one outputted by Gubbins, we found that there was a Robinson Foulds distance of 592 between the two trees, meaning that one tree would need 592 changes to be converted to the other. The Gubbins phylogeny possessed better nodal support (75.7% of internal nodes with bootstrap value of 75 or above) than the phylogeny produced from the original alignment (70.1% of internal nodes with bootstrap value of 75 or above; <xref ref-type="supplementary-material" rid="SM1">Supplementary Table 2</xref>), which suggests that it provides a better approximation of the evolutionary history between the <italic>M. bovis</italic> isolates in this dataset, and therefore, was used both for the rooting method comparison and for simulating the <italic>M. bovis</italic> alignment. The best rooting strategy for the phylogeny was the Outgroup rooting method [using the reference AF2122/97 (NC002945.4)], since it led to fewer changes in the topology (12 and 14) when compared to the Midpoint and MAD rooting strategies (<xref ref-type="supplementary-material" rid="SM1">Supplementary Figure 3</xref>, <xref ref-type="supplementary-material" rid="SM1">Supplementary Table 3</xref>). The substitution model that best fit the data was the TPM2+F+R4. The <italic>M. bovis</italic> phylogeny produced clustering patterns that were distinguishable based on geographic region (<xref rid="fig3" ref-type="fig">Figure 3</xref>). The population clustering analysis identified eight distinct clusters that matched closely with geographical locations. For instance, Cluster 1 had the same isolates as the USA group. Clusters 6 and 7 were subpopulations of the UK group, and NZ isolates contained the remaining population clusters (Clusters 2, 3, 4, 5, and 8). The population clusters did not show a visible pattern based on host species, indicating that the estimated <italic>M. bovis</italic> population clusters are determined mostly by geographical location.</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p><italic>Mycobacterium bovis</italic> maximum likelihood phylogenetic tree. This tree was inferred from the 6,847 SNPs of 700 <italic>Mycobacterium bovis</italic> isolates extracted from three independent studies in endemic regions: New Zealand, United Kingdom, and the United States of America. The TPM2+F+R4 substitution model was the one with best statistical support (by Bayesian Information Criteria). Ultrafast bootstrap support for the internal nodes is indicated by a black circle, its presence indicating that the node has support of &#x003E;75%. The phylogeny is further annotated by the taxa&#x2019;s membership in a population cluster (left), country of origin (middle), and host species (right). The tree was rooted using the <italic>M. bovis</italic> reference genome AF2122/97 as an outgroup.</p>
</caption>
<graphic xlink:href="fmicb-13-787856-g003.tif"/>
</fig>
</sec>
<sec id="sec14">
<title>SNP density and selective sweep region identification</title>
<p>The simulated <italic>M. bovis</italic> alignment was produced using the TPM2+F+R4 model, which was also the best supported model. For the SNP Density analysis, we determined that the combination of a window size of 1,000&#x2009;bp and a step size of 50&#x2009;bp were the parameters that identified the most significant windows (<xref rid="fig4" ref-type="fig">Figure 4A</xref>). In each window, we counted the number of SNPs that were within that window and represented that data as a Poisson distribution. We estimated the Poisson distribution parameter lambda to be 2.84, which suggests that in our simulated alignment, we can expect to have 2.84 SNPs in a window on average with a variance of 2.84 (<xref rid="fig4" ref-type="fig">Figure 4B</xref>). For the selective sweep analysis, we identified that a window size of 5,000&#x2009;bp was needed to maximize the number of significant results (<xref rid="fig4" ref-type="fig">Figure 4C</xref>), and similarly we estimated the parameters of the Gamma null distribution to be 0.944 (the shape parameter estimates the typical omega value in a window), and 0.202 (the scale parameter estimate that variance of the omega value), respectively. Based on the inferred parameter values for both null distributions, the genome-wide hypothesis tests computed the probability of finding the amount of SNPs/evidence of a selective sweep within the tested window. After we merged the overlapping significant windows, we determined 14 unique SDRs and 132 unique SSRs (<xref ref-type="supplementary-material" rid="SM1">Supplementary Tables 4, 5</xref>). A few SDRs and SSRs overlapped regions with regions containging <italic>pe</italic> or <italic>ppe</italic> genes, but since the SNPs falling within these genes were removed from the analysis, these genes did not contribute to the identification of SDRs or SSRs.</p>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption>
<p>The process of creating the null distributions used for SNP Dense Regions (SDRs) and Selective Sweep Regions identification (SSRs). <bold>(A)</bold> Multiple sliding window analyses were conducted to find the window size and step size whose combination maximized the number of significant windows. The white to blue gradient represent the indicated increased number of significant results. <bold>(B)</bold> The genome wide hypothesis test for SDRs relied on a window size of 1,000 bp and a step size of 50 bp to generate a Poisson null distribution for subsequent tests. <bold>(C)</bold> Multiple selective sweep analyses were conducted by varying the window size to find the size that maximized the number of significant results normalized by the window length. <bold>(D)</bold> The genome wide hypothesis test for SSRs relied on the window size of 5,000 bp to generate a Gamma null distribution for subsequent tests.</p>
</caption>
<graphic xlink:href="fmicb-13-787856-g004.tif"/>
</fig>
</sec>
<sec id="sec15">
<title>Random forest analysis</title>
<sec id="sec16">
<title>Shared and unique regions across scales</title>
<p>Three random forest models were computed for each individual scale present in our data: country of origin, sub-population, and host-species. Models presented different accuracy depending on the scale in focus. For the host species model, the model accuracy to classify <italic>M. bovis</italic> hosts was low for the majority of the host species represented in the data, especially within phylogenetic clades where multiple host species were present (low host-species structure). However, the model achieved high accuracy in classifying isolates from the UK as being either from cattle or badger hosts, with accuracies of 94.5 and 94.6%, respectively.</p>
<p>For each model, SSRs achieved the highest importance rankings for model classification, with a few SDRs ranking as some of the top 20 predictors for each model. As for the other genomic evolutionary metrics (such as INDELS and GC content), which were included in the model to determine how important SSRs or SDRs were, they were not included amongst the top 20 predictors (<xref rid="fig5" ref-type="fig">Figures 5A</xref>&#x2013;<xref rid="fig5" ref-type="fig">C</xref>).</p>
<fig position="float" id="fig5">
<label>Figure 5</label>
<caption>
<p>The top 20 predictors across the different scales. The predictors were presented for the Country of origin <bold>(A)</bold>, Population cluster <bold>(B)</bold>, and Host species <bold>(C)</bold> random forest models. A Venn diagram was used to identify predictive genomic regions that were shared and unique across models <bold>(D)</bold>.</p>
</caption>
<graphic xlink:href="fmicb-13-787856-g005.tif"/>
</fig>
<p>Out of all the regions used in the random forest analysis, only 16 SSRs and one SDR were shared as the top 20 most important variables across all the models (<xref rid="fig5" ref-type="fig">Figure 5D</xref>). It was rare for the top SSRs and SDRs of the three random forest models to be shared by two models only. For instance, country-of-origin and population cluster, as well as population cluster and host species, shared one region amongst their top 20 predictors, but zero regions were shared between the country of origin and species models. Each individual model also had at least two SSRs/SDRs that were uniquely important for correct classification (<xref rid="fig5" ref-type="fig">Figure 5</xref>).</p>
<p>Amongst all predictors in our models, only four SSRs and one SDR helped improve the model accuracy as the scales went from being general (country-of-origin) to more specific (host-species). Conversely, only two SSRs were found to be of decreased importance (<xref rid="fig6" ref-type="fig">Figure 6</xref>). SSR53 was an important region regardless of the model, with MDAs for country of origin, population clustering, and species being 27.8, 38.8, and 44.9%, respectively. Additionally, three regions, SSR71, SSR6, and SDR10 were observed to be of lower importance in the country-of-origin model (SSR71: 13.2%, SSR6: 13.8%, and SDR10: 10.0%) but jumped in importance within the host-species models (SSR71: 25.3%, SSR6: 31.4%, and SDR10: 24.9%). SSR28 and SSR85 were the sole predictors to decrease in importance as the scale narrowed between the models, but this decrease was modest, especially for SSR28 (29.3 to 27.2%; <xref rid="fig6" ref-type="fig">Figure 6</xref>).</p>
<fig position="float" id="fig6">
<label>Figure 6</label>
<caption>
<p>Factors that affect model accuracy across scales. The seven regions between the &#x2018;across-scales model&#x2019; that increase or decrease the model accuracy as the scales move from general (Country of origin) to less specific (Host species).</p>
</caption>
<graphic xlink:href="fmicb-13-787856-g006.tif"/>
</fig>
</sec>
<sec id="sec17">
<title>Gene functions across the geographic, sub-population, and host-species models</title>
<p>We observed that the genes within genomic regions that increased (SSR53, SSR42, SSR6, SSR71, SDR10) or decreased in importance (SSR28, SSR85) across the different scales contained general functions that impact virulence and immunogenicity (<italic>phoP</italic> and <italic>phoR</italic>), anaerobic survival through nitrate reduction (<italic>narG</italic>, <italic>narH</italic>, <italic>narJ</italic>, and <italic>narI</italic>), membrane structure (<italic>pks5</italic>, <italic>papA4</italic>, and <italic>fadD25</italic>), and ESX-3 secretion (<italic>ecca3</italic>, <italic>eccb3</italic>, <italic>eccc3</italic>, <italic>esxG</italic>, <italic>esxH</italic>, <italic>espg3</italic>, and <italic>eccd3</italic>). Of particular interest, the ESX-1 Type VII secretion system was identified as a unique selective sweep target in the sub-population model (<italic>esxB</italic>, <italic>esxA</italic>, and <italic>espI</italic>; <xref ref-type="supplementary-material" rid="SM1">Supplementary Table 6</xref>).</p>
</sec>
<sec id="sec18">
<title>Regions impacted by geographic stratification of species</title>
<p>The models for determining the segregation between worldwide cattle and wildlife both possessed adequate accuracy, with the aggregated cattle model being 88% accurate in differentiating cattle from wildlife, while the stratified cattle model was 70, 83, and 94% accurate when classifying cattle from the USA, NZ, and UK, respectively, (<xref rid="fig7" ref-type="fig">Figures 7A</xref>,<xref rid="fig7" ref-type="fig">B</xref>). Between the two models, a total of 18 regions were shared as being the top 20 predictors of isolates originating in cattle versus wildlife (<xref rid="fig7" ref-type="fig">Figure 7C</xref>). Only two regions were found to be unique for each model. We recorded the top predictors with an absolute MDA change between the two models of at least 10 and found that all three predictors increased in importance from the aggregated cattle model compared to the stratified cattle model (<xref rid="fig7" ref-type="fig">Figure 7D</xref>). Of the predictors, the majority (2 out of 3) were SSRs. The only SDR noted with an absolute MDA change was SDR11. SSR24 was recorded as having the highest difference between the models with a 13.6% increase in importance (11% in the aggregated cattle model versus 24.7% in the stratified cattle model; <xref ref-type="supplementary-material" rid="SM1">Supplementary Table 6</xref>).</p>
<fig position="float" id="fig7">
<label>Figure 7</label>
<caption>
<p>The 20 top predictors for cattle related random forest models. The 20 top predictors were shown for the <bold>(A)</bold> aggregated cattle and <bold>(B)</bold> stratified cattle random forest models. <bold>(C)</bold> A Venn diagram to identify regions that were shared and unique across models. <bold>(D)</bold> Three genomic regions between the cattle vs. wildlife models that increase or decrease model accuracy as the scales move from aggregated to stratified.</p>
</caption>
<graphic xlink:href="fmicb-13-787856-g007.tif"/>
</fig>
</sec>
<sec id="sec19">
<title>Gene functions that define core cattle classification</title>
<p>Our results of the differences in genomic region importance for the aggregated cattle vs. stratified cattle models highlighted that for global identification of cattle vs. wildlife, evolution in genes impacting the mce4 operon, which influences cholesterol utilization and intracellular survival were unique to the aggregated cattle model (<italic>mce4D</italic>, <italic>mce4C</italic>, <italic>mce4B</italic>, <italic>mce4A</italic>, <italic>yrbE4B</italic>, and <italic>yrbE4A</italic>). When isolates were classified to identify the geographically separated cattle vs. wildlife, toxin-antitoxin system functions appeared to be uniquely important to make the distinction (<italic>vapc12</italic> and <italic>vapb12</italic>). Additionally, it was observed that mycobactin biosynthesis function (<italic>mbtH</italic>, <italic>mbtG</italic>, <italic>mbtF</italic>, <italic>mbtE</italic>, and <italic>mbtD</italic>), as well as toxin-antitoxin function (<italic>vapc7</italic>, <italic>vapb7</italic>, <italic>vapb8</italic>, and <italic>vapc8</italic>) increased in importance when differentiating global cattle vs. geographically stratified cattle.</p>
</sec>
<sec id="sec20">
<title>Regions that differentiate a wildlife badger from a cattle host</title>
<p>In the random forest model differentiating <italic>M. bovis</italic> isolated from UK Badger vs. UK Cattle, the model was accurate in classifying isolates extracted from badgers at 94.6% and isolates extracted from cattle at 94.5% (<xref rid="tab1" ref-type="table">Table 1</xref>). The top predictor of host in the wildlife badger and cattle random forest model was SSR53, and the other predictors were dominated by SSRs with very few SDRs (<xref rid="fig8" ref-type="fig">Figure 8</xref>). However, SDR10 was found to be the second highest important predictor in this model, and compared to every other computed random forest model, was the highest SDR model importance seen yet (<xref ref-type="supplementary-material" rid="SM1">Supplementary Table 6</xref>).</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Summary of the random forest classification accuracy for each developed model.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Model</th>
<th align="left" valign="top">Grouping</th>
<th align="center" valign="top">Accuracy</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top" char="." rowspan="3">Country of origin</td>
<td align="left" valign="top" char="&#x00B1;">United States of America</td>
<td align="center" valign="top" char="&#x00B1;">1</td>
</tr>
<tr>
<td align="left" valign="top" char="&#x00B1;">United Kingdom</td>
<td align="center" valign="top" char="&#x00B1;">0.996</td>
</tr>
<tr>
<td align="left" valign="top" char="&#x00B1;">New Zealand</td>
<td align="center" valign="top" char="&#x00B1;">0.996</td>
</tr>
<tr>
<td align="left" valign="top" char="." rowspan="8">Population cluster</td>
<td align="left" valign="top" char="&#x00B1;">Cluster 1</td>
<td align="center" valign="top" char="&#x00B1;">1</td>
</tr>
<tr>
<td align="left" valign="top" char="&#x00B1;">Cluster 2</td>
<td align="center" valign="top" char="&#x00B1;">1</td>
</tr>
<tr>
<td align="left" valign="top" char="&#x00B1;">Cluster 3</td>
<td align="center" valign="top" char="&#x00B1;">0.972</td>
</tr>
<tr>
<td align="left" valign="top" char="&#x00B1;">Cluster 4</td>
<td align="center" valign="top" char="&#x00B1;">1</td>
</tr>
<tr>
<td align="left" valign="top" char="&#x00B1;">Cluster 5</td>
<td align="center" valign="top" char="&#x00B1;">1</td>
</tr>
<tr>
<td align="left" valign="top" char="&#x00B1;">Cluster 6</td>
<td align="center" valign="top" char="&#x00B1;">1</td>
</tr>
<tr>
<td align="left" valign="top" char="&#x00B1;">Cluster 7</td>
<td align="center" valign="top" char="&#x00B1;">1</td>
</tr>
<tr>
<td align="left" valign="top" char="&#x00B1;">Cluster 8</td>
<td align="center" valign="top" char="&#x00B1;">0.989</td>
</tr>
<tr>
<td align="left" valign="top" char="." rowspan="10">Host species</td>
<td align="left" valign="top" char="&#x00B1;">UK Badger</td>
<td align="center" valign="top" char="&#x00B1;">0.946</td>
</tr>
<tr>
<td align="left" valign="top" char="&#x00B1;">UK Cattle</td>
<td align="center" valign="top" char="&#x00B1;">0.945</td>
</tr>
<tr>
<td align="left" valign="top" char="&#x00B1;">USA Cattle</td>
<td align="center" valign="top" char="&#x00B1;">0.8</td>
</tr>
<tr>
<td align="left" valign="top" char="&#x00B1;">USA Elk</td>
<td align="center" valign="top" char="&#x00B1;">0</td>
</tr>
<tr>
<td align="left" valign="top" char="&#x00B1;">USA White tailed Deer</td>
<td align="center" valign="top" char="&#x00B1;">1</td>
</tr>
<tr>
<td align="left" valign="top" char="&#x00B1;">New Zealand Ferret</td>
<td align="center" valign="top" char="&#x00B1;">0.05</td>
</tr>
<tr>
<td align="left" valign="top" char="&#x00B1;">New Zealand Porcine</td>
<td align="center" valign="top" char="&#x00B1;">0</td>
</tr>
<tr>
<td align="left" valign="top" char="&#x00B1;">New Zealand Possum</td>
<td align="center" valign="top" char="&#x00B1;">0.11</td>
</tr>
<tr>
<td align="left" valign="top" char="&#x00B1;">New Zealand Cattle</td>
<td align="center" valign="top" char="&#x00B1;">0.894</td>
</tr>
<tr>
<td align="left" valign="top" char="&#x00B1;">New Zealand Cervine</td>
<td align="center" valign="top" char="&#x00B1;">0</td>
</tr>
<tr>
<td align="left" valign="top" char="." rowspan="2">Aggregated cattle</td>
<td align="left" valign="top" char="&#x00B1;">Bovine</td>
<td align="center" valign="top" char="&#x00B1;">0.868</td>
</tr>
<tr>
<td align="left" valign="top" char="&#x00B1;">Wildlife</td>
<td align="center" valign="top" char="&#x00B1;">0.739</td>
</tr>
<tr>
<td align="left" valign="top" char="." rowspan="4">Stratified cattle</td>
<td align="left" valign="top" char="&#x00B1;">UK Bovine</td>
<td align="center" valign="top" char="&#x00B1;">0.937</td>
</tr>
<tr>
<td align="left" valign="top" char="&#x00B1;">USA Bovine</td>
<td align="center" valign="top" char="&#x00B1;">0.7</td>
</tr>
<tr>
<td align="left" valign="top" char="&#x00B1;">New Zealand Bovine</td>
<td align="center" valign="top" char="&#x00B1;">0.830</td>
</tr>
<tr>
<td align="left" valign="top" char="&#x00B1;">Wildlife</td>
<td align="center" valign="top" char="&#x00B1;">0.772</td>
</tr>
<tr>
<td align="left" valign="top" char="." rowspan="2">UK Badger and Cattle</td>
<td align="left" valign="top" char="&#x00B1;">Badger</td>
<td align="center" valign="top" char="&#x00B1;">0.946</td>
</tr>
<tr>
<td align="left" valign="top" char="&#x00B1;">Cattle</td>
<td align="center" valign="top" char="&#x00B1;">0.945</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Accuracy is defined as the proportion of correct classifications of a grouping over the total number of attempts. Values can range from 0 to 1.</p>
</table-wrap-foot>
</table-wrap>
<fig position="float" id="fig8">
<label>Figure 8</label>
<caption>
<p>The 20 top predictors for the United Kingdom badger versus cattle random forest models. SNP evolution occurring in SSR53 is the top predictor in differentiating the species in the United Kingdom. SDR10 increased substantially in model importance, indicating its ability to differentiate badger from cattle.</p>
</caption>
<graphic xlink:href="fmicb-13-787856-g008.tif"/>
</fig>
</sec>
</sec>
</sec>
<sec id="sec21" sec-type="discussions">
<title>Discussion</title>
<p>This analysis presents one of the first studies in <italic>Mycobacterium bovis</italic> research that uses comparative genomics and machine learning approaches to identify putative genomic factors that contribute to across-scale evolution of <italic>M. bovis</italic> isolates. Using publicly available <italic>M. bovis</italic> sequences from well characterized molecular epidemiological studies, we investigated specific genomic signatures across ecological scales such as geographical locations, host-species, and pathogen population structure. To best identify these genomic signatures, we used a probabilistic framework to identify regions possessing high SNP counts and/or regions with high evidence for selective sweeps. After, we used random forest models to assess the importance of these genomic regions and other genomic variables such as GC content and the number of INDELS in classifying the <italic>M. bovis</italic> genomics across the three different scales. The combined application of machine learning and comparative genomics of pathogenic organisms to better understand connections between evolution and molecular epidemiology is still in its early stages. For <italic>M. bovis</italic> research, machine learning techniques were first explored in the work by <xref ref-type="bibr" rid="ref10">Crispell et al. (2019)</xref>, where epidemiological metrics were used to investigate their contribution to the genomic similarity of isolates during an <italic>M. bovis</italic> outbreak in cattle and badger populations. Likewise, there have been numerous other studies that determined specific <italic>M. bovis</italic> evolutionary targets that coincide with various lineages (<xref ref-type="bibr" rid="ref41">Patan&#x00E9; et al., 2017</xref>; <xref ref-type="bibr" rid="ref59">Zimpel et al., 2020</xref>).</p>
<p>The random forest models we developed were helpful in investigating which genomic regions became more important as ecological scales changed. In both the Across-Scales model comparison and the Aggregated <italic>vs</italic>. Stratified cattle model, pathogenic genes were harbored within the genomic regions that had sharp increases or decreases when the predictor importance was compared between models. While the genomic regions importance is not an assertion of being directly related to host adaptation, it provides an early <italic>in-silico</italic> examination of genes that possibly could play a role in adaptation to new environments. We hope that by identifying and reporting these genes, researchers that study <italic>M. bovis</italic> pathogenesis can further elucidate how the functions we highlighted impact <italic>M. bovis</italic> transmission in the very common multi-host environments.</p>
<p>Across every single model, SSRs were the most consistently important predictor in differentiating isolates from different geographic, sub-population, or host species groupings. Amongst the top 20 predictors in every individual model, SSRs had very high MDAs and achieved high levels of importance, despite the presence of SDRs in the full data. This indicated that evolution in selective sweep regions play a major role in differentiating isolates across the different scales. This would match what is observed in other pathogenic bacteria that maintain multiple host-associated clonal lineages such as <italic>Campylobacter jejuni</italic>, <italic>Staphylococcus aureus</italic> and <italic>Salmonella enterica</italic> (<xref ref-type="bibr" rid="ref50">Sheppard et al., 2018</xref>). Since <italic>M. bovis</italic> evolution is also described as evolving clonally, the main avenue of adapting to new host populations would likely be achieved predominantly through mutations that confer high selective advantage within a particular niche. Monitoring the genomic regions with SNPs in high linkage disequilibrium could be leveraged to identify genes that are essential for <italic>M. bovis</italic> transmission between different ecological scales and perhaps provide a signal that <italic>M. bovis</italic> is being maintained within a new population.</p>
<p>While SSRs were predominant amongst the models, a few SDRs did rank as the top 20 most important predictors (SDR3, SDR9, SDR10, SDR11, and SDR14). In the UK badger and cattle model, SDR10 was the second highest predictor of wildlife badger status and additionally had the highest importance rank amongst all the SDRs. SDR10 contains genes that are implicated in mycobacterial adaptation (<italic>vapb18</italic> and <italic>vapc18</italic>) as well as lipoproteins that affect lipid modifications (<italic>lppA</italic>, <italic>lppB</italic>, and <italic>lprR</italic>). Elevated mutation rates in these genes could have a direct impact on adaptation to the immunological pressures present within the UK wildlife reservoir through changes in membrane structure. Additionally, since genes within the SDR10 region gained high importance in the badger and cattle model, genomic regions of high SNP density might play a larger role in classifying local isolates than global ones. Altogether, this provides some indication that while selective sweep sites are important genomic signatures to investigate <italic>M. bovis</italic> evolution, the contribution of genomic regions that undergo excessive mutation should be further investigated. The accuracy of the random forest models to classify country-of-origin and sub-population groups was consistently high, but in the case of the host-species model, the model failed in classifying correctly all the host-species, having just a few adequately classified. The reasons for the poor performance of the host-species model in certain host-species groups could be due to lack of host species structure along the phylogeny. When compared to the number of SNPs within key SSRs, the SNP count metric correlated poorly with host species (<xref ref-type="supplementary-material" rid="SM1">Supplementary Figures 4&#x2013;6</xref>). We analyzed the top three SSRs (SSR53, SSR85, and SSR11) that were used to classify host species and <xref ref-type="supplementary-material" rid="SM1">Supplementary Figures 4&#x2013;6</xref> show that in clades with low host-population structure, the number of SNPs in these regions did not have enough resolution to distinguish host-species within clades. Despite this poor accuracy, the utilization of sub-populations, first described by <xref ref-type="bibr" rid="ref45">Rodrigues et al. (2021)</xref> to define <italic>M. bovis</italic> in Brazil, sub-divided into eight distinct population clusters, showed remarkable accuracy for correctly identifying the proper sub-population. The inferred population clusters encompassed <italic>M. bovis</italic> genomic similarity isolated from multiple host species, so this method is most likely overcoming the accuracy problem found in the host species model by defining communities of hosts that share a similar evolutionary history. Since our results show high accuracy in differentiating the samples based on their fastBAPS derived clusters rather than host species, future studies should further explore the relationship between population clusters, transmission, and selective sweeps in order to dissect <italic>M. bovis</italic> adaptation. Furthermore, these results suggest that when researchers investigate the genomic factors that contribute to <italic>M. bovis</italic> adaptation, it is more suitable to conduct these comparative analyses at local scales rather than global scales (where evolution occurred separately for a prolonged time between distinct regions). Otherwise, comparisons made between isolates based solely on geographic distance has the potential to mask biologically relevant results that are contributing to putative adaptation.</p>
<p>One limitation of this study was the amount of data available for each host-species in the different geographical locations. The data were sparse regarding certain host-species populations, making it difficult to draw conclusions about the model&#x2019;s ability to differentiate host species populations. In terms of accuracy, we noticed that low accuracy classes typically were not heavily sampled, but exceptions also existed. For example, 5 USA elk were never predicted accurately in our species model, while 13 USA cattle were correctly classified 80%. It appears that sub-population inference might be the best approach to mitigate the classification issue and identify variation occurring amongst <italic>M. bovis</italic> isolates, but without proper sampling of particular host-species it will be difficult to conclude if the low accuracy in classifying species is due to factors other than sample size. Additionally, we focused on selective sweep sites and SNP dense regions as the main genomic signatures of interest, and while our analyses produced insights regarding the relationship between <italic>M. bovis</italic> genomic evolution and ecological niche membership, there are other genomic signatures that could be investigated further. The data for this analysis was based on the number of SNPs within a SDR or SSR, but information about the exact SNPs that provided statistical support to differentiate <italic>M. bovis</italic> amongst various niches was not investigated. In future work, specialized bacterial genome-wide association studies (bacGWAS) would be useful to find influential SNPs, and furthermore answer the question of how these mutations impact the structure of particular genes (<xref ref-type="bibr" rid="ref27">Lees et al., 2018</xref>).</p>
</sec>
<sec id="sec22" sec-type="data-availability">
<title>Data availability statement</title>
<p>The WGS datasets and corresponding metadata analysed during the current study were downloaded from the NCBI Sequence Read Archive under the Bioproject Accession numbers: PRJNA363037 (NZ), PRJNA523164 (UK), and PRJNA251692 (USA). The scripts used for this publication are freely available on the following link: <ext-link xlink:href="https://github.com/salvadorlab/LegallSalvador2022_MbovisAcrossScales" ext-link-type="uri">https://github.com/salvadorlab/LegallSalvador2022_MbovisAcrossScales</ext-link>.</p>
</sec>
<sec id="sec23">
<title>Author contributions</title>
<p>NL designed the study, analyzed data, and wrote first draft of manuscript. LCMS designed the study, supervised data analysis, contributed to study interpretation, and revised the manuscript. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec id="sec24" sec-type="funding-information">
<title>Funding</title>
<p>This work was supported by the National Science Foundation under grant no. DGE-1545433 501 to NL and startup funds to LS from the University of Georgia Office of Research.</p>
</sec>
<sec id="conf1" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="sec100" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<ack>
<p>The authors thank lab mates within the Salvador lab (Ruijie Xu, Rodrigo Campos, Assel Akhmetova, and Ehsan Suez) alongside peers in the Bahl lab (Justin Bahl, Swan Tan, Jiani Chen, Lambodhar Damodaran, Leke Lyu, Gabriella Veytsal, and Cody Dailey). Additional thanks to Tod Stuber from the United States Department of Agriculture for help with bioinformatic related questions. Finally, as with most endeavors NL takes on in life, he could not have tackled this without the support of his parents Theodore Legall Jr. and Susan Legall.</p>
</ack>
<sec id="sec26" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary material for this article can be found online at: <ext-link xlink:href="https://www.frontiersin.org/articles/10.3389/fmicb.2022.787856/full#supplementary-material" ext-link-type="uri">https://www.frontiersin.org/articles/10.3389/fmicb.2022.787856/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.ZIP" id="SM1" mimetype="application/ZIP" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Akhmetova</surname> <given-names>A.</given-names></name> <name><surname>Guerrero</surname> <given-names>J.</given-names></name> <name><surname>McAdam</surname> <given-names>P.</given-names></name> <name><surname>Salvador</surname> <given-names>L. C. M.</given-names></name> <name><surname>Crispell</surname> <given-names>J.</given-names></name> <name><surname>Lavery</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Genomic epidemiology of <italic>Mycobacterium bovis</italic> infection in sympatric badger and cattle populations in Northern Ireland</article-title>. <source>BioRxiv</source> <volume>2021</volume>:<fpage>435101</fpage>. doi: <pub-id pub-id-type="doi">10.1101/2021.03.12.435101</pub-id></citation></ref>
<ref id="ref2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alachiotis</surname> <given-names>N.</given-names></name> <name><surname>Stamatakis</surname> <given-names>A.</given-names></name> <name><surname>Pavlidis</surname> <given-names>P.</given-names></name></person-group> (<year>2012</year>). <article-title>OmegaPlus: a scalable tool for rapid detection of selective sweeps in whole-genome datasets</article-title>. <source>Bioinformatics</source> <volume>28</volume>, <fpage>2274</fpage>&#x2013;<lpage>2275</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/bts419</pub-id>, PMID: <pub-id pub-id-type="pmid">22760304</pub-id></citation></ref>
<ref id="ref3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Allen</surname> <given-names>A. R.</given-names></name></person-group> (<year>2017</year>). <article-title>One bacillus to rule them all? - investigating broad range host adaptation in <italic>Mycobacterium bovis</italic></article-title>. <source>Infect. Genet. Evol.</source> <volume>53</volume>, <fpage>68</fpage>&#x2013;<lpage>76</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.meegid.2017.04.018</pub-id>, PMID: <pub-id pub-id-type="pmid">28434972</pub-id></citation></ref>
<ref id="ref4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Berg</surname> <given-names>S.</given-names></name> <name><surname>Garcia-Pelayo</surname> <given-names>M. C.</given-names></name> <name><surname>M&#x00FC;ller</surname> <given-names>B.</given-names></name> <name><surname>Hailu</surname> <given-names>E.</given-names></name> <name><surname>Asiimwe</surname> <given-names>B.</given-names></name> <name><surname>Kremer</surname> <given-names>K.</given-names></name> <etal/></person-group>. (<year>2011</year>). <article-title>African 2, a clonal complex of <italic>Mycobacterium bovis</italic> epidemiologically important in East Africa</article-title>. <source>J. Bacteriol.</source> <volume>193</volume>, <fpage>670</fpage>&#x2013;<lpage>678</lpage>. doi: <pub-id pub-id-type="doi">10.1128/JB.00750-10</pub-id>, PMID: <pub-id pub-id-type="pmid">21097608</pub-id></citation></ref>
<ref id="ref5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Biek</surname> <given-names>R.</given-names></name> <name><surname>O&#x2019;Hare</surname> <given-names>A.</given-names></name> <name><surname>Wright</surname> <given-names>D.</given-names></name> <name><surname>Mallon</surname> <given-names>T.</given-names></name> <name><surname>McCormick</surname> <given-names>C.</given-names></name> <name><surname>Orton</surname> <given-names>R. J.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>Whole genome sequencing reveals local transmission patterns of <italic>Mycobacterium bovis</italic> in sympatric cattle and badger populations</article-title>. <source>PLoS Pathog.</source> <volume>8</volume>:<fpage>e1003008</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.ppat.1003008</pub-id>, PMID: <pub-id pub-id-type="pmid">23209404</pub-id></citation></ref>
<ref id="ref6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Breiman</surname> <given-names>L.</given-names></name></person-group> (<year>2001</year>). <article-title>Random forests</article-title>. <source>Mach. Learn.</source> <volume>45</volume>, <fpage>5</fpage>&#x2013;<lpage>32</lpage>. doi: <pub-id pub-id-type="doi">10.1023/A:1010933404324</pub-id></citation></ref>
<ref id="ref7"><citation citation-type="other"><person-group person-group-type="author"><collab id="coll1">Broadinstitute/Picard</collab></person-group>. (<year>2021</year>). [Java]. Broad Institute. <ext-link xlink:href="https://github.com/broadinstitute/picard" ext-link-type="uri">https://github.com/broadinstitute/picard</ext-link> (Original work published 2014).</citation></ref>
<ref id="ref8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brosch</surname> <given-names>R.</given-names></name> <name><surname>Gordon</surname> <given-names>S. V.</given-names></name> <name><surname>Marmiesse</surname> <given-names>M.</given-names></name> <name><surname>Brodin</surname> <given-names>P.</given-names></name> <name><surname>Buchrieser</surname> <given-names>C.</given-names></name> <name><surname>Eiglmeier</surname> <given-names>K.</given-names></name> <etal/></person-group>. (<year>2002</year>). <article-title>A new evolutionary scenario for the <italic>Mycobacterium tuberculosis</italic> complex</article-title>. <source>Proc. Natl. Acad. Sci.</source> <volume>99</volume>, <fpage>3684</fpage>&#x2013;<lpage>3689</lpage>. doi: <pub-id pub-id-type="doi">10.1073/pnas.052548299</pub-id>, PMID: <pub-id pub-id-type="pmid">11891304</pub-id></citation></ref>
<ref id="ref9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>S.</given-names></name> <name><surname>Zhou</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Gu</surname> <given-names>J.</given-names></name></person-group> (<year>2018</year>). <article-title>fastp: an ultra-fast all-in-one FASTQ preprocessor</article-title>. <source>Bioinformatics</source> <volume>34</volume>, <fpage>i884</fpage>&#x2013;<lpage>i890</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/bty560</pub-id>, PMID: <pub-id pub-id-type="pmid">30423086</pub-id></citation></ref>
<ref id="ref10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Crispell</surname> <given-names>J.</given-names></name> <name><surname>Benton</surname> <given-names>C. H.</given-names></name> <name><surname>Balaz</surname> <given-names>D.</given-names></name> <name><surname>De Maio</surname> <given-names>N.</given-names></name> <name><surname>Ahkmetova</surname> <given-names>A.</given-names></name> <name><surname>Allen</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Combining genomics and epidemiology to analyse bi-directional transmission of <italic>Mycobacterium bovis</italic> in a multi-host system</article-title>. <source>elife</source> <volume>8</volume>:<fpage>e45833</fpage>. doi: <pub-id pub-id-type="doi">10.7554/eLife.45833</pub-id>, PMID: <pub-id pub-id-type="pmid">31843054</pub-id></citation></ref>
<ref id="ref11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Crispell</surname> <given-names>J.</given-names></name> <name><surname>Cassidy</surname> <given-names>S.</given-names></name> <name><surname>Kenny</surname> <given-names>K.</given-names></name> <name><surname>McGrath</surname> <given-names>G.</given-names></name> <name><surname>Warde</surname> <given-names>S.</given-names></name> <name><surname>Cameron</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title><italic>Mycobacterium bovis</italic> genomics reveals transmission of infection between cattle and deer in Ireland. <italic>Microbial</italic></article-title>. <source>Genomics</source> <volume>6</volume>:<fpage>mgen000388</fpage>. doi: <pub-id pub-id-type="doi">10.1099/mgen.0.000388</pub-id>, PMID: <pub-id pub-id-type="pmid">32553050</pub-id></citation></ref>
<ref id="ref12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Crispell</surname> <given-names>J.</given-names></name> <name><surname>Zadoks</surname> <given-names>R. N.</given-names></name> <name><surname>Harris</surname> <given-names>S. R.</given-names></name> <name><surname>Paterson</surname> <given-names>B.</given-names></name> <name><surname>Collins</surname> <given-names>D. M.</given-names></name> <name><surname>de-Lisle</surname> <given-names>G. W.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>Using whole genome sequencing to investigate transmission in a multi-host system: bovine tuberculosis in New Zealand</article-title>. <source>BMC Genomics</source> <volume>18</volume>:<fpage>180</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s12864-017-3569-x</pub-id>, PMID: <pub-id pub-id-type="pmid">28209138</pub-id></citation></ref>
<ref id="ref13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Croucher</surname> <given-names>N. J.</given-names></name> <name><surname>Page</surname> <given-names>A. J.</given-names></name> <name><surname>Connor</surname> <given-names>T. R.</given-names></name> <name><surname>Delaney</surname> <given-names>A. J.</given-names></name> <name><surname>Keane</surname> <given-names>J. A.</given-names></name> <name><surname>Bentley</surname> <given-names>S. D.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Rapid phylogenetic analysis of large samples of recombinant bacterial whole genome sequences using Gubbins</article-title>. <source>Nucleic Acids Res.</source> <volume>43</volume>:<fpage>e15</fpage>. doi: <pub-id pub-id-type="doi">10.1093/nar/gku1196</pub-id>, PMID: <pub-id pub-id-type="pmid">25414349</pub-id></citation></ref>
<ref id="ref14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Delignette-Muller</surname> <given-names>M. L.</given-names></name> <name><surname>Dutang</surname> <given-names>C.</given-names></name></person-group> (<year>2015</year>). <article-title>Fitdistrplus: an R Package for fitting distributions</article-title>. <source>J. Stat. Softw.</source> <volume>64</volume>, <fpage>1</fpage>&#x2013;<lpage>34</lpage>. doi: <pub-id pub-id-type="doi">10.18637/jss.v064.i04</pub-id></citation></ref>
<ref id="ref15"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Delogu</surname> <given-names>G.</given-names></name> <name><surname>Brennan</surname> <given-names>M. J.</given-names></name> <name><surname>Manganelli</surname> <given-names>R.</given-names></name></person-group> (<year>2017</year>). <article-title>PE and PPE genes: a tale of conservation and diversity</article-title>. In <person-group person-group-type="editor"><name><surname>Gagneux</surname> <given-names>S.</given-names></name></person-group> (Ed.), <source>Strain Variation in the <italic>Mycobacterium tuberculosis</italic> Complex: Its Role in Biology, Epidemiology and Control</source> (pp. <fpage>191</fpage>&#x2013;<lpage>207</lpage>). <publisher-loc>Cham, Switzerland</publisher-loc>: <publisher-name>Springer International Publishing AG</publisher-name>., PMID: <pub-id pub-id-type="pmid">29116636</pub-id></citation></ref>
<ref id="ref16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dunn</surname> <given-names>O. J.</given-names></name></person-group> (<year>1961</year>). <article-title>Multiple comparisons among means</article-title>. <source>J. Am. Stat. Assoc.</source> <volume>56</volume>, <fpage>52</fpage>&#x2013;<lpage>64</lpage>. doi: <pub-id pub-id-type="doi">10.1080/01621459.1961.10482090</pub-id></citation></ref>
<ref id="ref17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fuente</surname> <given-names>J. d. l.</given-names></name> <name><surname>D&#x00ED;ez-Delgado</surname> <given-names>I.</given-names></name> <name><surname>Contreras</surname> <given-names>M.</given-names></name> <name><surname>Vicente</surname> <given-names>J.</given-names></name> <name><surname>Cabezas-Cruz</surname> <given-names>A.</given-names></name> <name><surname>Tobes</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Comparative genomics of field isolates of <italic>Mycobacterium bovis</italic> and <italic>M. caprae</italic> provides evidence for possible correlates with bacterial viability and virulence</article-title>. <source>PLoS Negl. Trop. Dis.</source> <volume>9</volume>:<fpage>e0004232</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pntd.0004232</pub-id>, PMID: <pub-id pub-id-type="pmid">26583774</pub-id></citation></ref>
<ref id="ref18"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Garrison</surname> <given-names>E.</given-names></name> <name><surname>Marth</surname> <given-names>G.</given-names></name></person-group> (<year>2012</year>). Haplotype-based variant detection from short-read sequencing. <italic>ArXiv: 1207.3907 [q-bio]</italic>. Available at: <ext-link xlink:href="http://arxiv.org/abs/1207.3907" ext-link-type="uri">http://arxiv.org/abs/1207.3907</ext-link> (Accessed August 10, 2022).</citation></ref>
<ref id="ref19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gormley</surname> <given-names>E.</given-names></name> <name><surname>Corner</surname> <given-names>L. A. L.</given-names></name></person-group> (<year>2018</year>). <article-title>Wild animal tuberculosis: stakeholder value systems and management of disease</article-title>. <source>Front. Vet. Sci.</source> <volume>5</volume>:<fpage>327</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fvets.2018.00327</pub-id>, PMID: <pub-id pub-id-type="pmid">30622951</pub-id></citation></ref>
<ref id="ref20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gutierrez</surname> <given-names>M. C.</given-names></name> <name><surname>Brisse</surname> <given-names>S.</given-names></name> <name><surname>Brosch</surname> <given-names>R.</given-names></name> <name><surname>Fabre</surname> <given-names>M.</given-names></name> <name><surname>Oma&#x00EF;s</surname> <given-names>B.</given-names></name> <name><surname>Marmiesse</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2005</year>). <article-title>Ancient origin and gene Mosaicism of the progenitor of <italic>Mycobacterium tuberculosis</italic></article-title>. <source>PLoS Pathog.</source> <volume>1</volume>:<fpage>e5</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.ppat.0010005</pub-id>, PMID: <pub-id pub-id-type="pmid">16201017</pub-id></citation></ref>
<ref id="ref21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hallmaier-Wacker</surname> <given-names>L. K.</given-names></name> <name><surname>Munster</surname> <given-names>V. J.</given-names></name> <name><surname>Knauf</surname> <given-names>S.</given-names></name></person-group> (<year>2017</year>). <article-title>Disease reservoirs: from conceptual frameworks to applicable criteria</article-title>. <source>Emerg. Microbes Infect.</source> <volume>6</volume>:<fpage>e79</fpage>. doi: <pub-id pub-id-type="doi">10.1038/emi.2017.65</pub-id>, PMID: <pub-id pub-id-type="pmid">28874791</pub-id></citation></ref>
<ref id="ref22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Haydon</surname> <given-names>D. T.</given-names></name> <name><surname>Cleaveland</surname> <given-names>S.</given-names></name> <name><surname>Taylor</surname> <given-names>L. H.</given-names></name> <name><surname>Laurenson</surname> <given-names>M. K.</given-names></name></person-group> (<year>2002</year>). <article-title>Identifying reservoirs of infection: a conceptual and practical challenge</article-title>. <source>Emerg. Infect. Dis.</source> <volume>8</volume>, <fpage>1468</fpage>&#x2013;<lpage>1473</lpage>. doi: <pub-id pub-id-type="doi">10.3201/eid0812.010317</pub-id>, PMID: <pub-id pub-id-type="pmid">12498665</pub-id></citation></ref>
<ref id="ref23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hoang</surname> <given-names>D. T.</given-names></name> <name><surname>Chernomor</surname> <given-names>O.</given-names></name> <name><surname>von Haeseler</surname> <given-names>A.</given-names></name> <name><surname>Minh</surname> <given-names>B. Q.</given-names></name> <name><surname>Vinh</surname> <given-names>L. S.</given-names></name></person-group> (<year>2018</year>). <article-title>UFBoot2: improving the ultrafast bootstrap approximation</article-title>. <source>Mol. Biol. Evol.</source> <volume>35</volume>, <fpage>518</fpage>&#x2013;<lpage>522</lpage>. doi: <pub-id pub-id-type="doi">10.1093/molbev/msx281</pub-id>, PMID: <pub-id pub-id-type="pmid">29077904</pub-id></citation></ref>
<ref id="ref24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kalyaanamoorthy</surname> <given-names>S.</given-names></name> <name><surname>Minh</surname> <given-names>B. Q.</given-names></name> <name><surname>Wong</surname> <given-names>T. K. F.</given-names></name> <name><surname>von Haeseler</surname> <given-names>A.</given-names></name> <name><surname>Jermiin</surname> <given-names>L. S.</given-names></name></person-group> (<year>2017</year>). <article-title>ModelFinder: fast model selection for accurate phylogenetic estimates</article-title>. <source>Nat. Methods</source> <volume>14</volume>, <fpage>587</fpage>&#x2013;<lpage>589</lpage>. doi: <pub-id pub-id-type="doi">10.1038/nmeth.4285</pub-id>, PMID: <pub-id pub-id-type="pmid">28481363</pub-id></citation></ref>
<ref id="ref25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kinene</surname> <given-names>T.</given-names></name> <name><surname>Wainaina</surname> <given-names>J.</given-names></name> <name><surname>Maina</surname> <given-names>S.</given-names></name> <name><surname>Boykin</surname> <given-names>L. M.</given-names></name></person-group> (<year>2016</year>). <article-title>Rooting trees, methods for</article-title>. <source>Encyclopedia Evol. Biol.</source>, <fpage>489</fpage>&#x2013;<lpage>493</lpage>. doi: <pub-id pub-id-type="doi">10.1016/B978-0-12-800049-6.00215-8</pub-id></citation></ref>
<ref id="ref26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kuhn</surname> <given-names>M.</given-names></name></person-group> (<year>2008</year>). <article-title>Building predictive models in <italic>R</italic> using the caret package</article-title>. <source>J. Stat. Softw.</source> <volume>28</volume>, <fpage>1</fpage>&#x2013;<lpage>26</lpage>. doi: <pub-id pub-id-type="doi">10.18637/jss.v028.i05</pub-id></citation></ref>
<ref id="ref27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lees</surname> <given-names>J. A.</given-names></name> <name><surname>Galardini</surname> <given-names>M.</given-names></name> <name><surname>Bentley</surname> <given-names>S. D.</given-names></name> <name><surname>Weiser</surname> <given-names>J. N.</given-names></name> <name><surname>Corander</surname> <given-names>J.</given-names></name></person-group> (<year>2018</year>). <article-title>Pyseer: a comprehensive tool for microbial pangenome-wide association studies</article-title>. <source>Bioinformatics</source> <volume>34</volume>, <fpage>4310</fpage>&#x2013;<lpage>4312</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/bty539</pub-id>, PMID: <pub-id pub-id-type="pmid">30535304</pub-id></citation></ref>
<ref id="ref28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lekko</surname> <given-names>Y. M.</given-names></name> <name><surname>Ooi</surname> <given-names>P. T.</given-names></name> <name><surname>Omar</surname> <given-names>S.</given-names></name> <name><surname>Mazlan</surname> <given-names>M.</given-names></name> <name><surname>Ramanoon</surname> <given-names>S. Z.</given-names></name> <name><surname>Jasni</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title><italic>Mycobacterium tuberculosis</italic> complex in wildlife: review of current applications of antemortem and postmortem diagnosis</article-title>. <source>Vet. World</source> <volume>13</volume>, <fpage>1822</fpage>&#x2013;<lpage>1836</lpage>. doi: <pub-id pub-id-type="doi">10.14202/vetworld.2020.1822-1836</pub-id>, PMID: <pub-id pub-id-type="pmid">33132593</pub-id></citation></ref>
<ref id="ref500"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Durbin</surname> <given-names>R.</given-names></name></person-group> (<year>2009</year>). <article-title>Fast and accurate short read alignment with Burrows&#x2013;Wheeler transform</article-title>. <source>Bioinformatics</source> <volume>25</volume>, <fpage>1754</fpage>&#x2013;<lpage>1760</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btp324</pub-id>, PMID: <pub-id pub-id-type="pmid">30622951</pub-id></citation></ref>
<ref id="ref29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>H.</given-names></name></person-group> (<year>2011</year>). <article-title>A statistical framework for SNP calling, mutation discovery, association mapping and population genetical parameter estimation from sequencing data</article-title>. <source>Bioinformatics</source> <volume>27</volume>, <fpage>2987</fpage>&#x2013;<lpage>2993</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btr509</pub-id>, PMID: <pub-id pub-id-type="pmid">21903627</pub-id></citation></ref>
<ref id="ref30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Loiseau</surname> <given-names>C.</given-names></name> <name><surname>Menardo</surname> <given-names>F.</given-names></name> <name><surname>Aseffa</surname> <given-names>A.</given-names></name> <name><surname>Hailu</surname> <given-names>E.</given-names></name> <name><surname>Gumi</surname> <given-names>B.</given-names></name> <name><surname>Ameni</surname> <given-names>G.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>An African origin for <italic>Mycobacterium bovis</italic></article-title>. <source>Evol. Med. Public Health</source> <volume>2020</volume>, <fpage>49</fpage>&#x2013;<lpage>59</lpage>. doi: <pub-id pub-id-type="doi">10.1093/emph/eoaa005</pub-id>, PMID: <pub-id pub-id-type="pmid">32211193</pub-id></citation></ref>
<ref id="ref31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ly-Trong</surname> <given-names>N.</given-names></name> <name><surname>Naser-Khdour</surname> <given-names>S.</given-names></name> <name><surname>Lanfear</surname> <given-names>R.</given-names></name> <name><surname>Minh</surname> <given-names>B. Q.</given-names></name></person-group> (<year>2022</year>). <article-title>AliSim: a fast and versatile phylogenetic sequence simulator for the genomic era</article-title>. <source>Mol. Biol. Evol.</source> <volume>39</volume>, <fpage>msac092</fpage>. doi: <pub-id pub-id-type="doi">10.1093/molbev/msac092</pub-id>, PMID: <pub-id pub-id-type="pmid">35511713</pub-id></citation></ref>
<ref id="ref32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>McCluskey</surname> <given-names>B.</given-names></name> <name><surname>Lombard</surname> <given-names>J.</given-names></name> <name><surname>Strunk</surname> <given-names>S.</given-names></name> <name><surname>Nelson</surname> <given-names>D.</given-names></name> <name><surname>Robbe-Austerman</surname> <given-names>S.</given-names></name> <name><surname>Naugle</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title><italic>Mycobacterium bovis</italic> in California dairies: a case series of 2002&#x2013;2013 outbreaks</article-title>. <source>Prev. Vet. Med.</source> <volume>115</volume>, <fpage>205</fpage>&#x2013;<lpage>216</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.prevetmed.2014.04.010</pub-id>, PMID: <pub-id pub-id-type="pmid">24856878</pub-id></citation></ref>
<ref id="ref33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mehta</surname> <given-names>H. H.</given-names></name> <name><surname>Prater</surname> <given-names>A. G.</given-names></name> <name><surname>Beabout</surname> <given-names>K.</given-names></name> <name><surname>Elworth</surname> <given-names>R. A. L.</given-names></name> <name><surname>Karavis</surname> <given-names>M.</given-names></name> <name><surname>Gibbons</surname> <given-names>H. S.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>The essential role of hypermutation in rapid adaptation to antibiotic stress</article-title>. <source>Antimicrob. Agents Chemother.</source> <volume>63</volume>, <fpage>e00744</fpage>&#x2013;<lpage>e00719</lpage>. doi: <pub-id pub-id-type="doi">10.1128/AAC.00744-19</pub-id>, PMID: <pub-id pub-id-type="pmid">31036684</pub-id></citation></ref>
<ref id="ref34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Milian-Suazo</surname> <given-names>F.</given-names></name> <name><surname>Garcia-Casanova</surname> <given-names>L.</given-names></name> <name><surname>Robbe-Austerman</surname> <given-names>S.</given-names></name> <name><surname>Canto-Alarcon</surname> <given-names>G. J.</given-names></name> <name><surname>Barcenas-Reyes</surname> <given-names>I.</given-names></name> <name><surname>Stuber</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>Molecular relationship between strains of <italic>M. bovis</italic> from Mexico and those from countries with free trade of cattle with Mexico</article-title>. <source>PLoS One</source> <volume>11</volume>:<fpage>e0155207</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0155207</pub-id>, PMID: <pub-id pub-id-type="pmid">27171239</pub-id></citation></ref>
<ref id="ref35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Minh</surname> <given-names>B. Q.</given-names></name> <name><surname>Schmidt</surname> <given-names>H. A.</given-names></name> <name><surname>Chernomor</surname> <given-names>O.</given-names></name> <name><surname>Schrempf</surname> <given-names>D.</given-names></name> <name><surname>Woodhams</surname> <given-names>M. D.</given-names></name> <name><surname>von Haeseler</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>IQ-TREE 2: new models and efficient methods for phylogenetic inference in the genomic era</article-title>. <source>Mol. Biol. Evol.</source> <volume>37</volume>, <fpage>1530</fpage>&#x2013;<lpage>1534</lpage>. doi: <pub-id pub-id-type="doi">10.1093/molbev/msaa015</pub-id>, PMID: <pub-id pub-id-type="pmid">32011700</pub-id></citation></ref>
<ref id="ref36"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>M&#x00FC;ller</surname> <given-names>B.</given-names></name> <name><surname>Hilty</surname> <given-names>M.</given-names></name> <name><surname>Berg</surname> <given-names>S.</given-names></name> <name><surname>Garcia-Pelayo</surname> <given-names>M. C.</given-names></name> <name><surname>Dale</surname> <given-names>J.</given-names></name> <name><surname>Boschiroli</surname> <given-names>M. L.</given-names></name> <etal/></person-group>. (<year>2009</year>). <article-title>African 1, an epidemiologically important clonal complex of <italic>Mycobacterium bovis</italic> dominant in Mali, Nigeria, Cameroon, and Chad</article-title>. <source>J. Bacteriol.</source> <volume>191</volume>, <fpage>1951</fpage>&#x2013;<lpage>1960</lpage>. doi: <pub-id pub-id-type="doi">10.1128/JB.01590-08</pub-id>, PMID: <pub-id pub-id-type="pmid">19136597</pub-id></citation></ref>
<ref id="ref37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Orloski</surname> <given-names>K.</given-names></name> <name><surname>Robbe-Austerman</surname> <given-names>S.</given-names></name> <name><surname>Stuber</surname> <given-names>T.</given-names></name> <name><surname>Hench</surname> <given-names>B.</given-names></name> <name><surname>Schoenbaum</surname> <given-names>M.</given-names></name></person-group> (<year>2018</year>). <article-title>Whole genome sequencing of <italic>Mycobacterium bovis</italic> isolated from livestock in the United States, 1989&#x2013;2018</article-title>. <source>Front. Vet. Sci.</source> <volume>5</volume>:<fpage>253</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fvets.2018.00253</pub-id>, PMID: <pub-id pub-id-type="pmid">30425994</pub-id></citation></ref>
<ref id="ref38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Page</surname> <given-names>A. J.</given-names></name> <name><surname>Taylor</surname> <given-names>B.</given-names></name> <name><surname>Delaney</surname> <given-names>A. J.</given-names></name> <name><surname>Soares</surname> <given-names>J.</given-names></name> <name><surname>Seemann</surname> <given-names>T.</given-names></name> <name><surname>Keane</surname> <given-names>J. A.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>SNP-sites: rapid efficient extraction of SNPs from multi-FASTA alignments</article-title>. <source>Microb. Genom.</source> <volume>2</volume>:<fpage>e000056</fpage>. doi: <pub-id pub-id-type="doi">10.1099/mgen.0.000056</pub-id>, PMID: <pub-id pub-id-type="pmid">28348851</pub-id></citation></ref>
<ref id="ref39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Palmer</surname> <given-names>M. V.</given-names></name></person-group> (<year>2013</year>). <article-title><italic>Mycobacterium bovis</italic>: characteristics of wildlife reservoir hosts</article-title>. <source>Transbound. Emerg. Dis.</source> <volume>60</volume>, <fpage>1</fpage>&#x2013;<lpage>13</lpage>. doi: <pub-id pub-id-type="doi">10.1111/tbed.12115</pub-id>, PMID: <pub-id pub-id-type="pmid">24171844</pub-id></citation></ref>
<ref id="ref40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Palmer</surname> <given-names>M. V.</given-names></name> <name><surname>Thacker</surname> <given-names>T. C.</given-names></name> <name><surname>Waters</surname> <given-names>W. R.</given-names></name> <name><surname>Gort&#x00E1;zar</surname> <given-names>C.</given-names></name> <name><surname>Corner</surname> <given-names>L. A. L.</given-names></name></person-group> (<year>2012</year>). <article-title><italic>Mycobacterium bovis</italic>: a model pathogen at the interface of livestock, wildlife, and humans</article-title>. <source>Vet. Med. Int.</source> <volume>2012</volume>, <fpage>1</fpage>&#x2013;<lpage>17</lpage>. doi: <pub-id pub-id-type="doi">10.1155/2012/236205</pub-id>, PMID: <pub-id pub-id-type="pmid">22737588</pub-id></citation></ref>
<ref id="ref41"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Patan&#x00E9;</surname> <given-names>J. S. L.</given-names></name> <name><surname>Martins</surname> <given-names>J.</given-names></name> <name><surname>Castel&#x00E3;o</surname> <given-names>A. B.</given-names></name> <name><surname>Nishibe</surname> <given-names>C.</given-names></name> <name><surname>Montera</surname> <given-names>L.</given-names></name> <name><surname>Bigi</surname> <given-names>F.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>Patterns and processes of <italic>Mycobacterium bovis</italic> evolution revealed by phylogenomic analyses</article-title>. <source>Genome Biol. Evol.</source> <volume>9</volume>, <fpage>521</fpage>&#x2013;<lpage>535</lpage>. doi: <pub-id pub-id-type="doi">10.1093/gbe/evx022</pub-id>, PMID: <pub-id pub-id-type="pmid">28201585</pub-id></citation></ref>
<ref id="ref42"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pattengale</surname> <given-names>N. D.</given-names></name> <name><surname>Gottlieb</surname> <given-names>E. J.</given-names></name> <name><surname>Moret</surname> <given-names>B. M. E.</given-names></name></person-group> (<year>2007</year>). <article-title>Efficiently computing the Robinson-Foulds metric</article-title>. <source>J. Comput. Biol.</source> <volume>14</volume>, <fpage>724</fpage>&#x2013;<lpage>735</lpage>. doi: <pub-id pub-id-type="doi">10.1089/cmb.2007.R012</pub-id>, PMID: <pub-id pub-id-type="pmid">17691890</pub-id></citation></ref>
<ref id="ref43"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Quinlan</surname> <given-names>A. R.</given-names></name> <name><surname>Hall</surname> <given-names>I. M.</given-names></name></person-group> (<year>2010</year>). <article-title>BEDTools: a flexible suite of utilities for comparing genomic features</article-title>. <source>Bioinformatics</source> <volume>26</volume>, <fpage>841</fpage>&#x2013;<lpage>842</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btq033</pub-id>, PMID: <pub-id pub-id-type="pmid">20110278</pub-id></citation></ref>
<ref id="ref600"><citation citation-type="other"><person-group person-group-type="author"><collab id="coll2999">R Core Team</collab></person-group> (<year>2022</year>). R: A language and environment for statistical computing. R Foundation for Statistical Computing, Vienna, Austria. Available at: <ext-link xlink:href="https://www.R-project.org/" ext-link-type="uri">https://www.R-project.org/</ext-link></citation></ref>
<ref id="ref44"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Reis</surname> <given-names>A. C.</given-names></name> <name><surname>Cunha</surname> <given-names>M. V.</given-names></name></person-group> (<year>2021</year>). <article-title>Genome-wide estimation of recombination, mutation and positive selection enlightens diversification drivers of <italic>Mycobacterium bovis</italic></article-title>. <source>Sci. Rep.</source> <volume>11</volume>:<fpage>18789</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41598-021-98226-y</pub-id>, PMID: <pub-id pub-id-type="pmid">34552144</pub-id></citation></ref>
<ref id="ref45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rodrigues</surname> <given-names>R. d. A.</given-names></name> <name><surname>Ribeiro Ara&#x00FA;jo</surname> <given-names>F.</given-names></name> <name><surname>Rivera D&#x00E1;vila</surname> <given-names>A. M.</given-names></name> <name><surname>Etges</surname> <given-names>R. N.</given-names></name> <name><surname>Parkhill</surname> <given-names>J.</given-names></name> <name><surname>van Tonder</surname> <given-names>A. J.</given-names></name></person-group> (<year>2021</year>). <article-title>Genomic and temporal analyses of <italic>Mycobacterium bovis</italic> in southern Brazil</article-title>. <source>Microb. Genom.</source> <volume>7</volume>:<fpage>000569</fpage>. doi: <pub-id pub-id-type="doi">10.1099/mgen.0.000569</pub-id>, PMID: <pub-id pub-id-type="pmid">34016251</pub-id></citation></ref>
<ref id="ref46"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rodriguez-Campos</surname> <given-names>S.</given-names></name> <name><surname>Sch&#x00FC;rch</surname> <given-names>A. C.</given-names></name> <name><surname>Dale</surname> <given-names>J.</given-names></name> <name><surname>Lohan</surname> <given-names>A. J.</given-names></name> <name><surname>Cunha</surname> <given-names>M. V.</given-names></name> <name><surname>Botelho</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>European 2: a clonal complex of <italic>Mycobacterium bovis</italic> dominant in the Iberian Peninsula</article-title>. <source>Infect. Genet. Evol.</source> <volume>12</volume>, <fpage>866</fpage>&#x2013;<lpage>872</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.meegid.2011.09.004</pub-id>, PMID: <pub-id pub-id-type="pmid">21945286</pub-id></citation></ref>
<ref id="ref47"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rossi</surname> <given-names>G.</given-names></name> <name><surname>Crispell</surname> <given-names>J.</given-names></name> <name><surname>Brough</surname> <given-names>T.</given-names></name> <name><surname>Lycett</surname> <given-names>S. J.</given-names></name> <name><surname>White</surname> <given-names>P. C. L.</given-names></name> <name><surname>Allen</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Phylodynamic analysis of an emergent <italic>Mycobacterium bovis</italic> outbreak in an area with no previously known wildlife infections</article-title>. <source>J. Appl. Ecol.</source> <volume>59</volume>, <fpage>210</fpage>&#x2013;<lpage>222</lpage>. doi: <pub-id pub-id-type="doi">10.1111/1365-2664.14046</pub-id></citation></ref>
<ref id="ref48"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Salvador</surname> <given-names>L. C. M.</given-names></name> <name><surname>O&#x2019;Brien</surname> <given-names>D. J.</given-names></name> <name><surname>Cosgrove</surname> <given-names>M. K.</given-names></name> <name><surname>Stuber</surname> <given-names>T. P.</given-names></name> <name><surname>Schooley</surname> <given-names>A. M.</given-names></name> <name><surname>Crispell</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Disease management at the wildlife-livestock interface: using whole-genome sequencing to study the role of elk in <italic>Mycobacterium bovis</italic> transmission in Michigan, USA</article-title>. <source>Mol. Ecol.</source> <volume>28</volume>, <fpage>2192</fpage>&#x2013;<lpage>2205</lpage>. doi: <pub-id pub-id-type="doi">10.1111/mec.15061</pub-id>, PMID: <pub-id pub-id-type="pmid">30807679</pub-id></citation></ref>
<ref id="ref49"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sampson</surname> <given-names>S. L.</given-names></name></person-group> (<year>2011</year>). <article-title>Mycobacterial PE/PPE proteins at the host-pathogen Interface</article-title>. <source>Clin. Dev. Immunol.</source> <volume>2011</volume>:<fpage>497203</fpage>, <fpage>1</fpage>&#x2013;<lpage>11</lpage>. doi: <pub-id pub-id-type="doi">10.1155/2011/497203</pub-id>, PMID: <pub-id pub-id-type="pmid">21318182</pub-id></citation></ref>
<ref id="ref50"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sheppard</surname> <given-names>S. K.</given-names></name> <name><surname>Guttman</surname> <given-names>D. S.</given-names></name> <name><surname>Fitzgerald</surname> <given-names>J. R.</given-names></name></person-group> (<year>2018</year>). <article-title>Population genomics of bacterial host adaptation</article-title>. <source>Nat. Rev. Genet.</source> <volume>19</volume>, <fpage>549</fpage>&#x2013;<lpage>565</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41576-018-0032-z</pub-id></citation></ref>
<ref id="ref51"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Smith</surname> <given-names>N. H.</given-names></name> <name><surname>Berg</surname> <given-names>S.</given-names></name> <name><surname>Dale</surname> <given-names>J.</given-names></name> <name><surname>Allen</surname> <given-names>A.</given-names></name> <name><surname>Rodriguez</surname> <given-names>S.</given-names></name> <name><surname>Romero</surname> <given-names>B.</given-names></name> <etal/></person-group>. (<year>2011</year>). <article-title>European 1: a globally important clonal complex of <italic>Mycobacterium bovis</italic></article-title>. <source>Infect. Genet. Evol.</source> <volume>11</volume>, <fpage>1340</fpage>&#x2013;<lpage>1351</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.meegid.2011.04.027</pub-id>, PMID: <pub-id pub-id-type="pmid">21571099</pub-id></citation></ref>
<ref id="ref52"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Stephan</surname> <given-names>W.</given-names></name></person-group> (<year>2019</year>). <article-title>Selective sweeps</article-title>. <source>Genetics</source> <volume>211</volume>, <fpage>5</fpage>&#x2013;<lpage>13</lpage>. doi: <pub-id pub-id-type="doi">10.1534/genetics.118.301319</pub-id>, PMID: <pub-id pub-id-type="pmid">30626638</pub-id></citation></ref>
<ref id="ref53"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Swings</surname> <given-names>T.</given-names></name> <name><surname>Van den Bergh</surname> <given-names>B.</given-names></name> <name><surname>Wuyts</surname> <given-names>S.</given-names></name> <name><surname>Oeyen</surname> <given-names>E.</given-names></name> <name><surname>Voordeckers</surname> <given-names>K.</given-names></name> <name><surname>Verstrepen</surname> <given-names>K. J.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>Adaptive tuning of mutation rates allows fast response to lethal stress in <italic>Escherichia coli</italic></article-title>. <source>elife</source> <volume>6</volume>:<fpage>e22939</fpage>. doi: <pub-id pub-id-type="doi">10.7554/eLife.22939</pub-id>, PMID: <pub-id pub-id-type="pmid">28460660</pub-id></citation></ref>
<ref id="ref54"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tajima</surname> <given-names>F.</given-names></name></person-group> (<year>1991</year>). <article-title>Determination of window size for analyzing DNA sequences</article-title>. <source>J. Mol. Evol.</source> <volume>33</volume>, <fpage>470</fpage>&#x2013;<lpage>473</lpage>. doi: <pub-id pub-id-type="doi">10.1007/BF02103140</pub-id>, PMID: <pub-id pub-id-type="pmid">1960744</pub-id></citation></ref>
<ref id="ref55"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tonder</surname> <given-names>A. J. v.</given-names></name> <name><surname>Thornton</surname> <given-names>M. J.</given-names></name> <name><surname>Conlan</surname> <given-names>A. J. K.</given-names></name> <name><surname>Jolley</surname> <given-names>K. A.</given-names></name> <name><surname>Goolding</surname> <given-names>L.</given-names></name> <name><surname>Mitchell</surname> <given-names>A. P.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Inferring <italic>Mycobacterium bovis</italic> transmission between cattle and badgers using isolates from the randomised badger culling trial</article-title>. <source>PLoS Pathog.</source> <volume>17</volume>:<fpage>e1010075</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.ppat.1010075</pub-id>, PMID: <pub-id pub-id-type="pmid">34843579</pub-id></citation></ref>
<ref id="ref56"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tonkin-Hill</surname> <given-names>G.</given-names></name> <name><surname>Lees</surname> <given-names>J. A.</given-names></name> <name><surname>Bentley</surname> <given-names>S. D.</given-names></name> <name><surname>Frost</surname> <given-names>S. D. W.</given-names></name> <name><surname>Corander</surname> <given-names>J.</given-names></name></person-group> (<year>2019</year>). <article-title>Fast hierarchical Bayesian analysis of population structure</article-title>. <source>Nucleic Acids Res.</source> <volume>47</volume>, <fpage>5539</fpage>&#x2013;<lpage>5549</lpage>. doi: <pub-id pub-id-type="doi">10.1093/nar/gkz361</pub-id>, PMID: <pub-id pub-id-type="pmid">31076776</pub-id></citation></ref>
<ref id="ref57"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tria</surname> <given-names>F. D. K.</given-names></name> <name><surname>Landan</surname> <given-names>G.</given-names></name> <name><surname>Dagan</surname> <given-names>T.</given-names></name></person-group> (<year>2017</year>). <article-title>Phylogenetic rooting using minimal ancestor deviation</article-title>. <source>Nat. Ecol. Evol.</source> <volume>1</volume>, <fpage>1</fpage>&#x2013;<lpage>7</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41559-017-0193</pub-id>, PMID: <pub-id pub-id-type="pmid">29388565</pub-id></citation></ref>
<ref id="ref58"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Viana</surname> <given-names>M.</given-names></name> <name><surname>Mancy</surname> <given-names>R.</given-names></name> <name><surname>Biek</surname> <given-names>R.</given-names></name> <name><surname>Cleaveland</surname> <given-names>S.</given-names></name> <name><surname>Cross</surname> <given-names>P. C.</given-names></name> <name><surname>Lloyd-Smith</surname> <given-names>J. O.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>Assembling evidence for identifying reservoirs of infection</article-title>. <source>Trends Ecol. Evol.</source> <volume>29</volume>, <fpage>270</fpage>&#x2013;<lpage>279</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.tree.2014.03.002</pub-id>, PMID: <pub-id pub-id-type="pmid">24726345</pub-id></citation></ref>
<ref id="ref59"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zimpel</surname> <given-names>C. K.</given-names></name> <name><surname>Patan&#x00E9;</surname> <given-names>J. S. L.</given-names></name> <name><surname>Guedes</surname> <given-names>A. C. P.</given-names></name> <name><surname>de Souza</surname> <given-names>R. F.</given-names></name> <name><surname>Silva-Pereira</surname> <given-names>T. T.</given-names></name> <name><surname>Camargo</surname> <given-names>N. C. S.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Global distribution and evolution of <italic>Mycobacterium bovis</italic> lineages</article-title>. <source>Front. Microbiol.</source> <volume>11</volume>:<fpage>843</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fmicb.2020.00843</pub-id>, PMID: <pub-id pub-id-type="pmid">32477295</pub-id></citation></ref></ref-list>
</back>
</article>