<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<?covid-19-tdm?>
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="methods-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Microbiol.</journal-id>
<journal-title>Frontiers in Microbiology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Microbiol.</abbrev-journal-title>
<issn pub-type="epub">1664-302X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmicb.2022.859241</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Microbiology</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A New Way to Trace SARS-CoV-2 Variants Through Weighted Network Analysis of Frequency Trajectories of Mutations</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Huang</surname> <given-names>Qiang</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x2020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1701520/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhang</surname> <given-names>Qiang</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x2020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1671705/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Bible</surname> <given-names>Paul W.</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1701970/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Liang</surname> <given-names>Qiaoxing</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1550163/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Zheng</surname> <given-names>Fangfang</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1702423/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Wang</surname> <given-names>Ying</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1702376/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Hao</surname> <given-names>Yuantao</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/708605/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Liu</surname> <given-names>Yu</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1644573/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Medical Statistics and Epidemiology, School of Public Health, Sun Yat-sen University</institution>, <addr-line>Guangzhou</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>College of Computer, Chengdu University</institution>, <addr-line>Chengdu</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>College of Arts and Sciences, Marian University</institution>, <addr-line>Indianapolis, IN</addr-line>, <country>United States</country></aff>
<aff id="aff4"><sup>4</sup><institution>State Key Laboratory of Ophthalmology, Zhongshan Ophthalmic Center, Sun Yat-sen University</institution>, <addr-line>Guangzhou</addr-line>, <country>China</country></aff>
<aff id="aff5"><sup>5</sup><institution>School of Traditional Chinese Medicine Healthcare, Guangdong Food and Drug Vocational College</institution>, <addr-line>Guangzhou</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Pragya Dhruv Yadav, National Institute of Virology (ICMR), India</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Anna Bernasconi, Politecnico di Milano, Italy; Arbind Kumar Patel, Indian Institute of Technology Gandhinagar, India</p></fn>
<corresp id="c001">&#x002A;Correspondence: Yuantao Hao, <email>haoyt@mail.sysu.edu.cn</email></corresp>
<corresp id="c002">Yu Liu, <email>liuy683@mail.sysu.edu.cn</email></corresp>
<fn fn-type="equal" id="fn002"><p><sup>&#x2020;</sup>These authors have contributed equally to this work</p></fn>
<fn fn-type="other" id="fn004"><p>This article was submitted to Virology, a section of the journal Frontiers in Microbiology</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>16</day>
<month>03</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>859241</elocation-id>
<history>
<date date-type="received">
<day>21</day>
<month>01</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>18</day>
<month>02</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2022 Huang, Zhang, Bible, Liang, Zheng, Wang, Hao and Liu.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Huang, Zhang, Bible, Liang, Zheng, Wang, Hao and Liu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Early detection of SARS-CoV-2 variants enables timely tracking of clinically important strains in order to inform the public health response. Current subtype-based variant surveillance depending on prior subtype assignment according to lag features and their continuous risk assessment may delay this process. We proposed a weighted network framework to model the frequency trajectories of mutations (FTMs) for SARS-CoV-2 variant tracing, without requiring prior subtype assignment. This framework modularizes the FTMs and conglomerates synchronous FTMs together to represent the variants. It also generates module clusters to unveil the epidemic stages and their contemporaneous variants. Eventually, the module-based variants are assessed by phylogenetic tree through sub-sampling to facilitate communication and control of the epidemic. This process was benchmarked using worldwide GISAID data, which not only demonstrated all the methodology features but also showed the module-based variant identification had highly specific and sensitive mapping with the global phylogenetic tree. When applying this process to regional data like India and South Africa for SARS-CoV-2 variant surveillance, the approach clearly elucidated the national dispersal history of the viral variants and their co-circulation pattern, and provided much earlier warning of Beta (B.1.351), Delta (B.1.617.2), and Omicron (B.1.1.529). In summary, our work showed that the weighted network modeling of FTMs enables us to rapidly and easily track down SARS-CoV-2 variants overcoming prior viral subtyping with lag features, accelerating the understanding and surveillance of COVID-19.</p>
</abstract>
<kwd-group>
<kwd>SARS-CoV-2</kwd>
<kwd>mutations</kwd>
<kwd>frequency trajectories</kwd>
<kwd>weighted network analysis</kwd>
<kwd>variant tracing</kwd>
</kwd-group>
<contract-sponsor id="cn001">National Natural Science Foundation of China<named-content content-type="fundref-id">10.13039/501100001809</named-content></contract-sponsor>
<contract-sponsor id="cn002">Basic and Applied Basic Research Foundation of Guangdong Province<named-content content-type="fundref-id">10.13039/501100021171</named-content></contract-sponsor>
<counts>
<fig-count count="5"/>
<table-count count="0"/>
<equation-count count="4"/>
<ref-count count="37"/>
<page-count count="12"/>
<word-count count="6625"/>
</counts>
</article-meta>
</front>
<body>
<sec id="S1" sec-type="intro">
<title>Introduction</title>
<p>The severe acute respiratory syndrome-related coronavirus 2 (SARS-CoV-2) causing coronavirus disease 2019 (COVID-19) has been running rampant all over the world since December 2019. The current pandemic has triggered an unprecedented scale of whole-genome sequencing and sharing of the virus&#x2019;s genome. Surveillance of SARS-CoV-2 variants using sequence data provides insight into disease virulence, pathogenesis, host range or immune escape, as well as the effectiveness of SARS-CoV-2 diagnostics and therapeutics (<xref ref-type="bibr" rid="B5">Grubaugh et al., 2021</xref>; <xref ref-type="bibr" rid="B27">Tegally et al., 2021</xref>). Viral subtyping methods such as GISAID (<xref ref-type="bibr" rid="B7">Han et al., 2019</xref>), Pangolin (<xref ref-type="bibr" rid="B22">Rambaut et al., 2020</xref>) and CMM (<xref ref-type="bibr" rid="B21">Qin et al., 2021</xref>) have greatly aided this process. Designating a subtype (e.g., lineage) for each genome according to predetermined genetic features (e.g., mutations) followed by continuous risk assessment of these subtypes serves to identify clinically important emerging variants. However, subtype assignment depends on lag features that may delay the detection of newly emerging variants or the descendants of circulating variants. In addition, a too detailed subtyping (e.g., Pangolin) of the SARS-CoV-2 population has resulted in excess burden on risk monitoring while a rough categorization (e.g., GISAID) delays the detection and communication of dangerous variants (<xref ref-type="bibr" rid="B20">Oude Munnink et al., 2021</xref>; <xref ref-type="bibr" rid="B21">Qin et al., 2021</xref>; <xref ref-type="bibr" rid="B26">Tang et al., 2021</xref>).</p>
<p>It is well known that new SARS-CoV-2 variants with their specific mutation features gradually dominate through spatial and temporal expansion (<xref ref-type="bibr" rid="B16">Mascola et al., 2021</xref>). The frequencies of different mutations throughout the viral genome can now be tracked over time with high resolution and reliability. Mutations with synchronous frequency trajectories are likely to define a variant or a group of variants (<xref ref-type="bibr" rid="B37">Zhao et al., 2020</xref>; <xref ref-type="bibr" rid="B1">Bernasconi et al., 2021</xref>; <xref ref-type="bibr" rid="B21">Qin et al., 2021</xref>). Thereby, the frequency trajectories of mutations (FTMs) contain information that could allow very sensitive detection of prevalent mutations highlighting important variants, e.g., variants under investigation (VUI) or variants of concern (VOC). Leveraging FTMs to develop new analytics will allow truly real-time surveillance of SARS-CoV-2 variants and improve the lead time for public health interventions.</p>
<p>In this paper, we developed a module-based variant surveillance method that enables real-time tracking of historical and circulating SARS-CoV-2 variants without designating their subtypes in advance allowing newly emerging variants or the descendants of circulating variants to be tracked earlier. This method views mutations represented by FTMs as nodes of a network and describes their relationships using network connections. We found that closely connected nodes in the network forming a biologically meaningful module indicate a potential variant, and module clusters indicate potential contemporaneous variants. We demonstrate the FTM network construction and interpretation through analysis of worldwide data of SARS-CoV-2 genomes and validate its variant surveillance capability <italic>via</italic> tracking the variants circulating in two COVID-19 hotspots, India and South Africa.</p>
</sec>
<sec id="S2" sec-type="materials|methods">
<title>Materials and Methods</title>
<p>A comparison of the workflows between subtype-based and FTM-based variant surveillance methods has been shown in <xref ref-type="fig" rid="F1">Figure 1A</xref>. The outline of our FTM-based SARS-CoV-2 variant identification framework using weighted network modeling is shown in <xref ref-type="fig" rid="F1">Figure 1B</xref>. This framework uses FTMs as an input and is comprised of the following main steps: sequence curation, mutation calling, calculation, and filtering of FTMs, network construction, variant identification and determination using core mutations, and variant validation. We used the worldwide data and the pandemic variants (<xref ref-type="supplementary-material" rid="TS1">Supplementary Table 1</xref>) as a benchmark and further illustrate the surveillance features of our method using regional data from India and South Africa. Below, we focus on the delineation of each step.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption><p>Workflows for SARS-CoV-2 variant surveillance. <bold>(A)</bold> Workflow comparison between subtype-based and FTM-based variant surveillance methods. <bold>(B)</bold> Outline of a weighted network framework for variant surveillance using FTMs.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-13-859241-g001.tif"/>
</fig>
<sec id="S2.SS1">
<title>Data Curation</title>
<p>SARS-CoV-2 genomes were retrieved from GISAID database (<xref ref-type="bibr" rid="B24">Shu and McCauley, 2017</xref>). Only viruses from human submitted before 2021-11-30 with sample collection date between 2020-01-05 and 2021-11-27 were extracted, filtering sequences with flags, &#x201C;complete sample collection date&#x201D;, &#x201C;complete genome&#x201D; (genome length &#x003E; 29,000 bp) and &#x201C;low coverage excluded&#x201D; (exclude genomes with &#x003E; 5% Ns). Consequently, a total of 5,043,950 genomes were collected. Because significant sampling date errors were found in metadata of some genomes (<xref ref-type="supplementary-material" rid="FS1">Supplementary Figure 1</xref>), they were firstly excluded from downstream analysis according to their mutation numbers (see below).</p>
</sec>
<sec id="S2.SS2">
<title>Bioinformatic Analysis for Mutation Calling</title>
<p>Whole genome genetic variations, including single nucleotide polymorphisms (SNPs) and insertions/deletions (INDELs), were determined and annotated using a bioinformatic framework proposed by <xref ref-type="bibr" rid="B17">Massacci et al. (2020)</xref> with Wuhan-Hu-1 (GenBank NC_045512.2) (<xref ref-type="bibr" rid="B32">Wu et al., 2020</xref>) as the reference. In summary, the viral sequences were first aligned against the reference using the nucmer command with default settings except requiring only the forward matching of the query sequences (&#x2013;forward), provided by the MUMmer package (version 3.23) (<xref ref-type="bibr" rid="B15">Marcais et al., 2018</xref>). The generated delta encoded alignment files were then parsed by the show-snps command to produce a catalog of all SNPs and INDELs. Show-snps outputs were summarized and translated to proteins using a R script adapted from <xref ref-type="bibr" rid="B18">Mercatelli and Giorgi (2020)</xref>. Eventually, an annotated list including 186,399,389 mutational events was exported. The number of mutational events for each study sample was firstly calculated. Since high mutation numbers are not likely to appear in the early stage of the COVID-19 pandemic (<xref ref-type="supplementary-material" rid="FS1">Supplementary Figure 1</xref>), we excluded genomes with mutation numbers far beyond other samples collected in the same month, where the cutoffs were set to be the average plus 5 standard deviations.</p>
</sec>
<sec id="S2.SS3">
<title>Calculation and Filtration of Frequency Trajectories of Mutations</title>
<p>Mutations present at least once across all genomes were extracted and their frequency time series were generated according to calendar weeks of sampling. Specifically, a mutation frequency, denoted by <italic>y</italic><sub>st</sub>, at a sampling week <italic>t</italic> on a specific site <italic>s</italic> was calculated as the fraction of genomes with the mutation of all genomes sampled at that week. Then the frequency trajectory of a mutation <italic>s</italic>(1 = <italic>s</italic> = <italic>S</italic>) can be denoted as</p>
<disp-formula id="S2.E1">
<label>(1)</label>
<mml:math id="M1">
<mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mtext mathvariant="bold">y</mml:mtext>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>:</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>t</italic> denotes the week number and <italic>t</italic> = 1 represents the first complete calendar week of 2020 (from January 5 to 11, 2020). When aggregating the mutation events for each mutation site, a large amount of multi-directional mutations (e.g., C&#x2192;T and C&#x2192;G) were detected (<xref ref-type="supplementary-material" rid="FS2">Supplementary Figure 2</xref>). All possible mutation directions were considered in our study to allow the distinction of different variant branches (e.g., G23012A for B.1.351 and G23012C for B.1.617.1) and to avoid erroneous clustering in the network construction due to missing mutation directions.</p>
<p>A myriad of mutations (i.e., large <italic>S</italic>) were detected across the viral genome but most were less informative with the temporal frequency pattern of fluctuating near zero (<xref ref-type="supplementary-material" rid="FS3">Supplementary Figure 3</xref>). Therefore, FTMs with all mutation frequencies less than a threshold [e.g., 1%, a threshold above which a mutation is considered fixed in a natural population (<xref ref-type="bibr" rid="B30">Wong et al., 2003</xref>)] were first excluded. For worldwide data, 1,178 (1.4%) were kept after this filtration and the majority of these FTMs maintained a frequency of &#x2265;1% only for a limited period, as described by <xref ref-type="bibr" rid="B3">Chiara et al. (2021)</xref>. To facilitate the demonstration of the methodology features, a hierarchical clustering analysis using Ward&#x2019;s method (<xref ref-type="bibr" rid="B29">Ward, 1963</xref>) was additionally applied to group and exclude them before investigating the temporal clustering patterns.</p>
</sec>
<sec id="S2.SS4">
<title>Weighted Network Construction</title>
<p>In the network model, nodes correspond to mutations, or more precisely to scaled FTMs with</p>
<disp-formula id="S2.E2">
<label>(2)</label>
<mml:math id="M2">
<mml:mrow>
<mml:msubsup>
<mml:mtext mathvariant="bold">y</mml:mtext>
<mml:mi>s</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mtext mathvariant="bold">y</mml:mtext>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mo>-</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext mathvariant="bold">y</mml:mtext>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext mathvariant="bold">y</mml:mtext>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula><mml:math id="INEQ3"><mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>e</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>a</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>n</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mtext mathvariant="bold">y</mml:mtext><mml:mi>s</mml:mi></mml:msub><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mi>T</mml:mi></mml:mfrac><mml:mo>&#x2062;</mml:mo><mml:mrow><mml:msub><mml:mo largeop="true" symmetric="true">&#x2211;</mml:mo><mml:mi>t</mml:mi></mml:msub><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>s</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="INEQ4"><mml:mrow><mml:mrow><mml:mi>v</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>a</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>r</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mtext mathvariant="bold">y</mml:mtext><mml:mi>s</mml:mi></mml:msub><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mi>T</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mfrac><mml:mo>&#x2062;</mml:mo><mml:mrow><mml:msub><mml:mo largeop="true" symmetric="true">&#x2211;</mml:mo><mml:mi>t</mml:mi></mml:msub><mml:msup><mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>s</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>e</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>a</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>n</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mtext mathvariant="bold">y</mml:mtext><mml:mi>s</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mrow><mml:mo stretchy="false">]</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mrow></mml:mrow></mml:math></inline-formula>. The edges between mutations are determined by the pairwise Pearson correlations between FTMs. Then two FTMs will have a correlation coefficient close to 1 if they are synchronous, and non-synchronous relationships will deviate from 1. The connection strength between mutation <italic>i</italic> and <italic>j</italic> were quantified with an adjacency score using a power function (<xref ref-type="bibr" rid="B8">Horvath, 2011</xref>),</p>
<disp-formula id="S2.E3">
<label>(3)</label>
<mml:math id="M3">
<mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0.5</mml:mn>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mn>0.5</mml:mn>
<mml:mo>&#x22C5;</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msubsup>
<mml:mtext mathvariant="bold">y</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mo rspace="7.5pt">,</mml:mo>
<mml:msubsup>
<mml:mtext mathvariant="bold">y</mml:mtext>
<mml:mi>j</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mi mathvariant="normal">&#x03B2;</mml:mi>
</mml:msup>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where<inline-formula><mml:math id="INEQ5"><mml:mrow><mml:mi>c</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>o</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>r</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>r</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:msubsup><mml:mtext mathvariant="bold">y</mml:mtext><mml:mi>i</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msubsup><mml:mo rspace="7.5pt">,</mml:mo><mml:msubsup><mml:mtext mathvariant="bold">y</mml:mtext><mml:mi>j</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msubsup><mml:mo rspace="7.5pt">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>is the Pearson correlation coefficient between <inline-formula><mml:math id="INEQ6"><mml:msubsup><mml:mtext mathvariant="bold">y</mml:mtext><mml:mi>i</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msubsup></mml:math></inline-formula> and <inline-formula><mml:math id="INEQ7"><mml:msubsup><mml:mtext mathvariant="bold">y</mml:mtext><mml:mi>j</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msubsup></mml:math></inline-formula>. The transformation in the parentheses is applied to map the correlations onto the interval [0, 1] to satisfy the requirement of an adjacency matrix and the exponential transformation with &#x03B2;&#x2265;1 is used to emphasize strong correlations at the expense of weak correlations. This leads to a weighted network and &#x03B2; is determined based on the scale-free topology criterion (<xref ref-type="bibr" rid="B36">Zhang and Horvath, 2005</xref>).</p>
<p>The network connectivity (<italic>k_s</italic>) of the <italic>s</italic>th mutation is the sum of the connection strengths with the other mutations, <italic>k</italic><sub><italic>s</italic></sub> = &#x2211;<sub><italic>i</italic>&#x2260;<italic>s</italic></sub><italic>A</italic><sub><italic>si</italic></sub>. The summation performed over all mutations in a particular module is the intra-modular connectivity (<italic>k</italic><sub>s,intra</sub>).</p>
</sec>
<sec id="S2.SS5">
<title>Network Module Identification</title>
<p>In weighted networks, modules are subsets of mutations which are tightly connected. Identifying these modules facilitates rapid identification and designation of a variant. Since the adjacency between two nodes cannot reflect their connectivity with other intra- or inter-modular nodes, we use a topological overlap measure (TOM) instead. The topological overlap is defined by:</p>
<disp-formula id="S2.E4">
<label>(4)</label>
<mml:math id="M4">
<mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>O</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mtable displaystyle="true" rowspacing="0pt">
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mpadded width="+5pt">
<mml:mstyle displaystyle="false">
<mml:mfrac>
<mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="false">
<mml:msub>
<mml:mo largeop="true" symmetric="true">&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2260;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:msub>
</mml:mstyle>
<mml:mrow>
<mml:msub>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi>min</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:msub>
<mml:mi>k</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>k</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>-</mml:mo>
<mml:msub>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mstyle>
</mml:mpadded>
<mml:mo>&#x2062;</mml:mo>
<mml:mpadded>
<mml:mtext>i</mml:mtext>
<mml:mo>&#x2260;</mml:mo>
<mml:mtext> j</mml:mtext>
</mml:mpadded>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo mathvariant="italic" separator="true">&#x2003;&#x2003;&#x2003;&#x2003;</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mi/>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where &#x2211;<sub><italic>l</italic>&#x2260;<italic>i</italic>,<italic>j</italic></sub><italic>A</italic><sub><italic>il</italic></sub><italic>A</italic><sub><italic>lj</italic></sub> quantifies the indirect connection strengths between <italic>i</italic> and <italic>j</italic> through their shared neighbors and the denominator serves as a normalization factor. The topological overlap between mutation <italic>i</italic> and <italic>j</italic> reflects their relative interconnectedness as mediated through other mutation nodes (<xref ref-type="bibr" rid="B34">Yip and Horvath, 2007</xref>). Module identification was done using the TOM-based dissimilarity matrix diss<italic>TOM</italic> = (1&#x2212;<italic>TOM</italic><sub>ij</sub>) coupled with average linkage hierarchical clustering. Modules corresponded to branches of the resulting hierarchical clustering tree. We used a dynamic cut-tree algorithm to determine the branches (<xref ref-type="bibr" rid="B13">Langfelder et al., 2008</xref>). All of these were realized with the R WGCNA package (<xref ref-type="bibr" rid="B12">Langfelder and Horvath, 2008</xref>).</p>
<p>To intuitively display the relationship between nodes of the weighted network, the topological overlap matrix was partitioned by different cutoffs (e.g., 0.1 or higher) and visualized using the R igraph package (<xref ref-type="bibr" rid="B4">Cs&#x00E1;rdi and Nepusz, 2006</xref>). To distinguish between modules, each module was designated with a visually friendly color.</p>
</sec>
<sec id="S2.SS6">
<title>Core Mutations for Variant Determination</title>
<p>According to our hypothesis, modules in our network are expected to be sets of synchronous FTMs that represent variants. Emerging variants develop mutations quickly, but they are characterized by a highly correlated set of characteristic mutations. These characteristic mutations form densely connected intra-modular sub-networks. These sub-networks represent the &#x201C;core&#x201D; of a module and are detected using a high-pass adjacency score threshold. The threshold value is determined empirically by mapping benchmark modules to the global phylogeny (see below) with statistical evaluation of specificity and sensitivity. The historical classification and nomenclature for these variants were extracted from the GISAID metadata.</p>
</sec>
<sec id="S2.SS7">
<title>Phylogenetic Assessment of Detected Variants</title>
<p>We assessed variants determined by our module &#x201C;core&#x201D; mutations against a global reference dataset provided by GISAID using the pipeline proposed by Nextstrain (<xref ref-type="bibr" rid="B6">Hadfield et al., 2018</xref>). First, the metadata of the global SARS-CoV-2 phylogenic tree, with 4,506,129 high quality genomes created on December 24, 2021, were retrieved from the GISAID database. A subsample randomly selected from these data was used for the skeleton construction of global SARS-CoV-2 phylogenic tree. Second, the module genomes determined by the module &#x201C;core&#x201D; mutations were extracted. Specifically, the pandemic module genomes pointing to S, V, G, GH, GV, GR, GRY, and GK were directly taken from the global reference dataset to show the consistency with the skeleton tree. Other module genomes were extracted from the source data but down-sampling was introduced if the number exceeded 200. Then, the pipeline successively performs an alignment of genomes in MAFFT (<xref ref-type="bibr" rid="B10">Katoh and Standley, 2013</xref>), phylogenetic inference in IQ-Tree (<xref ref-type="bibr" rid="B19">Minh et al., 2020</xref>), tree dating and ancestral state construction and annotation. The phylogenetic trees were visualized using the R ggtree package (<xref ref-type="bibr" rid="B35">Yu et al., 2018</xref>).</p>
</sec>
</sec>
<sec id="S3" sec-type="results">
<title>Results</title>
<sec id="S3.SS1">
<title>Variant-Specific Frequency Trajectories of Mutations Present Synchronous Temporal Changes</title>
<p>A total of 5,043,950 SARS-CoV-2 sequences during our study period were retrieved. After excluding those with probable sampling date error, 5,042,287 (&#x003E;99.9%) were eventually included. These viral sequences have been accumulating over time at an unprecedented speed, from a few to hundreds of thousands a week according to their sampling time (<xref ref-type="fig" rid="F2">Figure 2A</xref>). Changes in the prevalence of the SARS-CoV-2 variants over time have been imprinted through these sequences (<xref ref-type="fig" rid="F2">Figure 2B</xref>). Using Wuhan-Hu-1 genome (NC_045512.2) as the reference, 186,253,697 mutation events were detected at 29,825 nucleotide sites, including 28,972 (97.1%) sites with 2 or more mutation directions (<xref ref-type="supplementary-material" rid="FS2">Supplementary Figure 2</xref>). The time series plots of FTMs showed that majority of them had very low occurrence rate over time (<xref ref-type="supplementary-material" rid="FS3">Supplementary Figure 3</xref>), indicating a high chance of random or unstable mutations, or even sequencing artifacts. A few mutations with synchronous temporal changes (e.g., C241T, C3037T, C14408T and A23403G) were also observed.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption><p>Synchronous temporal changes between variant-specific FTMs and variant prevalence. <bold>(A)</bold> Weekly distribution of SARS-CoV-2 genome sequences according to sampling time. <bold>(B)</bold> Time course of major variant distribution in collected sequences. <bold>(C)</bold> A Wald&#x2019;s linkage hierarchical cluster tree of frequency trajectories of mutations. One hundred and fifty-eight mutations passing filtration were analyzed, annotated and displayed. Variant-specific mutations were flagged.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-13-859241-g002.tif"/>
</fig>
<p>To show the association of epidemic variants and genetic variations of SARS-CoV-2 across time, a clustering process using Ward&#x2019;s method was done for the FTMs. Due to ultra-high analytic dimensionality, the cluster having randomly fluctuated series was firstly identified and excluded (see section &#x201C;Materials and Methods&#x201D;). In consequence, 158 time series were left. The clustering analysis showed that mutations with consistent temporal change patterns were clustered together and some of these clusters were clearly linked to variant features (<xref ref-type="fig" rid="F2">Figure 2C</xref>). This suggests that frequency trajectories of variant-specific mutations can be used for identifying and tracking variants. Moreover, there exist other mutation trajectories within each cluster having synchronous temporal changes (<xref ref-type="fig" rid="F2">Figure 2C</xref>), which indicates the availability of more information that can be used to trace the same variant.</p>
</sec>
<sec id="S3.SS2">
<title>Identification of Variants Using the Weighted Network</title>
<p>The weighted network workflow for SARS-CoV-2 variant tracking has been summarized in <xref ref-type="fig" rid="F1">Figure 1B</xref> and detailed in section &#x201C;Materials and Methods&#x201D;. Briefly, the Pearson correlation coefficient is calculated for all pair-wise comparisons of the scaled FTMs across the viral genome. This correlation matrix is then transformed into a matrix of connection strengths using a power function (connection strength = (0.5 + 0.5 &#x00D7; correlation)<sup>&#x03B2;</sup>). Mutations with similar patterns of connection strengths are speculated to form network modules while each node represents an FTM-related mutation. Topological overlaps are used to assess the similarity of the synchronous relationship of two FTMs with all the other FTMs in the network. Modules with high topological overlaps are detected using average linkage hierarchical clustering coupled with a dynamic tree-cutting algorithm. Each module is analyzed separately to identify &#x201C;core&#x201D; mutations for variant determination.</p>
<p>We used the 158 most frequent mutations from worldwide data for module detection and variant identification to show the capability of the method to track variants using a weighted network. This may lead to some information loss about the endemic variants, but we will illustrate later that this workflow will be more sensitive when it is applied to regional data. As showed in <xref ref-type="fig" rid="F3">Figure 3A</xref>, FTMs were grouped into 20 distinct modules with 5 module clusters. Most modules (19/20, 95.0%) point to well-defined variants supported by the module genomes, which were identified from the viral population through the module core mutations (<xref ref-type="supplementary-material" rid="TS2">Supplementary Table 2</xref>). More precisely, 8 modules were clearly linked to global epidemic clades (S, V, G, GH, GV, GR, GRY, and GK) and 11 were identified as variants or sub-variants causing tens of thousands of COVID-19 cases, including two sub-variants of GRY that were not assigned Pangolin lineages. All the identification showed a very high specificity approaching 100% and a high sensitivity exceeding 70% when using the global phylogeny as a reference with an adjacency cutoff 0.7, an appropriate compromise between area under the receiver-operating characteristic curve and module-based variant discovery (<xref ref-type="supplementary-material" rid="TS3">Supplementary Table 3</xref>). Another module showed low connection strengths (&#x003C;0.4) between nodes indicating asynchronous FTMs; thus, it was ignored. In addition, the time course prevalence of the module-based variants suggested that the 5 module clusters represented the five worldwide epidemic stages until the late of November, 2021, with co-circulation of multiple major variants defined by intra-cluster modules during each period (<xref ref-type="fig" rid="F3">Figure 3B</xref>).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption><p>A benchmark to use weighted network framework for identification of worldwide pandemic variants. <bold>(A)</bold> Clustering dendrogram of 158 FTMs from GISAID worldwide data. The module numbers are labeled and module clusters are highlighted with different colors. <bold>(B)</bold> The heatmap of module-based variant prevalence. The variants were determined by core mutations within each module. The modules were reordered and colored according to their module clusters and time course. <bold>(C)</bold> Network graph with topology overlap values &#x003E; 10<sup>&#x2013; 3</sup> to show the relationship between nodes and modules of the weighted network. <bold>(D)</bold> Network graph with topology overlap values &#x003E; 0.1. <bold>(E)</bold> Phylogenic evaluation of detected worldwide pandemic variants. Time-resolved maximum clade credibility phylogeny is shown and identified variants are highlighted and annotated with visually friendly colors.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-13-859241-g003.tif"/>
</fig>
<p>Network graphs were used to further demonstrate the relationships among nodes within a module as well as to inspect how any module is related to the rest of the network and how closely any two modules are related. The continuous network topology was dichotomized by different cutoffs, and modules were individually colored. These network graphs highlighted our FTM-based weighted network conglomerated variant-specific mutations as modules with contemporaneous variants forming module clusters. First, mutations pointing to the same variant were clustered together to form closely connected modules (<xref ref-type="fig" rid="F3">Figure 3C</xref>). Second, the modules pointing to cotemporaneous variants were likely to be connected to each other (<xref ref-type="fig" rid="F3">Figure 3C</xref>). Third, with the increasing cutoff, linkages were broken in turn, first between module clusters and then between intra-modular nodes (<xref ref-type="fig" rid="F3">Figure 3D</xref> and <xref ref-type="supplementary-material" rid="FS4">Supplementary Figure 4</xref>). All of these method features provided us with fresh insights to track down the historical, current, or emerging variants.</p>
</sec>
<sec id="S3.SS3">
<title>Validation Using Phylogenic Analysis</title>
<p>Variants determined by core mutations (<xref ref-type="supplementary-material" rid="TS2">Supplementary Table 2</xref>) were evaluated using phylogenic analysis. Data randomly sampled from the global SARS-CoV-2 phylogenic tree of the GISAID repository were used to establish the phylogenic skeleton (<xref ref-type="supplementary-material" rid="FS5">Supplementary Figure 5A</xref>). Genomes with module &#x201C;core&#x201D; mutations of S, V, G, GH, GR, GRY, and GK in the skeleton showed almost perfect consistency with the expectation (<xref ref-type="supplementary-material" rid="FS5">Supplementary Figure 5B</xref>). Samples with other module &#x201C;core&#x201D; mutations were selected from the source data, an updated phylogenetic tree was generated, and nodes were colored by their modules. As shown in <xref ref-type="fig" rid="F3">Figure 3E</xref>, module &#x201C;core&#x201D; mutations detected by our weighted network successfully identified their lineages.</p>
</sec>
<sec id="S3.SS4">
<title>Workflow Application for Variant Surveillance in India and South Africa</title>
<p>After showing the capability of weighted network analysis of FTMs in module-based variant identification, we applied this workflow for SARS-CoV-2 variant surveillance in regional data and further tested its efficacy. All 59,069 SARS-CoV-2 genomes in the study period from India were first included. Since the genome numbers in the sampling weeks showed a high fluctuation (<xref ref-type="fig" rid="F4">Figure 4A</xref>), from zero to several thousand, we only kept mutations that have occurred in 10% or more of genomes with occurrences &#x003E; 10 in at least one sampling week. This resulted in 165 FTMs left for the weighted network construction. Following the automatic parameter selection and clustering process, these mutations were grouped into 33 modules among which 30 (30/33, 90.9%) had sets of mutations with strong synchronous FTMs (<xref ref-type="supplementary-material" rid="TS4">Supplementary Table 4</xref>). Five module clusters were detected in this process (<xref ref-type="supplementary-material" rid="FS6">Supplementary Figure 6</xref>). According to this module clustering feature, the heatmap of module-based variant prevalence clearly showed the SARS-CoV-2 epidemic in India by November 2021 could be divided into at least five stages, with the major variants during each stage determined by the module core mutations (<xref ref-type="fig" rid="F4">Figure 4A</xref>). Phylogenetic assessment through a module-based sampling confirmed the results of network analysis and showed the modules corresponded to B.1.617.2 (Delta), B.1.617.1 (Kappa), B.1.1.7 (Alpha), B.1.36, B.1.1.306, B.1.1.326 or their sub-variants (<xref ref-type="fig" rid="F4">Figure 4B</xref>). It is noteworthy that the weighted network would provide much earlier warning of Delta (B.1.617.2) than the date it was reported as VUI by WHO (January 3 2021 vs. April 4 2021, <xref ref-type="fig" rid="F4">Figure 4A</xref>), if the time delay between sample collection, sequencing and analysis could be sufficiently overcome. In addition, the phylogenetic tree suggested that the network analysis detected multiple descendants of the major SARS-CoV-2 variants previously or currently circulating in India. Specifically, four primary descendent variants of B.1.617.2 (<xref ref-type="fig" rid="F4">Figure 4B</xref>), which continued circulating as a dominant lineage in India until the end of November 2021, were tracked down. In contrast, CMM classified this variant to G3.14.1 with no subtype surveillance and Pangolin gave various subtypes of this variant (<xref ref-type="supplementary-material" rid="TS5">Supplementary Table 5</xref>).</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption><p>A demonstration to identify SARS-CoV-2 variants prevalent in India using the weighted network. <bold>(A)</bold> Weekly distribution of SARS-CoV-2 genome sequences according to sampling time (top) and the heatmap of module-based variant prevalence (bottom). Core mutations within each module were used to define the variants. The modules were reordered and colored according to their module clusters and time course. The weeks when Delta (B.1.617.2) was identified as a prevalent variant by network model (green) or reported as a VUI by WHO (blue) are highlighted by rectangles. <bold>(B)</bold> Phylogenic evaluation of detected endemic SARS-CoV-2 variants in South Africa. The detected variants are highlighted and annotated with visually friendly colors.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-13-859241-g004.tif"/>
</fig>
<p>The same pipeline was applied in South Africa for SARS-CoV-2 variant tracking. The weighted network modeling for FTMs generated by the total available 17,778 SARS-CoV-2 genomes showed viral population in South Africa has gone through four prevalent stages with variant cluster pattern (<xref ref-type="fig" rid="F5">Figure 5A</xref>), including a rapid surging of suspected variants with numerous spike protein mutations detected since November 7, 2021 (<xref ref-type="supplementary-material" rid="TS6">Supplementary Table 6</xref> and <xref ref-type="supplementary-material" rid="FS7">Supplementary Figure 7</xref>). The newly circulating variants seemed to split from module 25 with mutation C10029T and C22995A according to the prevalence rate. Phylogenetic analysis using module-based sampling data showed that the dominant variants at the four stages were B.1.1.529 (Omicron), B.1.617.2, B.1.351, and C.1, respectively, from near to far (<xref ref-type="fig" rid="F5">Figures 5B&#x2013;E</xref>). The descendants of these variants were also tracked down by the weighted network, having consistent but more dedicated subtypes compared with Pangolin classification and more detailed than CMM grouping (<xref ref-type="supplementary-material" rid="TS7">Supplementary Table 7</xref>).</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption><p>SARS-CoV-2 variant surveillance in South Africa using the weighted network. <bold>(A)</bold> Weekly distribution of SARS-CoV-2 genome sequences according to sampling time (top) and the heatmap of module-based variant prevalence (bottom). Core mutations within each module were used to define the variants. The modules were reordered and colored according to their module clusters and time course. The weeks when Omicron (B.1.1.529) and Beta (B.1.351) were identified as prevalent variants by network model (green) or reported as VUI or VOC by WHO (blue) are highlighted by rectangles. <bold>(B&#x2013;E)</bold> Phylogenic evaluation of every major SARS-CoV-2 variant detected in South Africa, including Omicron <bold>(B)</bold>, Delta <bold>(C)</bold>, Beta <bold>(D)</bold> and C.1 <bold>(E)</bold>. The module-based variants having consistent classification with Pangolin lineages are labeled.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-13-859241-g005.tif"/>
</fig>
</sec>
</sec>
<sec id="S4" sec-type="discussion">
<title>Discussion</title>
<p>Scientists are keeping their eyes open for the mutating SARS-CoV-2 virus and making every effort to detect, investigate, and monitor clinically important variants (<xref ref-type="bibr" rid="B2">Chakraborty et al., 2021</xref>; <xref ref-type="bibr" rid="B5">Grubaugh et al., 2021</xref>; <xref ref-type="bibr" rid="B16">Mascola et al., 2021</xref>). In this study, we proposed a module-based variant surveillance framework through weighted network modeling of FTMs, enabling us to rapidly gain insights into the time-scaled dispersal history of SARS-CoV-2 variants without requiring prior lineage assignment of each viral sequence (<xref ref-type="fig" rid="F1">Figure 1A</xref>). This framework modularizes the FTMs, with synchronous FTMs conglomerating together to represent the variants and module clusters reflecting contemporaneous variants (<xref ref-type="fig" rid="F3">Figure 3C</xref>). The module-based variants are assessed by phylogenetic tree through sub-sampling to facilitate communication and control of the epidemic.</p>
<p>The ad hoc viral classification may delay the detection of newly emerging variants or their descendants. Viral subtyping followed by their characterization, prevalence monitoring and risk assessment is continuing to be used in SARS-CoV-2 variant surveillance (<xref ref-type="bibr" rid="B31">World Health Organization [WHO], 2021</xref>). Either phylogenetic-tree-based partition of GISAID (<xref ref-type="bibr" rid="B7">Han et al., 2019</xref>), Nextstrain (<xref ref-type="bibr" rid="B6">Hadfield et al., 2018</xref>) and Pangolin (<xref ref-type="bibr" rid="B22">Rambaut et al., 2020</xref>), or genetic-feature-based grouping of CMMs (<xref ref-type="bibr" rid="B21">Qin et al., 2021</xref>) and ISMs (<xref ref-type="bibr" rid="B37">Zhao et al., 2020</xref>), captured viral subtype features according to historical data, resulting in lag signals of classification, and then false subtyping at the early stage of their emergence delayed the public health response. Our module-based variant surveillance would have provided much earlier warning about newly surging variants of B.1.617.2 in India (<xref ref-type="fig" rid="F4">Figure 4A</xref>) and B.1.351/B.1.1.529 in South Africa (<xref ref-type="fig" rid="F5">Figure 5A</xref>) prior to their announced VUI/VOC dates by WHO.</p>
<p>Our investigation also reveals other advantages of module-based variant monitoring. First, the surveillance system will automatically divide the whole epidemic period into multiple stages and detect variant co-circulation pattern during each stage (<xref ref-type="fig" rid="F3">Figures 3B</xref>, <xref ref-type="fig" rid="F4">4A</xref>, <xref ref-type="fig" rid="F5">5A</xref>). This may give an important insight into viral evolution (<xref ref-type="bibr" rid="B11">Kostaki et al., 2021</xref>). Second, the methodology provides variant surveillance at moderate resolution, facilitating an overview of epidemic variants. Our framework focuses on the tracking of prevalent variants rather than comprehensive surveillance. In spite of a rough filtration process, the benchmark analysis using worldwide data tracked down all the major pandemic variants and some regionally epidemic variants (<xref ref-type="fig" rid="F3">Figure 3E</xref>). National level analysis in India and South Africa further demonstrated that this approach not only provided a variant profile (<xref ref-type="fig" rid="F4">Figures 4B</xref>, <xref ref-type="fig" rid="F5">5B&#x2013;E</xref>) consistent with previous studies (<xref ref-type="bibr" rid="B25">Singh et al., 2021</xref>; <xref ref-type="bibr" rid="B27">Tegally et al., 2021</xref>), but also gave more detailed variant monitoring than CMM. The weighted network analysis also provided a much more enriched variant investigation than Pangolin (<xref ref-type="supplementary-material" rid="TS5">Supplementary Tables 5</xref>, <xref ref-type="supplementary-material" rid="TS7">7</xref>), which were confirmed by previous reports. Third, our framework allows insertion, deletion and recombination events to be included. This highly extends the surveillance because current variant monitoring mainly involves substitution events (<xref ref-type="bibr" rid="B37">Zhao et al., 2020</xref>; <xref ref-type="bibr" rid="B21">Qin et al., 2021</xref>) and poses a great challenge in phylogenetic inference (<xref ref-type="bibr" rid="B14">Liu et al., 2021</xref>).</p>
<p>Our approach can be an alternative method for rapid investigation and early detection of prevalent variants to facilitate regional SARS-CoV-2 genomic surveillance. An efficient variant surveillance is firstly dependent on the timely availability of viral genomes (<xref ref-type="bibr" rid="B9">Kalia et al., 2021</xref>). To compensate and minimize the time delay between sample collection and submission, surveillance activities at national and sub-national levels, where first hand data are actually acquired, are highly recommended (<xref ref-type="bibr" rid="B31">World Health Organization [WHO], 2021</xref>). Meanwhile, simple surveillance systems, especially employing time-based analysis of SARS-CoV-2 mutations, are developed to assist in the identification of candidate variants of clinical importance. Nevertheless, most of them focus on trend survey of viral mutations (<xref ref-type="bibr" rid="B28">Wada et al., 2020</xref>; <xref ref-type="bibr" rid="B23">Showers et al., 2022</xref>) or their phenetic clustering (<xref ref-type="bibr" rid="B33">Yang et al., 2020</xref>; <xref ref-type="bibr" rid="B3">Chiara et al., 2021</xref>) but not real variant monitoring. Based on similar motivation, <xref ref-type="bibr" rid="B1">Bernasconi et al. (2021)</xref> applied standard time-series clustering to group 1-month-long FTMs for detection of all SARS-CoV-2 variants at national level. Due to the segmenting and complete analysis of FTMs, they have to face the challenge of handling the discrepancies between cluster features of the same variants, especially when these variants are new and not included in the lineage dictionary. Our module-based variant monitoring overcomes these difficulties by concentrating on high-frequency FTMs for prevalent variant identification.</p>
<p>Some limitations are also acknowledged. First, the mutation modules detected by our workflow may not represent a nominated lineage, but the analysis offers perceptive insights into novel variants which could be causing more transmission. Second, the independence between FTMs were assumed in the analysis. This might not be true especially for multiple direction mutations at the same nucleotide sites. However, as we can see in our analysis, the assumption may not highly influence our results. Lastly, the threshold value of FTM filtration is empirically chosen. This may result in the loss of less frequent variants. We believe it is a trade-off between detectability and discriminability in variant monitoring. When more samples are available and the cutoff is thought to be too big, analysis at a higher spatial resolution is recommended.</p>
<p>In summary, an efficient and easy-to-use weighted network framework was proposed for SARS-CoV-2 variants tracing that could help to accelerate the understanding, surveillance, and control of the emerging viral variants.</p>
</sec>
<sec id="S5" sec-type="data-availability">
<title>Data Availability Statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="FS1">Supplementary Material</xref>, further inquiries can be directed to the corresponding author/s.</p>
</sec>
<sec id="S6">
<title>Author Contributions</title>
<p>YL and YH conceived, designed, and supervised the project. QL and FZ collected the data. QH, QZ, YW, and YL performed computations, analyzed the results, and drafted the manuscript. PB and YH provided critical revision for important intellectual content. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec id="conf1" sec-type="COI-statement">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="pudiscl1" sec-type="disclaimer">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<sec id="S7" sec-type="funding-information">
<title>Funding</title>
<p>This work was supported by the National Natural Science Foundation of China (81973150 to YH), the Guangdong Basic and Applied Basic Research Foundation (2021A1515011591 to YL), and the Guangdong Medical Science and Technology Research Foundation (A2021104 to YL).</p>
</sec>
<ack><p>We would like to thank Prof. Jinghua Li, Drs. Zhicheng Du, and Xiao Lin for the fruitful discussions and Mr. Shuming Zhu for his precious IT support.</p>
</ack>
<sec id="S9" sec-type="supplementary-material">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fmicb.2022.859241/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fmicb.2022.859241/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Image_1.TIF" id="FS1" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Figure 1</label>
<caption><p>Frequency distribution of mutational number of each SARS-CoV-2 genome at each sampling week.</p></caption>
</supplementary-material>
<supplementary-material xlink:href="Image_2.TIF" id="FS2" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Figure 2</label>
<caption><p>Frequency distribution of number of mutation directions at each mutation sites.</p></caption>
</supplementary-material>
<supplementary-material xlink:href="Image_3.TIF" id="FS3" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Figure 3</label>
<caption><p>Frequency trajectories of mutations by SARS-CoV-2 genome regions. UTR, Untranslated region; NSP, non-structural protein; S, Spike protein; ORF, Open reading frame; M, Membrane protein; N, Nucleocapsid protein; E, Envelope protein.</p></caption>
</supplementary-material>
<supplementary-material xlink:href="Image_4.TIF" id="FS4" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Figure 4</label>
<caption><p>Network graphs with different topological overlap cutoffs for identification of worldwide pandemic variants.</p></caption>
</supplementary-material>
<supplementary-material xlink:href="Image_5.TIF" id="FS5" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Figure 5</label>
<caption><p>The global SARS-CoV-2 phylogenic skeleton. <bold>(A)</bold> The SARS-CoV-2 phylogenic skeleton generated by the Nextstrain pipeline based on a random sample of the global phylogenic tree from the GISAID database, with the edges colored by the GISAID clade nomenclature system. <bold>(B)</bold> Comparison of the genome classification consistency between the expectation and those determined by the &#x201C;core&#x201D; mutations.</p></caption>
</supplementary-material>
<supplementary-material xlink:href="Image_6.TIF" id="FS6" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Figure 6</label>
<caption><p>Clustering dendrogram of 165 FTMs from India, with dissimilarity based on topological overlap. The module numbers were labeled and module clusters were highlighted with different colors.</p></caption>
</supplementary-material>
<supplementary-material xlink:href="Image_7.TIF" id="FS7" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Supplementary Figure 7</label>
<caption><p>Clustering dendrogram of 223 FTMs from South Africa, with dissimilarity based on topological overlap. The module numbers were labeled and module clusters were highlighted with different colors.</p></caption>
</supplementary-material>
<supplementary-material xlink:href="Table_1.XLSX" id="TS1" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_2.XLSX" id="TS2" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_3.XLSX" id="TS3" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_4.XLSX" id="TS4" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_5.XLSX" id="TS5" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_6.XLSX" id="TS6" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_7.XLSX" id="TS7" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bernasconi</surname> <given-names>A.</given-names></name> <name><surname>Mari</surname> <given-names>L.</given-names></name> <name><surname>Casagrandi</surname> <given-names>R.</given-names></name> <name><surname>Ceri</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>Data-driven analysis of amino acid change dynamics timely reveals SARS-CoV-2 variant emergence.</article-title> <source><italic>Sci. Rep.</italic></source> <volume>11</volume>:<issue>21068</issue>. <pub-id pub-id-type="doi">10.1038/s41598-021-00496-z</pub-id> <pub-id pub-id-type="pmid">34702903</pub-id></citation></ref>
<ref id="B2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chakraborty</surname> <given-names>D.</given-names></name> <name><surname>Agrawal</surname> <given-names>A.</given-names></name> <name><surname>Maiti</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>Rapid identification and tracking of SARS-CoV-2 variants of concern.</article-title> <source><italic>Lancet</italic></source> <volume>397</volume> <fpage>1346</fpage>&#x2013;<lpage>1347</lpage>. <pub-id pub-id-type="doi">10.1016/s0140-6736(21)00470-0</pub-id></citation></ref>
<ref id="B3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chiara</surname> <given-names>M.</given-names></name> <name><surname>Horner</surname> <given-names>D. S.</given-names></name> <name><surname>Gissi</surname> <given-names>C.</given-names></name> <name><surname>Pesole</surname> <given-names>G.</given-names></name></person-group> (<year>2021</year>). <article-title>Comparative genomics reveals early emergence and biased spatiotemporal distribution of SARS-CoV-2.</article-title> <source><italic>Mol. Biol. Evol.</italic></source> <volume>38</volume> <fpage>2547</fpage>&#x2013;<lpage>2565</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msab049</pub-id> <pub-id pub-id-type="pmid">33605421</pub-id></citation></ref>
<ref id="B4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cs&#x00E1;rdi</surname> <given-names>G.</given-names></name> <name><surname>Nepusz</surname> <given-names>T.</given-names></name></person-group> (<year>2006</year>). <article-title>The igraph software package for complex network research.</article-title> <source><italic>InterJ. Complex Syst.</italic></source> <volume>1695</volume> <fpage>1</fpage>&#x2013;<lpage>9</lpage>.</citation></ref>
<ref id="B5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Grubaugh</surname> <given-names>N. D.</given-names></name> <name><surname>Hodcroft</surname> <given-names>E. B.</given-names></name> <name><surname>Fauver</surname> <given-names>J. R.</given-names></name> <name><surname>Phelan</surname> <given-names>A. L.</given-names></name> <name><surname>Cevik</surname> <given-names>M.</given-names></name></person-group> (<year>2021</year>). <article-title>Public health actions to control new SARS-CoV-2 variants.</article-title> <source><italic>Cell</italic></source> <volume>184</volume> <fpage>1127</fpage>&#x2013;<lpage>1132</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2021.01.044</pub-id> <pub-id pub-id-type="pmid">33581746</pub-id></citation></ref>
<ref id="B6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hadfield</surname> <given-names>J.</given-names></name> <name><surname>Megill</surname> <given-names>C.</given-names></name> <name><surname>Bell</surname> <given-names>S. M.</given-names></name> <name><surname>Huddleston</surname> <given-names>J.</given-names></name> <name><surname>Potter</surname> <given-names>B.</given-names></name> <name><surname>Callender</surname> <given-names>C.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>Nextstrain: real-time tracking of pathogen evolution.</article-title> <source><italic>Bioinformatics</italic></source> <volume>34</volume> <fpage>4121</fpage>&#x2013;<lpage>4123</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty407</pub-id> <pub-id pub-id-type="pmid">29790939</pub-id></citation></ref>
<ref id="B7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Han</surname> <given-names>A. X.</given-names></name> <name><surname>Parker</surname> <given-names>E.</given-names></name> <name><surname>Scholer</surname> <given-names>F.</given-names></name> <name><surname>Maurer-Stroh</surname> <given-names>S.</given-names></name> <name><surname>Russell</surname> <given-names>C. A.</given-names></name></person-group> (<year>2019</year>). <article-title>Phylogenetic clustering by linear integer programming (PhyCLIP).</article-title> <source><italic>Mol. Biol. Evol.</italic></source> <volume>36</volume> <fpage>1580</fpage>&#x2013;<lpage>1595</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msz053</pub-id> <pub-id pub-id-type="pmid">30854550</pub-id></citation></ref>
<ref id="B8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Horvath</surname> <given-names>S.</given-names></name></person-group> (<year>2011</year>). &#x201C;<article-title>Chapter 5 Correlation and Gene Co-Expression Networks</article-title>,&#x201D; in <source><italic>Weighted Network Analysis: Applications in Genomics and Systems Biology</italic></source>, (<publisher-loc>New York</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>90</fpage>&#x2013;<lpage>121</lpage>.</citation></ref>
<ref id="B9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kalia</surname> <given-names>K.</given-names></name> <name><surname>Saberwal</surname> <given-names>G.</given-names></name> <name><surname>Sharma</surname> <given-names>G.</given-names></name></person-group> (<year>2021</year>). <article-title>The lag in SARS-CoV-2 genome submissions to GISAID.</article-title> <source><italic>Nat. Biotechnol.</italic></source> <volume>39</volume> <fpage>1058</fpage>&#x2013;<lpage>1060</lpage>. <pub-id pub-id-type="doi">10.1038/s41587-021-01040-0</pub-id> <pub-id pub-id-type="pmid">34376850</pub-id></citation></ref>
<ref id="B10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Katoh</surname> <given-names>K.</given-names></name> <name><surname>Standley</surname> <given-names>D. M.</given-names></name></person-group> (<year>2013</year>). <article-title>MAFFT multiple sequence alignment software version 7: improvements in performance and usability.</article-title> <source><italic>Mol. Biol. Evol.</italic></source> <volume>30</volume> <fpage>772</fpage>&#x2013;<lpage>780</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/mst010</pub-id> <pub-id pub-id-type="pmid">23329690</pub-id></citation></ref>
<ref id="B11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kostaki</surname> <given-names>E. G.</given-names></name> <name><surname>Tseti</surname> <given-names>I.</given-names></name> <name><surname>Tsiodras</surname> <given-names>S.</given-names></name> <name><surname>Pavlakis</surname> <given-names>G. N.</given-names></name> <name><surname>Sfikakis</surname> <given-names>P. P.</given-names></name> <name><surname>Paraskevis</surname> <given-names>D.</given-names></name></person-group> (<year>2021</year>). <article-title>Temporal dominance of B.1.1.7 over B.1.354 SARS-CoV-2 variant: a hypothesis based on areas of variant co-circulation.</article-title> <source><italic>Life</italic></source> <volume>11</volume>:<issue>375</issue>. <pub-id pub-id-type="doi">10.3390/life11050375</pub-id> <pub-id pub-id-type="pmid">33921938</pub-id></citation></ref>
<ref id="B12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Langfelder</surname> <given-names>P.</given-names></name> <name><surname>Horvath</surname> <given-names>S.</given-names></name></person-group> (<year>2008</year>). <article-title>WGCNA: an R package for weighted correlation network analysis.</article-title> <source><italic>BMC Bioinformatics</italic></source> <volume>9</volume>:<issue>559</issue>. <pub-id pub-id-type="doi">10.1186/1471-2105-9-559</pub-id> <pub-id pub-id-type="pmid">19114008</pub-id></citation></ref>
<ref id="B13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Langfelder</surname> <given-names>P.</given-names></name> <name><surname>Zhang</surname> <given-names>B.</given-names></name> <name><surname>Horvath</surname> <given-names>S.</given-names></name></person-group> (<year>2008</year>). <article-title>Defining clusters from a hierarchical cluster tree: the Dynamic Tree Cut package for R.</article-title> <source><italic>Bioinformatics</italic></source> <volume>24</volume> <fpage>719</fpage>&#x2013;<lpage>720</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btm563</pub-id> <pub-id pub-id-type="pmid">18024473</pub-id></citation></ref>
<ref id="B14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Guo</surname> <given-names>L.</given-names></name> <name><surname>Xu</surname> <given-names>T.</given-names></name> <name><surname>Lu</surname> <given-names>X.</given-names></name> <name><surname>Ma</surname> <given-names>M.</given-names></name> <name><surname>Sheng</surname> <given-names>W.</given-names></name><etal/></person-group> (<year>2021</year>). <article-title>A comprehensive evolutionary and epidemiological characterization of insertion and deletion mutations in SARS-CoV-2 genomes.</article-title> <source><italic>Virus Evol.</italic></source> <volume>7</volume>:<issue>veab104</issue>. <pub-id pub-id-type="doi">10.1093/ve/veab104</pub-id> <pub-id pub-id-type="pmid">35039785</pub-id></citation></ref>
<ref id="B15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Marcais</surname> <given-names>G.</given-names></name> <name><surname>Delcher</surname> <given-names>A. L.</given-names></name> <name><surname>Phillippy</surname> <given-names>A. M.</given-names></name> <name><surname>Coston</surname> <given-names>R.</given-names></name> <name><surname>Salzberg</surname> <given-names>S. L.</given-names></name> <name><surname>Zimin</surname> <given-names>A.</given-names></name></person-group> (<year>2018</year>). <article-title>MUMmer4: a fast and versatile genome alignment system.</article-title> <source><italic>PLoS Comput. Biol.</italic></source> <volume>14</volume>:<issue>e1005944</issue>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1005944</pub-id> <pub-id pub-id-type="pmid">29373581</pub-id></citation></ref>
<ref id="B16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mascola</surname> <given-names>J. R.</given-names></name> <name><surname>Graham</surname> <given-names>B. S.</given-names></name> <name><surname>Fauci</surname> <given-names>A. S.</given-names></name></person-group> (<year>2021</year>). <article-title>SARS-CoV-2 viral variants - tackling a moving target.</article-title> <source><italic>JAMA</italic></source> <volume>325</volume> <fpage>1261</fpage>&#x2013;<lpage>1262</lpage>. <pub-id pub-id-type="doi">10.1001/jama.2021.2088</pub-id> <pub-id pub-id-type="pmid">33571363</pub-id></citation></ref>
<ref id="B17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Massacci</surname> <given-names>A.</given-names></name> <name><surname>Sperandio</surname> <given-names>E.</given-names></name> <name><surname>D&#x2019;ambrosio</surname> <given-names>L.</given-names></name> <name><surname>Maffei</surname> <given-names>M.</given-names></name> <name><surname>Palombo</surname> <given-names>F.</given-names></name> <name><surname>Aurisicchio</surname> <given-names>L.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>Design of a companion bioinformatic tool to detect the emergence and geographical distribution of SARS-CoV-2 Spike protein genetic variants.</article-title> <source><italic>J. Transl. Med.</italic></source> <volume>18</volume>:<issue>494</issue>. <pub-id pub-id-type="doi">10.1186/s12967-020-02675-4</pub-id> <pub-id pub-id-type="pmid">33380328</pub-id></citation></ref>
<ref id="B18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mercatelli</surname> <given-names>D.</given-names></name> <name><surname>Giorgi</surname> <given-names>F. M.</given-names></name></person-group> (<year>2020</year>). <article-title>Geographic and genomic distribution of SARS-CoV-2 mutations.</article-title> <source><italic>Front. Microbiol.</italic></source> <volume>11</volume>:<issue>1800</issue>. <pub-id pub-id-type="doi">10.3389/fmicb.2020.01800</pub-id> <pub-id pub-id-type="pmid">32793182</pub-id></citation></ref>
<ref id="B19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Minh</surname> <given-names>B. Q.</given-names></name> <name><surname>Schmidt</surname> <given-names>H. A.</given-names></name> <name><surname>Chernomor</surname> <given-names>O.</given-names></name> <name><surname>Schrempf</surname> <given-names>D.</given-names></name> <name><surname>Woodhams</surname> <given-names>M. D.</given-names></name> <name><surname>Von Haeseler</surname> <given-names>A.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>IQ-TREE 2: new models and efficient methods for phylogenetic inference in the genomic era.</article-title> <source><italic>Mol. Biol. Evol.</italic></source> <volume>37</volume> <fpage>1530</fpage>&#x2013;<lpage>1534</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msaa015</pub-id> <pub-id pub-id-type="pmid">32011700</pub-id></citation></ref>
<ref id="B20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Oude Munnink</surname> <given-names>B. B.</given-names></name> <name><surname>Worp</surname> <given-names>N.</given-names></name> <name><surname>Nieuwenhuijse</surname> <given-names>D. F.</given-names></name> <name><surname>Sikkema</surname> <given-names>R. S.</given-names></name> <name><surname>Haagmans</surname> <given-names>B.</given-names></name> <name><surname>Fouchier</surname> <given-names>R. A. M.</given-names></name><etal/></person-group> (<year>2021</year>). <article-title>The next phase of SARS-CoV-2 surveillance: real-time molecular epidemiology.</article-title> <source><italic>Nat. Med.</italic></source> <volume>27</volume> <fpage>1518</fpage>&#x2013;<lpage>1524</lpage>. <pub-id pub-id-type="doi">10.1038/s41591-021-01472-w</pub-id> <pub-id pub-id-type="pmid">34504335</pub-id></citation></ref>
<ref id="B21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qin</surname> <given-names>L.</given-names></name> <name><surname>Ding</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>Q.</given-names></name> <name><surname>Meng</surname> <given-names>J.</given-names></name> <name><surname>Jiang</surname> <given-names>T.</given-names></name></person-group> (<year>2021</year>). <article-title>Co-mutation modules capture the evolution and transmission patterns of SARS-CoV-2.</article-title> <source><italic>Brief. Bioinform.</italic></source> <volume>22</volume>:<issue>bbab222</issue>. <pub-id pub-id-type="doi">10.1093/bib/bbab222</pub-id> <pub-id pub-id-type="pmid">34121111</pub-id></citation></ref>
<ref id="B22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rambaut</surname> <given-names>A.</given-names></name> <name><surname>Holmes</surname> <given-names>E. C.</given-names></name> <name><surname>O&#x2019;toole</surname> <given-names>A.</given-names></name> <name><surname>Hill</surname> <given-names>V.</given-names></name> <name><surname>Mccrone</surname> <given-names>J. T.</given-names></name> <name><surname>Ruis</surname> <given-names>C.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>A dynamic nomenclature proposal for SARS-CoV-2 lineages to assist genomic epidemiology.</article-title> <source><italic>Nat. Microbiol.</italic></source> <volume>5</volume> <fpage>1403</fpage>&#x2013;<lpage>1407</lpage>. <pub-id pub-id-type="doi">10.1038/s41564-020-0770-5</pub-id> <pub-id pub-id-type="pmid">32669681</pub-id></citation></ref>
<ref id="B23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Showers</surname> <given-names>W. M.</given-names></name> <name><surname>Leach</surname> <given-names>S. M.</given-names></name> <name><surname>Kechris</surname> <given-names>K.</given-names></name> <name><surname>Strong</surname> <given-names>M.</given-names></name></person-group> (<year>2022</year>). <article-title>Longitudinal analysis of SARS-CoV-2 spike and RNA-dependent RNA polymerase protein sequences reveals the emergence and geographic distribution of diverse mutations.</article-title> <source><italic>Infect. Genet. Evol.</italic></source> <volume>97</volume>:<issue>105153</issue>. <pub-id pub-id-type="doi">10.1016/j.meegid.2021.105153</pub-id> <pub-id pub-id-type="pmid">34801754</pub-id></citation></ref>
<ref id="B24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shu</surname> <given-names>Y.</given-names></name> <name><surname>McCauley</surname> <given-names>J.</given-names></name></person-group> (<year>2017</year>). <article-title>GISAID: global initiative on sharing all influenza data - from vision to reality.</article-title> <source><italic>Eurosurveillance</italic></source> <volume>22</volume> <fpage>2</fpage>&#x2013;<lpage>4</lpage>. <pub-id pub-id-type="doi">10.2807/1560-7917.es.2017.22.13.30494</pub-id> <pub-id pub-id-type="pmid">28382917</pub-id></citation></ref>
<ref id="B25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Singh</surname> <given-names>J.</given-names></name> <name><surname>Rahman</surname> <given-names>S. A.</given-names></name> <name><surname>Ehtesham</surname> <given-names>N. Z.</given-names></name> <name><surname>Hira</surname> <given-names>S.</given-names></name> <name><surname>Hasnain</surname> <given-names>S. E.</given-names></name></person-group> (<year>2021</year>). <article-title>SARS-CoV-2 variants of concern are emerging in India.</article-title> <source><italic>Nat. Med.</italic></source> <volume>27</volume> <fpage>1131</fpage>&#x2013;<lpage>1133</lpage>. <pub-id pub-id-type="doi">10.1038/s41591-021-01397-4</pub-id> <pub-id pub-id-type="pmid">34045737</pub-id></citation></ref>
<ref id="B26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tang</surname> <given-names>X.</given-names></name> <name><surname>Ying</surname> <given-names>R.</given-names></name> <name><surname>Yao</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>G.</given-names></name> <name><surname>Wu</surname> <given-names>C.</given-names></name> <name><surname>Tang</surname> <given-names>Y.</given-names></name><etal/></person-group> (<year>2021</year>). <article-title>Evolutionary analysis and lineage designation of SARS-CoV-2 genomes.</article-title> <source><italic>Sci. Bull.</italic></source> <volume>66</volume> <fpage>2297</fpage>&#x2013;<lpage>2311</lpage>. <pub-id pub-id-type="doi">10.1016/j.scib.2021.02.012</pub-id> <pub-id pub-id-type="pmid">33585048</pub-id></citation></ref>
<ref id="B27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tegally</surname> <given-names>H.</given-names></name> <name><surname>Wilkinson</surname> <given-names>E.</given-names></name> <name><surname>Giovanetti</surname> <given-names>M.</given-names></name> <name><surname>Iranzadeh</surname> <given-names>A.</given-names></name> <name><surname>Fonseca</surname> <given-names>V.</given-names></name> <name><surname>Giandhari</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2021</year>). <article-title>Detection of a SARS-CoV-2 variant of concern in South Africa.</article-title> <source><italic>Nature</italic></source> <volume>592</volume> <fpage>438</fpage>&#x2013;<lpage>443</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-021-03402-9</pub-id> <pub-id pub-id-type="pmid">33690265</pub-id></citation></ref>
<ref id="B28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wada</surname> <given-names>K.</given-names></name> <name><surname>Wada</surname> <given-names>Y.</given-names></name> <name><surname>Ikemura</surname> <given-names>T.</given-names></name></person-group> (<year>2020</year>). <article-title>Time-series analyses of directional sequence changes in SARS-CoV-2 genomes and an efficient search method for candidates for advantageous mutations for growth in human cells.</article-title> <source><italic>Gene X</italic></source> <volume>5</volume>:<issue>100038</issue>. <pub-id pub-id-type="doi">10.1016/j.gene.2020.100038</pub-id></citation></ref>
<ref id="B29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ward</surname> <given-names>J. H.</given-names></name></person-group> (<year>1963</year>). <article-title>Hierarchical grouping to optimize an objective function.</article-title> <source><italic>J. Am. Stat. Assoc.</italic></source> <volume>58</volume> <fpage>236</fpage>&#x2013;<lpage>244</lpage>. <pub-id pub-id-type="doi">10.1080/01621459.1963.10500845</pub-id></citation></ref>
<ref id="B30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wong</surname> <given-names>G. K.</given-names></name> <name><surname>Yang</surname> <given-names>Z.</given-names></name> <name><surname>Passey</surname> <given-names>D. A.</given-names></name> <name><surname>Kibukawa</surname> <given-names>M.</given-names></name> <name><surname>Paddock</surname> <given-names>M.</given-names></name> <name><surname>Liu</surname> <given-names>C. R.</given-names></name><etal/></person-group> (<year>2003</year>). <article-title>A population threshold for functional polymorphisms.</article-title> <source><italic>Genome Res.</italic></source> <volume>13</volume> <fpage>1873</fpage>&#x2013;<lpage>1879</lpage>. <pub-id pub-id-type="doi">10.1101/gr.1324303</pub-id> <pub-id pub-id-type="pmid">12902381</pub-id></citation></ref>
<ref id="B31"><citation citation-type="journal"><collab>World Health Organization [WHO]</collab> (<year>2021</year>). <article-title>Guidance for Surveillance of SARS-CoV-2 Variants: Interim Guidance, 9 August 2021.</article-title> Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.who.int/publications/i/item/WHO_2019-nCoV_surveillance_variants">https://www.who.int/publications/i/item/WHO_2019-nCoV_surveillance_variants</ext-link> <comment>(accessed January 12, 2022)</comment>.</citation></ref>
<ref id="B32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>F.</given-names></name> <name><surname>Zhao</surname> <given-names>S.</given-names></name> <name><surname>Yu</surname> <given-names>B.</given-names></name> <name><surname>Chen</surname> <given-names>Y. M.</given-names></name> <name><surname>Wang</surname> <given-names>W.</given-names></name> <name><surname>Song</surname> <given-names>Z. G.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>A new coronavirus associated with human respiratory disease in China.</article-title> <source><italic>Nature</italic></source> <volume>579</volume> <fpage>265</fpage>&#x2013;<lpage>269</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-020-2008-3</pub-id></citation></ref>
<ref id="B33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>H. C.</given-names></name> <name><surname>Chen</surname> <given-names>C. H.</given-names></name> <name><surname>Wang</surname> <given-names>J. H.</given-names></name> <name><surname>Liao</surname> <given-names>H. C.</given-names></name> <name><surname>Yang</surname> <given-names>C. T.</given-names></name> <name><surname>Chen</surname> <given-names>C. W.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>Analysis of genomic distributions of SARS-CoV-2 reveals a dominant strain type with strong allelic associations.</article-title> <source><italic>Proc. Natl. Acad. Sci. U. S. A.</italic></source> <volume>117</volume> <fpage>30679</fpage>&#x2013;<lpage>30686</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.2007840117</pub-id> <pub-id pub-id-type="pmid">33184173</pub-id></citation></ref>
<ref id="B34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yip</surname> <given-names>A. M.</given-names></name> <name><surname>Horvath</surname> <given-names>S.</given-names></name></person-group> (<year>2007</year>). <article-title>Gene network interconnectedness and the generalized topological overlap measure.</article-title> <source><italic>BMC Bioinformatics</italic></source> <volume>8</volume>:<issue>22</issue>. <pub-id pub-id-type="doi">10.1186/1471-2105-8-22</pub-id> <pub-id pub-id-type="pmid">17250769</pub-id></citation></ref>
<ref id="B35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>G.</given-names></name> <name><surname>Lam</surname> <given-names>T. T.</given-names></name> <name><surname>Zhu</surname> <given-names>H.</given-names></name> <name><surname>Guan</surname> <given-names>Y.</given-names></name></person-group> (<year>2018</year>). <article-title>Two methods for mapping and visualizing associated data on phylogeny using ggtree.</article-title> <source><italic>Mol. Biol. Evol.</italic></source> <volume>35</volume> <fpage>3041</fpage>&#x2013;<lpage>3043</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msy194</pub-id> <pub-id pub-id-type="pmid">30351396</pub-id></citation></ref>
<ref id="B36"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>B.</given-names></name> <name><surname>Horvath</surname> <given-names>S.</given-names></name></person-group> (<year>2005</year>). <article-title>A general framework for weighted gene co-expression network analysis.</article-title> <source><italic>Stat. Appl. Genet. Mol. Biol.</italic></source> <volume>4</volume>:<issue>17</issue>. <pub-id pub-id-type="doi">10.2202/1544-6115.1128</pub-id> <pub-id pub-id-type="pmid">16646834</pub-id></citation></ref>
<ref id="B37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>Z.</given-names></name> <name><surname>Sokhansanj</surname> <given-names>B. A.</given-names></name> <name><surname>Malhotra</surname> <given-names>C.</given-names></name> <name><surname>Zheng</surname> <given-names>K.</given-names></name> <name><surname>Rosen</surname> <given-names>G. L.</given-names></name></person-group> (<year>2020</year>). <article-title>Genetic grouping of SARS-CoV-2 coronavirus sequences using informative subtype markers for pandemic spread visualization.</article-title> <source><italic>PLoS Comput. Biol.</italic></source> <volume>16</volume>:<issue>e1008269</issue>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1008269</pub-id> <pub-id pub-id-type="pmid">32941419</pub-id></citation></ref>
</ref-list>
</back>
</article>
