<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Bioinform.</journal-id>
<journal-title>Frontiers in Bioinformatics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Bioinform.</abbrev-journal-title>
<issn pub-type="epub">2673-7647</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1400003</article-id>
<article-id pub-id-type="doi">10.3389/fbinf.2024.1400003</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Bioinformatics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>AUTO-TUNE: selecting the distance threshold for inferring HIV transmission clusters</article-title>
<alt-title alt-title-type="left-running-head">Weaver et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fbinf.2024.1400003">10.3389/fbinf.2024.1400003</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Weaver</surname>
<given-names>Steven</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2306792/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>D&#xe1;vila Conn</surname>
<given-names>Vanessa M.</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2771526/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ji</surname>
<given-names>Daniel</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Verdonk</surname>
<given-names>Hannah</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>&#xc1;vila-R&#xed;os</surname>
<given-names>Santiago</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Leigh Brown</surname>
<given-names>Andrew J.</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wertheim</surname>
<given-names>Joel O.</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/565214/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Kosakovsky Pond</surname>
<given-names>Sergei L.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2722399/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Center for Viral Evolution</institution>, <institution>Temple University</institution>, <addr-line>Philadelphia</addr-line>, <addr-line>PA</addr-line>, <country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Center for Research in Infectious Diseases</institution>, <institution>National Institute of Respiratory Diseases</institution>, <addr-line>Mexico City</addr-line>, <country>Mexico</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Medicine</institution>, <institution>University of California San Diego</institution>, <addr-line>La Jolla</addr-line>, <addr-line>CA</addr-line>, <country>United States</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>National Institute of Respiratory Diseases</institution>, <addr-line>Mexico City</addr-line>, <country>Mexico</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/33596/overview">Helen Piontkivska</ext-link>, Kent State University, United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1909529/overview">Robin Paul</ext-link>, St. Jude Children&#x2019;s Research Hospital, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/544806/overview">Jean Lutamyo Mbisa</ext-link>, UK Health Security Agency (UKHSA), United Kingdom</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Sergei L. Kosakovsky Pond, <email>spond@temple.edu</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>10</day>
<month>07</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>4</volume>
<elocation-id>1400003</elocation-id>
<history>
<date date-type="received">
<day>12</day>
<month>03</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>17</day>
<month>05</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Weaver, D&#xe1;vila Conn, Ji, Verdonk, &#xc1;vila-R&#xed;os, Leigh Brown, Wertheim and Kosakovsky Pond.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Weaver, D&#xe1;vila Conn, Ji, Verdonk, &#xc1;vila-R&#xed;os, Leigh Brown, Wertheim and Kosakovsky Pond</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Molecular surveillance of viral pathogens and inference of transmission networks from genomic data play an increasingly important role in public health efforts, especially for HIV-1. For many methods, the genetic distance threshold used to connect sequences in the transmission network is a key parameter informing the properties of inferred networks. Using a distance threshold that is too high can result in a network with many spurious links, making it difficult to interpret. Conversely, a distance threshold that is too low can result in a network with too few links, which may not capture key insights into clusters of public health concern. Published research using the HIV-TRACE software package frequently uses the default threshold of 0.015 substitutions/site for HIV pol gene sequences, but in many cases, investigators heuristically select other threshold parameters to better capture the underlying dynamics of the epidemic they are studying. Here, we present a general heuristic scoring approach for tuning a distance threshold adaptively, which seeks to prevent the formation of giant clusters. We prioritize the ratio of the sizes of the largest and the second largest cluster, maximizing the number of clusters present in the network. We apply our scoring heuristic to outbreaks with different characteristics, such as regional or temporal variability, and demonstrate the utility of using the scoring mechanism&#x2019;s suggested distance threshold to identify clusters exhibiting risk factors that would have otherwise been more difficult to identify. For example, while we found that a 0.015 substitutions/site distance threshold is typical for US-like epidemics, recent outbreaks like the CRF07_BC subtype among men who have sex with men (MSM) in China have been found to have a lower optimal threshold of 0.005 to better capture the transition from injected drug use (IDU) to MSM as the primary risk factor. Alternatively, in communities surrounding Lake Victoria in Uganda, where there has been sustained heterosexual transmission for many years, we found that a larger distance threshold is necessary to capture a more risk factor-diverse population with sparse sampling over a longer period of time. Such identification may allow for more informed intervention action by respective public health officials.</p>
</abstract>
<kwd-group>
<kwd>molecular epidemiology</kwd>
<kwd>HIV</kwd>
<kwd>network</kwd>
<kwd>transmission cluster</kwd>
<kwd>surveillance</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Evolutionary Bioinformatics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>The use of genomic data to infer and characterize transmission networks of various pathogens has grown in prominence in the past 2 decades, with applications to a growing list of pathogens, including viruses such as HIV (<xref ref-type="bibr" rid="B52">Paraskevis et al., 2016</xref>), hepatitis C virus (HCV) (<xref ref-type="bibr" rid="B44">Murphy et al., 2019a</xref>), or influenza A virus (IAV) (<xref ref-type="bibr" rid="B33">Jombart et al., 2011</xref>), and bacteria such as <italic>M</italic>. <italic>tuberculosis</italic> (<xref ref-type="bibr" rid="B42">Mai et al., 2018</xref>) or <italic>A</italic>. <italic>baumanii</italic> (<xref ref-type="bibr" rid="B72">Thoma et al., 2022</xref>). Notably, genomic surveillance had a prominent role during the COVID-19 pandemic, including the use of sequencing for the study of transmission clusters (<xref ref-type="bibr" rid="B76">von Rotz et al., 2023</xref>; <xref ref-type="bibr" rid="B8">Campigotto et al., 2023</xref>).</p>
<p>Many competing approaches for inferring transmission clusters, transmission parameters, and source attribution have been described (e.g., for an HIV-1 centric review, see <xref ref-type="bibr" rid="B27">Grabowski et al. (2018)</xref>). These approaches can be roughly categorized into distance-based (infer clusters from pairwise genetic distances, e.g., (<xref ref-type="bibr" rid="B36">Kosakovsky Pond et al., 2018</xref>), phylogeny-based (infer phylogenetic trees from the data, then process the resulting tree, e.g., <xref ref-type="bibr" rid="B60">Ragonnet-Cronin et al. (2013)</xref>, or phylodynamic (transmission model is directly incorporated into tree inference, e.g., <xref ref-type="bibr" rid="B75">Volz et al. (2017)</xref>). These methodological categories differ considerably in model and computational complexity, as well as in interpretability of results. Comparisons of different methods have been undertaken, showing broad compatibility of results, but also highlighting application-specific differences between them (<xref ref-type="bibr" rid="B48">Novitsky et al., 2020</xref>). Our goal here is not to develop a conceptually new method, but rather to propose a systematic approach to selecting the key parameter (distance threshold) for the popular HIV-TRACE (<xref ref-type="bibr" rid="B36">Kosakovsky Pond et al., 2018</xref>) class distance-based methods for identifying transmission clusters.</p>
<p>Choosing an appropriate genetic distance threshold is an important part of using a molecular transmission network to track the spread of rapidly evolving pathogens (<xref ref-type="bibr" rid="B41">Liu et al., 2020</xref>; <xref ref-type="bibr" rid="B65">Rose et al., 2020</xref>). This distance threshold defines the degree of genetic closeness between pathogen sequences, isolated from two individuals, required to suggest them as potential transmission partners in the network. Using a distance threshold that is too large can result in a network with many spurious, making it difficult to interpret and analyze. On the other hand, using a distance threshold that is too small can result in a network with too few links, underestimating connections between individuals and making it difficult to accurately track the spread of the disease (<xref ref-type="bibr" rid="B26">Gore et al., 2022</xref>).</p>
<p>To enhance the utility of inferred transmission networks, it is important to carefully consider the appropriate distance threshold, <italic>d</italic>. This threshold may vary depending on the specific disease and the context in which it is spreading. For example, a highly contagious acute respiratory illness (e.g., SARS-CoV-2) may require a smaller <italic>d</italic> than a less contagious chronic illness that is primarily spread through direct contact (e.g., HIV-1). Viruses are more amenable to molecular studies compared to bacteria due to their high genetic divergence and compact genomes. Given the relatively high evolutionary rate of RNA viruses detectable genetic fingerprints can be prioritized for epidemiological studies over short time periods (<xref ref-type="bibr" rid="B52">Paraskevis et al., 2016</xref>).</p>
<p>For chronic infections such as HIV, the most appropriate genetic distance threshold should be determined according to the characteristics of the epidemic such as the speed of transmission, and the evolutionary rate of the genomic region analyzed (<xref ref-type="bibr" rid="B41">Liu et al., 2020</xref>). Sampling density and possible delays between infection and diagnosis should be considered, since samples close to the time of seroconversion are more likely to cluster than samples from well after infection. Lower thresholds will capture the most closely related sequences, while higher thresholds will capture long-term epidemics and chronically infected individuals (<xref ref-type="bibr" rid="B34">Junqueira et al., 2019</xref>).</p>
<p>Cluster analysis, i.e., identification and analysis of connected network components, in public health has been used for early identification of increased transmission (<xref ref-type="bibr" rid="B50">Oster et al., 2021</xref>; <xref ref-type="bibr" rid="B49">2018</xref>), monitoring response to an HIV outbreak (<xref ref-type="bibr" rid="B68">Sizemore et al., 2020</xref>; <xref ref-type="bibr" rid="B73">Tookes et al., 2020</xref>; <xref ref-type="bibr" rid="B74">Tumpney et al., 2020</xref>), evaluating the effectiveness of interventions (<xref ref-type="bibr" rid="B78">Wang et al., 2015</xref>; <xref ref-type="bibr" rid="B56">Peters et al., 2016</xref>; <xref ref-type="bibr" rid="B41">Liu et al., 2020</xref>) or predicting clusters that are most likely to grow in the near future (<xref ref-type="bibr" rid="B17">Erly et al., 2021</xref>; <xref ref-type="bibr" rid="B59">Ragonnet-Cronin et al., 2022</xref>). This balance can be achieved through careful analysis and consideration of the specific disease and context.</p>
<p>This study introduces AUTO-TUNE, a method that offers a systematic approach to select genetic distance thresholds for molecular HIV transmission network analysis, based purely on the structure of the collected data. By autonomously optimizing clustering metrics derived from pairwise genetic distances, AUTO-TUNE has the potential to improve the accuracy and reliability of network inference, irrespective of data attributes. The AUTO-TUNE methodology&#x2019;s independence from supplementary data makes it less sensitive to variations in data collection protocols and enhances its adaptability to various contexts, including potentially other viral diseases.</p>
</sec>
<sec sec-type="methods" id="s2">
<title>2 Methods</title>
<p>Assume that there are <italic>S</italic> aligned genomic sequences (full or partial, e.g., the HIV-1 <italic>pol</italic> gene) for a pathogen of interest, each representing the &#x201c;consensus&#x201d; circulating viral diversity at the time of sampling in a single infected individual. We shall infer a putative transmission network comprising <italic>S</italic> nodes, and <italic>E</italic> links (edges), where an edge is drawn between a pair of sequences if the genetic distance between them is at or below a threshold <italic>d</italic>. In such a network, there will be 0 &#x2264; <italic>C</italic> &#x3c; <italic>S</italic> connected components with more than one node (clusters), which are the primary object of inference. This network inference strategy is used by HIV-TRACE (<xref ref-type="bibr" rid="B36">Kosakovsky Pond et al., 2018</xref>), where the genetic distance is computed using the Tamura-Nei (TN93) (<xref ref-type="bibr" rid="B70">Tamura and Nei, 1993</xref>) model, with a variety of options controlling how to deal with ambiguous nucleotide bases; for HIV-1 such bases are informative since they often represent variants co-circulating in the infected individual at the time of sampling at substantial frequencies (<xref ref-type="bibr" rid="B35">Kosakovsky Pond et al., 2009</xref>).</p>
<p>We begin by describing an approach to assign a score to each of the choices of <italic>d</italic> in a plausible/informative range of distances. Note that while such a range is continuous, it is sufficient to only consider distance cutoffs that are in the array of pairwise distances between the sequences, as those are the cut-points where one or more additional edges will be added to the network as <italic>d</italic> is increased.</p>
<sec id="s2-1">
<title>2.1 Scoring heuristic procedure</title>
<p>The network threshold selection procedure proceeds as follows (we provide an example in the Results section as well).<list list-type="simple">
<list-item>
<p>1. For each candidate threshold <italic>d</italic>
<sub>
<italic>L</italic>
</sub>, in increasing order, ranging from the smallest genetic distance in the dataset, up to either the largest distance or a predetermined maximal threshold, we compute two network statistics: <italic>R</italic>
<sub>12</sub>, the ratio of the size of the largest cluster to the size of the second largest cluster, and <italic>C</italic>, the number of clusters in the network at this threshold. A cluster is defined as a connected component in the network with at least two nodes.</p>
</list-item>
<list-item>
<p>2. A priority score is assigned to each <italic>d</italic>
<sub>
<italic>L</italic>
</sub>. This score measures two properties of the threshold: Does <italic>R</italic>
<sub>12</sub> jump at <italic>d</italic>
<sub>
<italic>L</italic>
</sub>? How far is the number of clusters <italic>C</italic> at <italic>d</italic>
<sub>
<italic>L</italic>
</sub> from the maximal number of clusters computed over all threshold values? Let there be <italic>N</italic> overall <italic>d</italic>
<sub>
<italic>L</italic>
</sub> candidate values, and assume we are examining the <italic>i</italic>th candidate, <inline-formula id="inf1">
<mml:math id="m1">
<mml:msubsup>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> with <italic>W</italic> &#x3c; <italic>i</italic> &#x2264; <italic>N</italic> &#x2212; <italic>W</italic> (<italic>W</italic> is a positive integer defined below).</p>
<list list-type="simple">
<list-item>
<p>a. The <italic>R</italic>
<sub>12</sub> jump is computed by looking at the normalized ratio of the mean <italic>R</italic>
<sub>12</sub> values computed over the leading window <inline-formula id="inf2">
<mml:math id="m2">
<mml:msubsup>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2026;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> and the trailing window <inline-formula id="inf3">
<mml:math id="m3">
<mml:msubsup>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2026;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula>. The width of the window, <italic>W</italic>, is defined as <inline-formula id="inf4">
<mml:math id="m4">
<mml:mi>min</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>max</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>100</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mn>30</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula>. The distribution of ratios is converted to <italic>Z</italic> scores, and normalized relative to the largest positive <italic>Z</italic> score across all candidate distances, yielding the jump component of the score.</p>
</list-item>
<list-item>
<p>b. The number of clusters, <italic>C</italic>
<sub>
<italic>i</italic>
</sub> at threshold <inline-formula id="inf5">
<mml:math id="m5">
<mml:msubsup>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> is first normalized to [0,1] through <inline-formula id="inf6">
<mml:math id="m6">
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">min</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:math>
</inline-formula> and next gated via a Gompertz function transform <inline-formula id="inf7">
<mml:math id="m7">
<mml:msup>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>25</mml:mn>
<mml:mi>x</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>. This function provides an <italic>ad hoc</italic> means for penalizing having too few clusters relative to the maximum over all ranges. For example, a threshold that yields 95% of the maximal number of clusters receives a score of 0.996, a threshold that yields 85% - a score of 0.376, and a threshold that yields 60% - a score of 0.0009.</p>
</list-item>
<list-item>
<p>c. The priority score for <inline-formula id="inf8">
<mml:math id="m8">
<mml:msubsup>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> is the sum of the two components defined in a) and b), and ranges from 0 to 2.</p>
</list-item>
</list>
</list-item>
<list-item>
<p>3. The threshold with the highest priority score will be selected as the suggested automatic distance threshold, if the score is high enough (1.9 or more), and either of the two conditions hold.</p>
<list list-type="simple">
<list-item>
<p>a. No other thresholds have priority scores of 1.9 or higher</p>
</list-item>
<list-item>
<p>b. If other thresholds have priority scores of 1.9 or higher, then the range of thresholds represented by these options is small (no more than log&#x2009; <italic>N</italic> times the mean step between successive <inline-formula id="inf9">
<mml:math id="m9">
<mml:msubsup>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula>).</p>
</list-item>
</list>
</list-item>
<list-item>
<p>4. If no single threshold can be selected in step 3, then the one with the highest priority score is suggested, and an inspection of a plot of scores is recommended to ensure that the threshold is sensible.</p>
</list-item>
</list>
</p>
<p>The corresponding flowchart can be found in <xref ref-type="fig" rid="F1">Figure 1</xref>. The <italic>R</italic>
<sub>12</sub> jump component of the score is motivated by the giant component formation result from network theory: when the degree distribution satisfies particular conditions, most nodes in the network will belong to a single component, or cluster (<xref ref-type="bibr" rid="B43">Molloy and Reed, 1995</xref>). This situation leads to emidemiologically uninformative networks, and should be avoided. The default configuration considers all clusters (i.e., with two or more nodes), but users can specify that only clusters with <italic>K</italic> or more members (<italic>K</italic> &#x2265; 2) should be included in <italic>R</italic>
<sub>12</sub> and <italic>C</italic>
<sub>
<italic>i</italic>
</sub> calculations.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Method flowchart for computing and recommending a distance threshold. See text for details on normalization and specific transforms used.</p>
</caption>
<graphic xlink:href="fbinf-04-1400003-g001.tif"/>
</fig>
</sec>
<sec id="s2-2">
<title>2.2 Assortativity</title>
<p>Degree-weighted homophily (DWH) is a measure of similarity between nodes in a network based on their attributes (such as demographic characteristics or behaviors) and their degree centrality (i.e., the number of connections they have to other nodes in the network). It is used to quantify the extent to which nodes with similar attributes tend to be connected to each other more frequently than would be expected by chance (<xref ref-type="bibr" rid="B23">Golub and Jackson, 2012</xref>). DWH is calculated as the ratio of the observed number of connections between nodes with similar attributes to the expected number of connections between such nodes, based on their network degree.</p>
<p>For any two subsets <italic>A</italic> and <italic>B</italic> of nodes in a network without singletons (each node has a positive degree), define the weight between <italic>A</italic> and <italic>B</italic> as<disp-formula id="equ1">
<mml:math id="m10">
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>A</mml:mi>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:mi>B</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>A</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>B</mml:mi>
<mml:mo>,</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mtext>are&#x2009;connected</mml:mtext>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:math>
</disp-formula>where <italic>d</italic>
<sub>
<italic>i</italic>
</sub> is the degree of node <italic>i</italic>, and &#x7c;<italic>X</italic>&#x7c; is the cardinality (size) of subset <italic>X</italic>.</p>
<p>Then for any proper (not empty and not the complete network) subset of the network, <italic>G</italic>, e.g., a group of nodes sharing an attribute, e.g., transmission risk factor, define<disp-formula id="e1">
<mml:math id="m11">
<mml:mi>D</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>H</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>G</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mrow>
</mml:msub>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:math>
<label>(1)</label>
</disp-formula>with<list list-type="simple">
<list-item>
<p>&#x2022; <inline-formula id="inf10">
<mml:math id="m12">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula>: the complement of <italic>G</italic> (all nodes not in <italic>G</italic>)</p>
</list-item>
<list-item>
<p>&#x2022; <italic>d</italic>
<sub>
<italic>i</italic>
</sub>: the degree of node <italic>i</italic>
</p>
</list-item>
</list>
</p>
<p>DWH ranges from &#x2212;1 to 1. A DWH value of 0 indicates that there is no more homophily than expected by chance (conditioned on network structure), while a value of 1 indicates that there is perfect homophily (<italic>G</italic> consists of connected components disconnected from the rest of the network). A value of &#x2212;1 is achieved for perfectly disassortative networks (the only links are between <italic>G</italic> and <inline-formula id="inf11">
<mml:math id="m13">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula>).</p>
<p>Homophily metrics have been used in social network analysis and in the study of how different attributes are related to the formation of connections between individuals (<xref ref-type="bibr" rid="B58">Ragonnet-Cronin et al., 2021</xref>). To assess whether or not DWH is significantly different from 0 (and from random expectation), we generate the null distribution of DWH obtained by randomly reshuffling node attributes used to define group <italic>G</italic> and recomputing DWH for each such replicate.</p>
</sec>
<sec id="s2-3">
<title>2.3 Implementation</title>
<p>The software implementation involves a step-by-step process that utilizes the HIV-TRACE suite of packages. It starts with calculating pairwise distances with the <monospace>tn93</monospace> tool and a supplied multiple sequence alignment. Thus generated pairwise distances are supplied to the <monospace>hivnetworkcsv</monospace> script while providing the <monospace>-A</monospace> keyword argument. A brief outline of the software&#x2019;s implementation is as follows.<list list-type="simple">
<list-item>
<p>1. Calculate pairwise distances: The user first calculates the pairwise distances using the <monospace>tn93</monospace> fast pairwise distance calculator, providing the maximum threshold value to consider (0.03 in this case, which may be revised upwards for sufficiently divergent sequences, as this provides an upper bound of thresholds to consider) and the input FASTA file. The command for this step is</p>
<list list-type="simple">
<list-item>
<p>
<inline-graphic xlink:href="fbinf-04-1400003-fx1.tif"/>
</p>
</list-item>
</list>
</list-item>
</list>
</p>
<p>Please note that the threshold should include the maximal range one is intending to test.<list list-type="simple">
<list-item>
<label>2.</label>
<p>Compute priority scores for each candidate threshold: The <monospace>hivnetworkcsv</monospace> script is then executed with the required input file, format, and autotune option to generate a tab-separated output file, as shown below</p>
<list list-type="simple">
<list-item>
<p>
<inline-graphic xlink:href="fbinf-04-1400003-fx2.tif"/>
</p>
</list-item>
</list>
</list-item>
<list-item>
<label>3.</label>
<p>Visualize the report: Users can upload the generated autotune_report.tsv file to</p>
</list-item>
<list-item>
<p>&#x2009;&#x2009;&#x2009;&#x2009;&#x2009;<ext-link ext-link-type="uri" xlink:href="http://autotune.datamonkey.org/analyze">http://autotune.datamonkey.org/analyze</ext-link> for visualization and further analysis of the data. This web-based site extends the Datamonkey platform (<xref ref-type="bibr" rid="B79">Weaver et al., 2018</xref>) to provide an interactive environment to explore scores and other metrics across the range of tested outputs.</p>
</list-item>
<list-item>
<label>4.</label>
<p>Run HIV-TRACE: Once AUTO-TUNEd threshold(s) are settled upon after review, the user runs the HIV-TRACE command with the appropriate input FASTA file, distance threshold, and other required arguments. The output is saved as a JSON file. An example command is</p>
<list list-type="simple">
<list-item>
<p>
<inline-graphic xlink:href="fbinf-04-1400003-fx3.tif"/>
</p>
</list-item>
</list>
</list-item>
</list>
</p>
<sec id="s2-3-1">
<title>2.3.1 Optional: compute assortativity metrics</title>
<p>
<list list-type="simple">
<list-item>
<p>1. Annotate results: The <monospace>hivnetworkannotate</monospace> script is used to annotate the results obtained from the HIV-TRACE step with attributes. The script takes the JSON results file, node attributes file, schema file, and a resolve flag as input.</p>
<list list-type="simple">
<list-item>
<p>
<inline-graphic xlink:href="fbinf-04-1400003-fx4.tif"/>
</p>
</list-item>
</list>
</list-item>
</list>
</p>
<p>For more information, users can refer to the <monospace>hivnetworkannotate</monospace> documentation.<list list-type="simple">
<list-item>
<p>2. Analyze the results with DWH: After the results file has been annotated, the user can proceed to the assortativity page, <ext-link ext-link-type="uri" xlink:href="http://autotune.datamonkey.org/assortativity">http://autotune.datamonkey.org/assortativity</ext-link>, for further analysis of the output.</p>
</list-item>
</list>
</p>
<p>The described workflow offers a systematic approach to analyze potential distance thresholds for one&#x2019;s data with AUTO-TUNE, from calculating pairwise distances to visualizing and annotating results.</p>
</sec>
</sec>
<sec id="s2-4">
<title>2.4 Visualization</title>
<p>Visualizations of AUTO-TUNE results are accessible at <ext-link ext-link-type="uri" xlink:href="http://autotune.datamonkey.org/analyze">http://autotune.datamonkey.org/analyze</ext-link>. These include the priority score plot, and the two contributing statistics: cluster count relative to the maximum and the ratio of two largest cluster sizes (<xref ref-type="fig" rid="F2">Figure 2</xref>). An assortativity tool is available at <ext-link ext-link-type="uri" xlink:href="http://autotune.datamonkey.org/assortativity">http://autotune.datamonkey.org/assortativity</ext-link>, and is an analytical tool engineered to facilitate the calculation of Degree-weighted homophily (DWH) values. It utilizes the DWH NPM package to generate a tabular representation of DWH values corresponding to each value for a selected attribute annotation, providing an exhaustive examination of the interrelationships for the field. The tool also computes the panmictic (null) range, which involves a label permutation test to generate the null distribution of DWH values. This feature establishes a comparative baseline that aids in determining the significance of homophily <italic>versus</italic> what would be expected by chance.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The user interface of the AUTO-TUNE web application (<ext-link ext-link-type="uri" xlink:href="http://autotune.datamonkey.org/analyze">http://autotune.datamonkey.org/analyze</ext-link>). The platform provides a multi-faceted view of AUTO-TUNE&#x2019;s analysis, including a score plot that visualizes trends across different genetic distance thresholds. It also displays graphs of the number of clusters and the R1/R2 ratio&#x2014;both key metrics in AUTO-TUNE&#x2019;s heuristic scoring system. These interactive visualizations aid researchers in making nuanced decisions for threshold selection, especially when multiple thresholds yield similar scores.</p>
</caption>
<graphic xlink:href="fbinf-04-1400003-g002.tif"/>
</fig>
<p>The visualization code is available on Github (<ext-link ext-link-type="uri" xlink:href="https://github.com/stevenweaver/autotune-app/">https://github.com/stevenweaver/autotune-app/</ext-link>).</p>
</sec>
<sec id="s2-5">
<title>2.5 Comparisons with previously published analyses</title>
<p>First, we set out to compare the thresholds used in numerous published studies with those obtained by AUTO-TUNE. To select the data sets for this analysis, we conducted a scientific literature search to identify studies focused on HIV networks for public health purposes. We then filtered the studies that utilized HIV-TRACE to infer genetic networks and had publicly available sequences. Due to privacy concerns, HIV-1 sequences are frequently not released in the public domain (<xref ref-type="bibr" rid="B31">Inzaule et al., 2023</xref>). Some of the best-sampled datasets are national-level cohorts, such as the UK HIV Drug Resistance Database (<xref ref-type="bibr" rid="B16">Dunn and Pillay, 2007</xref>), the Swiss HIV cohort (<xref ref-type="bibr" rid="B66">Scherrer et al., 2022</xref>), or the Dutch ATHENA cohort (<xref ref-type="bibr" rid="B5">Boender et al., 2018</xref>). However, because sequences from these cohorts are not in the public domain and are typically subject to strong usage restrictions, we elected not to use such data, for reasons of reproducibility, practicality, and data transparency.</p>
<p>We also attempted to include studies from different countries and regions, enabling us to assess the performance of our method across various epidemic contexts, risk groups, and network sizes in real-data sets that used variable clustering thresholds.</p>
<p>Second, we compared AUTO-TUNE with the most direct published alternative: the <monospace>clustuneR</monospace> method (<xref ref-type="bibr" rid="B10">Chato et al., 2020</xref>). We procured datasets from <xref ref-type="bibr" rid="B81">Wolf et al. (2017)</xref> and <xref ref-type="bibr" rid="B77">Vrancken et al. (2017)</xref> utilizing the approach delineated in <xref ref-type="bibr" rid="B10">Chato et al. (2020)</xref>. These datasets, namely, Middle Tennessee, Seattle, and Alberta were processed using the workflow described in <xref ref-type="sec" rid="s2-3">Section 2.3</xref>. This enabled us to determine an optimal threshold for each dataset using AUTO-TUNE. We further executed the command as detailed in step 4 of <xref ref-type="sec" rid="s2-3">Section 2.3</xref>, deploying thresholds previously established as optimal by <xref ref-type="bibr" rid="B10">Chato et al. (2020)</xref>. Note that <monospace>clustuneR</monospace> requires and uses temporal information (dates sequences were collected), whereas AUTO-TUNE does not.</p>
<p>Lastly, we evaluated the effect of sampling density on the genetic distance threshold as determined by AUTO-TUNE, we implemented a strategy of random subsampling from the original dataset sourced from <xref ref-type="bibr" rid="B63">Rhee et al. (2019)</xref>. This study was selected due to its satisfactory AUTO-TUNE score when utilized in its entirety, as well as its inherent design as a Geographically-Stratified set of 716 <italic>pol</italic> Subtype/CRF (GSPS) reference sequence dataset. The dataset, which comprises 6034 samples gathered between 1989 and 2016, was subjected to random subsampling ten times at proportions of 25%, 50%, and 75% of the original sample size. For each subsample, the optimal threshold and associated scores were determined via AUTO-TUNE.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<sec id="s3-1">
<title>3.1 Comparisons with published HIV-1 molecular epidemiology studies</title>
<p>We selected several publications citing HIV-TRACE for our analysis, primarily because these studies not only referenced the tool but also made some or all of their sequence data publicly available (<xref ref-type="table" rid="T1">Tables 1</xref>, <xref ref-type="table" rid="T2">2</xref>). These studies adopted several different approaches for selecting genetic distance thresholds, including using US CDC guidelines (<xref ref-type="bibr" rid="B82">Yan et al., 2020</xref>), picking thresholds based on prior studies (<xref ref-type="bibr" rid="B67">Sivay et al., 2018</xref>), and visually inspecting the numbers of clusters and nodes in the networks across candidate distance thresholds (<xref ref-type="bibr" rid="B41">Liu et al., 2020</xref>). These thresholds, often qualitatively determined, tended to be round numbers, and were usually determined using <italic>ad hoc</italic> or subjective procedures. Some studies stratified their analyses by viral subtype (major clade), while others did not (or this was not applicable).</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Comparison of AUTO-TUNE and published thresholds from prior studies using partial HIV-1 polymerase gene sequences. <italic>N</italic>: the number of sequences; <italic>L</italic>: length of the multiple sequence alignment, bp; <italic>E</italic> [<italic>D</italic>] mean pairwise TN93 distance; (the studies are sorted on this column, in ascending order) &#xb6;: the original study performed threshold tuning (varied methods); &#x2020;: distance thresholds were specific to subtypes; &#x22c6;: the corresponding AUTO-TUNE score is <inline-formula id="inf12">
<mml:math id="m14">
<mml:mo>&#x2265;</mml:mo>
<mml:mn>1.9</mml:mn>
</mml:math>
</inline-formula>; &#x2022;: only a subset of the complete dataset was made available (privacy, data use restrictions, incomplete GenBank submissions), the number of sequences analyzed here is shown after the/symbol; N.R: not reported.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left">References</th>
<th rowspan="2" align="left">
<italic>N</italic>
</th>
<th rowspan="2" align="left">
<italic>L</italic>
</th>
<th rowspan="2" align="left">
<italic>E</italic> [<italic>D</italic>] (%)</th>
<th rowspan="2" align="left">Scope</th>
<th rowspan="2" align="left">Location/Country</th>
<th rowspan="2" align="left">Timespan</th>
<th rowspan="2" align="left">Common subtypes</th>
<th colspan="2" align="center">Distance threshold, sub/site</th>
</tr>
<tr>
<th align="left">Published</th>
<th align="left">AUTO-TUNE</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<xref ref-type="bibr" rid="B87">Zai et al. (2020)</xref>
</td>
<td align="left">209</td>
<td align="left">1056</td>
<td align="left">1.5</td>
<td align="left">Country</td>
<td align="left">China</td>
<td align="left">2007&#x2013;2015</td>
<td align="left">CRF55/01B</td>
<td align="left">
<italic>&#xb6;</italic> 0.002</td>
<td align="left">0.00255</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B41">Liu et al. (2020)</xref>
</td>
<td align="left">2087/1907 &#x2022;</td>
<td align="left">1053</td>
<td align="left">5.3</td>
<td align="left">City</td>
<td align="left">Shenyang, China</td>
<td align="left">2008&#x2013;2016</td>
<td align="left">CRF01, CRF07, B</td>
<td align="left">
<italic>&#xb6;</italic> 0.005/0.007 &#x2020;</td>
<td align="left">0.00621</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B12">Dalai et al. (2018)</xref>
</td>
<td align="left">317</td>
<td align="left">1044</td>
<td align="left">5.5</td>
<td align="left">City</td>
<td align="left">San Mateo, CA, United States of America</td>
<td align="left">1997&#x2013;2008</td>
<td align="left">96% B</td>
<td align="left">0.02</td>
<td align="left">0.01944</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B10">Chato et al. (2020)</xref>
</td>
<td align="left">1653/1840 &#x2022;</td>
<td align="left">1020</td>
<td align="left">5.5</td>
<td align="left">City</td>
<td align="left">Seattle, United States of America</td>
<td align="left">2000&#x2013;2013</td>
<td align="left">B</td>
<td align="left">
<italic>&#xb6;</italic>0.016</td>
<td align="left">0.01538</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B10">Chato et al. (2020)</xref>
</td>
<td align="left">808</td>
<td align="left">1017</td>
<td align="left">5.6</td>
<td align="left">Province</td>
<td align="left">Northern Alberta, Canada</td>
<td align="left">2007&#x2013;2013</td>
<td align="left">B</td>
<td align="left">
<italic>&#xb6;</italic>0.0104</td>
<td align="left">0.01201</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B40">Little et al. (2014)</xref>
</td>
<td align="left">648/646 &#x2022;</td>
<td align="left">1212</td>
<td align="left">5.9</td>
<td align="left">City</td>
<td align="left">San Diego, CA, United States of America</td>
<td align="left">1996&#x2013;2011</td>
<td align="left">98.5% B</td>
<td align="left">0.015</td>
<td align="left">0.02495</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B55">P&#xe9;rez-Losada et al. (2017)</xref>
</td>
<td align="left">1879/3411 &#x2022;</td>
<td align="left">1027</td>
<td align="left">6.0</td>
<td align="left">City</td>
<td align="left">Washington DC, United States of America</td>
<td align="left">1987&#x2013;2015</td>
<td align="left">B</td>
<td align="left">0.010</td>
<td align="left">0.01733</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B63">Rhee et al. (2019)</xref>
</td>
<td align="left">4553</td>
<td align="left">897</td>
<td align="left">6.1</td>
<td align="left">State</td>
<td align="left">CA, United States of America</td>
<td align="left">1998&#x2013;2016</td>
<td align="left">95.5% B</td>
<td align="left">0.015</td>
<td align="left">0.01139 &#x22c6;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B10">Chato et al. (2020)</xref>
</td>
<td align="left">2779/2750 &#x2022;</td>
<td align="left">1398</td>
<td align="left">6.3</td>
<td align="left">State</td>
<td align="left">Tennessee, United States of America</td>
<td align="left">2001&#x2013;2015</td>
<td align="left">B</td>
<td align="left">
<italic>&#xb6;</italic>0.016</td>
<td align="left">0.01872</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B71">Temereanca et al. (2017)</xref>
</td>
<td align="left">37</td>
<td align="left">1302</td>
<td align="left">6.7</td>
<td align="left">City</td>
<td align="left">Bucharest, Romania</td>
<td align="left">2010&#x2013;2013</td>
<td align="left">F1, G, B</td>
<td align="left">0.015</td>
<td align="left">0.00194 &#x22c6;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B6">Brenner et al. (2021)</xref>
</td>
<td align="left">10945/448 &#x2022;</td>
<td align="left">738</td>
<td align="left">6.7</td>
<td align="left">Province</td>
<td align="left">Quebec, Canada</td>
<td align="left">2002&#x2013;2020</td>
<td align="left">B</td>
<td align="left">0.015/0.025</td>
<td align="left">0.02741</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B67">Sivay et al. (2018)</xref>
</td>
<td align="left">201</td>
<td align="left">1302</td>
<td align="left">6.9</td>
<td align="left">Province</td>
<td align="left">Mpumalanga, South Africa</td>
<td align="left">2011&#x2013;2015</td>
<td align="left">C</td>
<td align="left">0.025</td>
<td align="left">0.02506</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B39">Li et al. (2022)</xref>
</td>
<td align="left">295</td>
<td align="left">1206</td>
<td align="left">7.8</td>
<td align="left">Prefecture</td>
<td align="left">Pu&#x2019;er, China</td>
<td align="left">2021</td>
<td align="left">CRF08, CRF01, CRF07</td>
<td align="left">
<italic>&#xb6;</italic> 0.013</td>
<td align="left">0.01483 &#x22c6;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B86">Yu et al. (2022)</xref>
</td>
<td align="left">316</td>
<td align="left">1074</td>
<td align="left">7.9</td>
<td align="left">Province</td>
<td align="left">Guangxi, China</td>
<td align="left">2012&#x2013;2018</td>
<td align="left">CRF01, CRF08, CRF07</td>
<td align="left">
<italic>&#xb6;</italic>0.013</td>
<td align="left">0.01178</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B83">Yan et al. (2021)</xref>
</td>
<td align="left">1695/1569 &#x2022;</td>
<td align="left">1569</td>
<td align="left">8.4</td>
<td align="left">City</td>
<td align="left">Guangzhou, China</td>
<td align="left">2008&#x2013;2012</td>
<td align="left">CRF01, CRF07, CRF55,G</td>
<td align="left">0.015</td>
<td align="left">0.00839</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B4">Billings et al. (2019)</xref>
</td>
<td align="left">150</td>
<td align="left">1597</td>
<td align="left">8.5</td>
<td align="left">City</td>
<td align="left">Lagos, Nigeria</td>
<td align="left">2013&#x2013;2016</td>
<td align="left">CRF02, URF</td>
<td align="left">0.015</td>
<td align="left">0.0233</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B11">Chen et al. (2023)</xref>
</td>
<td align="left">1975/209 &#x2022;</td>
<td align="left">1050</td>
<td align="left">8.7</td>
<td align="left">Province</td>
<td align="left">Guangxi, China</td>
<td align="left">2016&#x2013;2018</td>
<td align="left">CRF01, CRF07, CRF08</td>
<td align="left">
<italic>&#xb6;</italic>0.0075</td>
<td align="left">0.01295</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B18">Fabeni et al. (2020)</xref>
</td>
<td align="left">726/3499 &#x2022;</td>
<td align="left">1029</td>
<td align="left">9.2</td>
<td align="left">Country</td>
<td align="left">Italy</td>
<td align="left">1998&#x2013;2018</td>
<td align="left">B</td>
<td align="left">0.010</td>
<td align="left">0.0037</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B37">Leal et al. (2020)</xref>
</td>
<td align="left">630/633 &#x2022;</td>
<td align="left">990</td>
<td align="left">9.2</td>
<td align="left">State</td>
<td align="left">Maranh&#xe3;o, Brazil</td>
<td align="left">2008&#x2013;2017</td>
<td align="left">B</td>
<td align="left">0.020</td>
<td align="left">0.04033</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B3">Bbosa et al. (2020)</xref>
</td>
<td align="left">2018</td>
<td align="left">1257</td>
<td align="left">9.3</td>
<td align="left">Country</td>
<td align="left">Uganda</td>
<td align="left">2009&#x2013;2016</td>
<td align="left">N.R</td>
<td align="left">0.015</td>
<td align="left">0.02035 &#x22c6;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B69">Stecher et al. (2018)</xref>
</td>
<td align="left">2774</td>
<td align="left">1028</td>
<td align="left">12.1</td>
<td align="left">Multi-City</td>
<td align="left">Germany</td>
<td align="left">1999&#x2013;2016</td>
<td align="left">B</td>
<td align="left">0.015</td>
<td align="left">0.03056</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Network properties at the published and AUTO-TUNE thresholds. In cases when the original paper used more than one threshold, we selected the largest for comparison. The datasets are ordered by the AUTO-TUNE priority score from highest to lowest. <italic>&#x3c1;</italic> is the fitted characteristic scale-free exponent of the corresponding degree distributions.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left">References</th>
<th rowspan="2" align="left">AUTO-TUNE score</th>
<th colspan="2" align="center">Nodes in network</th>
<th colspan="2" align="center">Clusters in network</th>
<th colspan="2" align="center">
<italic>R</italic>
<sub>12</sub>
</th>
<th colspan="2" align="center">Scale parameter <italic>&#x3c1;</italic>
</th>
</tr>
<tr>
<th align="left">Published</th>
<th align="left">AUTO-TUNE</th>
<th align="left">Published</th>
<th align="left">AUTO-TUNE</th>
<th align="left">Published</th>
<th align="left">AUTO-TUNE</th>
<th align="left">Published</th>
<th align="left">AUTO-TUNE</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<xref ref-type="bibr" rid="B39">Li et al. (2022)</xref>
</td>
<td align="left">2.00</td>
<td align="left">1364</td>
<td align="left">1224</td>
<td align="left">277</td>
<td align="left">277</td>
<td align="left">1.7</td>
<td align="left">2.4</td>
<td align="left">2.8</td>
<td align="left">2.6</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B10">Chato et al. (2020)</xref> TN</td>
<td align="left">2.00</td>
<td align="left">394</td>
<td align="left">445</td>
<td align="left">108</td>
<td align="left">109</td>
<td align="left">1.0</td>
<td align="left">1.7</td>
<td align="left">2.7</td>
<td align="left">2.9</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B63">Rhee et al. (2019)</xref>
</td>
<td align="left">1.95</td>
<td align="left">2044</td>
<td align="left">1636</td>
<td align="left">524</td>
<td align="left">488</td>
<td align="left">13.2</td>
<td align="left">1.5</td>
<td align="left">2.6</td>
<td align="left">2.7</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B3">Bbosa et al. (2020)</xref>
</td>
<td align="left">1.93</td>
<td align="left">222</td>
<td align="left">296</td>
<td align="left">102</td>
<td align="left">119</td>
<td align="left">2.2</td>
<td align="left">1.6</td>
<td align="left">3.2</td>
<td align="left">2.6</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B12">Dalai et al. (2018)</xref>
</td>
<td align="left">1.89</td>
<td align="left">60</td>
<td align="left">54</td>
<td align="left">9</td>
<td align="left">11</td>
<td align="left">22</td>
<td align="left">2.6</td>
<td align="left">2.0</td>
<td align="left">2.2</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B71">Temereanca et al. (2017)</xref>
</td>
<td align="left">1.79</td>
<td align="left">30</td>
<td align="left">16</td>
<td align="left">5</td>
<td align="left">3</td>
<td align="left">3</td>
<td align="left">1.5</td>
<td align="left">N/A</td>
<td align="left">2.8</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B86">Yu et al. (2022)</xref>
</td>
<td align="left">1.76</td>
<td align="left">55</td>
<td align="left">51</td>
<td align="left">19</td>
<td align="left">19</td>
<td align="left">2.75</td>
<td align="left">1.75</td>
<td align="left">10.4</td>
<td align="left">34.0</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B67">Sivay et al. (2018)</xref>
</td>
<td align="left">1.42</td>
<td align="left">51</td>
<td align="left">51</td>
<td align="left">19</td>
<td align="left">19</td>
<td align="left">1.5</td>
<td align="left">1.5</td>
<td align="left">3.2</td>
<td align="left">3.0</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B87">Zai et al. (2020)</xref>
</td>
<td align="left">1.40</td>
<td align="left">96</td>
<td align="left">98</td>
<td align="left">26</td>
<td align="left">27</td>
<td align="left">1.5</td>
<td align="left">1.5</td>
<td align="left">24.1</td>
<td align="left">17.7</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B40">Little et al. (2014)</xref>
</td>
<td align="left">1.31</td>
<td align="left">301</td>
<td align="left">394</td>
<td align="left">98</td>
<td align="left">87</td>
<td align="left">2.5</td>
<td align="left">6.1</td>
<td align="left">3.6</td>
<td align="left">3.1</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B6">Brenner et al. (2021)</xref>
</td>
<td align="left">1.22</td>
<td align="left">363</td>
<td align="left">379</td>
<td align="left">71</td>
<td align="left">70</td>
<td align="left">5.6</td>
<td align="left">5.5</td>
<td align="left">2.7</td>
<td align="left">2.8</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B69">Stecher et al. (2018)</xref>
</td>
<td align="left">1.20</td>
<td align="left">97</td>
<td align="left">558</td>
<td align="left">36</td>
<td align="left">155</td>
<td align="left">2.2</td>
<td align="left">4.9</td>
<td align="left">3.2</td>
<td align="left">3.3</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B10">Chato et al. (2020)</xref> Seattle</td>
<td align="left">1.16</td>
<td align="left">505</td>
<td align="left">484</td>
<td align="left">148</td>
<td align="left">149</td>
<td align="left">2.5</td>
<td align="left">1.7</td>
<td align="left">2.7</td>
<td align="left">2.6</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B4">Billings et al. (2019)</xref>
</td>
<td align="left">1.16</td>
<td align="left">38</td>
<td align="left">78</td>
<td align="left">13</td>
<td align="left">23</td>
<td align="left">2</td>
<td align="left">2.3</td>
<td align="left">2.6</td>
<td align="left">11.5</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B83">Yan et al. (2021)</xref>
</td>
<td align="left">1.14</td>
<td align="left">1084</td>
<td align="left">753</td>
<td align="left">124</td>
<td align="left">116</td>
<td align="left">2.0</td>
<td align="left">1.8</td>
<td align="left">1.2</td>
<td align="left">2.0</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B11">Chen et al. (2023)</xref>
</td>
<td align="left">1.11</td>
<td align="left">20</td>
<td align="left">47</td>
<td align="left">8</td>
<td align="left">16</td>
<td align="left">1.3</td>
<td align="left">2.0</td>
<td align="left">
<italic>&#x221e;</italic>
</td>
<td align="left">
<italic>&#x221e;</italic>
</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B37">Leal et al. (2020)</xref>
</td>
<td align="left">1.11</td>
<td align="left">50</td>
<td align="left">270</td>
<td align="left">25</td>
<td align="left">57</td>
<td align="left">1</td>
<td align="left">1.6</td>
<td align="left">53.6</td>
<td align="left">3.1</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B55">P&#xe9;rez-Losada et al. (2017)</xref>
</td>
<td align="left">1.06</td>
<td align="left">172</td>
<td align="left">431</td>
<td align="left">76</td>
<td align="left">134</td>
<td align="left">5.1</td>
<td align="left">1.4</td>
<td align="left">5.2</td>
<td align="left">2.9</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B41">Liu et al. (2020)</xref>
</td>
<td align="left">1.05</td>
<td align="left">885</td>
<td align="left">797</td>
<td align="left">156</td>
<td align="left">161</td>
<td align="left">6.0</td>
<td align="left">4.5</td>
<td align="left">3.1</td>
<td align="left">3.0</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B10">Chato et al. (2020)</xref> Alberta</td>
<td align="left">1.03</td>
<td align="left">394</td>
<td align="left">445</td>
<td align="left">108</td>
<td align="left">109</td>
<td align="left">1.0</td>
<td align="left">1.7</td>
<td align="left">2.7</td>
<td align="left">2.9</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B18">Fabeni et al. (2020)</xref>
</td>
<td align="left">1.00</td>
<td align="left">626</td>
<td align="left">221</td>
<td align="left">197</td>
<td align="left">83</td>
<td align="left">2.1</td>
<td align="left">3.2</td>
<td align="left">2.1</td>
<td align="left">3.2</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>A direct comparison with published networks is not feasible because only the underlying sequence data (and often only some of the sequences) are made available, not the networks themselves. To facilitate comparison here, we used distance thresholds and all available sequences from primary publications to infer transmission networks anew (the scripts for doing so and the corresponding settings are available in github. com/veg/auto-tune) and compare them with the networks obtained using the highest scoring AUTO-TUNE threshold.</p>
<p>With a few exceptions (e.g., <xref ref-type="bibr" rid="B12">Dalai et al. (2018)</xref>; <xref ref-type="bibr" rid="B67">Sivay et al. (2018)</xref>), both the distance thresholds and the inferred networks were quite different, in terms of the numbers of connected nodes, clusters, degree distributions, and even hyper-parameters, such as the characteristic exponent of the scale free degree distribution, <italic>&#x3c1;</italic>. This is true even for the studies where the published threshold was tuned (typically to maximize the number of clusters). AUTO-TUNE thresholds were larger than the published values in 13/21 datasets, and smaller in 8/21 datasets.</p>
<sec id="s3-1-1">
<title>3.1.1 Examples of how changing thresholds affects inferred networks</title>
<sec id="s3-1-1-1">
<title>3.1.1.1 Cluster size reduction</title>
<p>The 0.02 subs/site (substitutions/site) threshold used by <xref ref-type="bibr" rid="B12">Dalai et al. (2018)</xref>, yielded one large cluster composed of two loosely connected components (one PWID/HSX, one MSM, see <xref ref-type="fig" rid="F3">Figure 3</xref> in that paper). A minute change to the threshold by AUTO-TUNE to 0.0194 subs/site splits one large cluster into three (some nodes also became disconnected), separating the two major risk groups; this is because the &#x201c;bridging&#x201d; connections were between these two thresholds (see <xref ref-type="fig" rid="F4">Figure 4A</xref>). This minor change also reduced <italic>R</italic>
<sub>12</sub> from 21 to 2.6.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>
<bold>(A)</bold> Box plot representing the AUTO-TUNE scores across ten random samples at 25%, 50%, and 75% of the (<xref ref-type="bibr" rid="B63">Rhee et al., 2019</xref>) dataset, showing a trend of increasing confidence in score estimates with denser sampling. <bold>(B)</bold> Box plot of the selected distance thresholds across the same random samples at 25%, 50%, and 75% proportions, demonstrating improved consistency in threshold selection with increased sample size. <bold>(C)</bold> Scatterplot of the chosen thresholds (<italic>Y</italic>-axis) against their corresponding AUTO-TUNE scores (<italic>X</italic>-axis) for the three subsample proportions.</p>
</caption>
<graphic xlink:href="fbinf-04-1400003-g003.tif"/>
</fig>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Examples of AUTO-TUNE scores profiles. <bold>(A)</bold> Lowering the genetic distance threshold removes some of the edges from the network (shown in grey) and disconnects a large cluster into color-coded smaller clusters; here &#x201c;None&#x201d; means that the node is not connected to anything at the lower threshold. <bold>(B)</bold> Raising the genetic distance threshold adds edges to the network (shown in grey) and connectes previously separte clusters into a larger component <bold>(C)</bold> Each circle is a cluster in the larger threshold network, and with a proportion of nodes removed when the threshold is lowered. <bold>(D)</bold> Changes to the node degree distribution (colors represent the counts of nodes with the same degree). <bold>(E)</bold> A significant enlargement of a small network at a higher threshold, with grey edges only present at the larger threshold.</p>
</caption>
<graphic xlink:href="fbinf-04-1400003-g004.tif"/>
</fig>
</sec>
<sec id="s3-1-1-2">
<title>3.1.1.2 Cluster size increase</title>
<p>Increasing the threshold from 0.015 to 0.02495 subs/site on data from <xref ref-type="bibr" rid="B40">Little et al. (2014)</xref> combined several small clusters (and singletons) into a single larger cluster, while preserving the overall size and properties of the network (see <xref ref-type="fig" rid="F4">Figure 4B</xref>). This change also reduced <italic>R</italic>
<sub>12</sub> from 2.5 to 1.5.</p>
</sec>
<sec id="s3-1-1-3">
<title>3.1.1.3 Thinning out the network</title>
<p>Reducing the threshold from 0.015 to 0.01139 subs/site on data from <xref ref-type="bibr" rid="B63">Rhee et al. (2019)</xref> dramatically reduced the size of the largest cluster, and thinned out most clusters with five or more nodes (see <xref ref-type="fig" rid="F4">Figure 4C</xref>). This is different from the <xref ref-type="bibr" rid="B12">Dalai et al. (2018)</xref> case above, because the entire network is affected, rather than a single or a few clusters.</p>
</sec>
<sec id="s3-1-1-4">
<title>3.1.1.4 Materially changing the degree distribution of the network</title>
<p>For the sequences from <xref ref-type="bibr" rid="B39">Li et al. (2022)</xref>, AUTO-TUNE suggests <italic>D</italic> &#x3d; 0.01483 subs/site with robust (1.76) confidence, whereas the original <italic>D</italic> &#x3d; 0.013 subs/site was selected based on maximizing the number of clusters (and likely rounding to the nearest decimal). While the total number of the clusters only increases by 1, the number of nodes connected in the network grows from 95 to 119, and the scale free exponent of the distribution is dramatically affected. The latter is informed by the degree distribution of the network, and <xref ref-type="fig" rid="F4">Figure 4D</xref> shows, the degree distribution is dramatically affected. The degree distribution of network, which tabulates the number edges connected to each node, is a fundamental feature of network analysis. For each integer 0 &#x2264; <italic>K</italic> &#x2264; <italic>K</italic>
<sub>max</sub>, the degree distribution function counts how many nodes have exactly <italic>K</italic> edges connected to them. <italic>K</italic>
<sub>max</sub> is simply the highest such number for a given network. Many commonly used network-derived correlates (e.g., degree centrality) can be strongly affected by such changes.</p>
</sec>
<sec id="s3-1-1-5">
<title>3.1.1.5 Expanding the network</title>
<p>Increasing the .015 subs/site threshold in <xref ref-type="bibr" rid="B4">Billings et al. (2019)</xref> to 0.0233 subs/site more than doubles the number of nodes included (<xref ref-type="fig" rid="F4">Figure 4E</xref>). This is distnict from the <xref ref-type="bibr" rid="B63">Rhee et al. (2019)</xref> case above, because, once again, most of the network is affected, rather than a few key clusters.</p>
</sec>
</sec>
</sec>
<sec id="s3-2">
<title>3.2 Interpretation</title>
<p>Networks with high AUTO-TUNE scores are exemplified by the alignment (in the distance space) of the points where the number of clusters is maximized and where the network transitions to having an &#x201c;unusually&#x201d; large cluster (see <xref ref-type="fig" rid="F5">Figure 5A</xref>). In cases of low scores, AUTO-TUNE effectively falls back to maximizing the number of clusters as a function of the distance thresholds, which is a common strategy found in empirical studies (see <xref ref-type="fig" rid="F5">Figure 5B</xref>).</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Examples of how changing thresholds affects inferred networks. <bold>(A)</bold> A high-scoring network <xref ref-type="bibr" rid="B3">Bbosa et al. (2020)</xref> has a distance threshold which achieves the number of clusters near the maximum, while also avoiding the formation of a large (weakly connected) cluster. <bold>(B)</bold> A low-scoring network <xref ref-type="bibr" rid="B41">Liu et al. (2020)</xref> has a misalignment between the distance for which the maximum number of clusters is found, and where the big jumps in the cluster size ratio occur. Here, AUTO-TUNE effectively optimizes the number of clusters while preventing excessive growth of the largest cluster.</p>
</caption>
<graphic xlink:href="fbinf-04-1400003-g005.tif"/>
</fig>
<p>As expected, AUTO-TUNE inferred smaller thresholds for younger (e.g., studies based in China) epidemics. While AUTO-TUNE will always return a score, in the majority of cases there is no clear &#x201c;winner&#x201d; threshold, with priority scores exceeding 1.5 in only 6/18 cases (<xref ref-type="table" rid="T2">Table 2</xref>). One interpretation for such lack of clarity is that the underlying network has several different (e.g., spatial, temporal, or subtype-specific) thresholds which cannot be well-represented by any single value. For instance, when analyzing the data from <xref ref-type="bibr" rid="B83">Yan et al. (2021)</xref>, AUTO-TUNE returned a low score of 1.14 for <italic>D</italic> &#x3d; 0.00839 subs/site. However, when we split the data into major constituent subtypes and ran AUTO-TUNE on each one separately, starkly discrepant thresholds were found for different subtypes: <italic>D</italic> &#x3d; 0.0102 subs/site (score &#x3d; 1.59) for CRF01, <italic>D</italic> &#x3d; 0.00193 subs/site (score &#x3d; 2) for CRF05, <italic>D</italic> &#x3d; 0.02615 subs/site (score &#x3d; 1.65) for B, and <italic>D</italic> &#x3d; 0.0111 subs/site (score &#x3d; 1.04) for CRF07. Although many networks from the literature tend to be dominated by sequences from the same subtype, in more heterogeneous settings it seems prudent to partition the data by subtype (corresponding to major phylogenetic clades), and perform network analyses within subtypes.</p>
</sec>
<sec id="s3-3">
<title>3.3 Minimum cluster size setting</title>
<p>We explored how selecting the minimum number of connected nodes needed to define a cluster affected the selected threshold and the score for the same collection of 21 empirical datasets (<xref ref-type="table" rid="T3">Table 3</xref>). For the majority of the datasets, requiring three or more connected nodes to define a cluster had a minor effect on the selected distance, with the maximal priority score achieved for the standard minimum of two nodes. The few exceptions where higher distances yield larger scores come from very small networks, where there are only a few clusters, e.g., <xref ref-type="bibr" rid="B11">Chen et al. (2023)</xref>; <xref ref-type="bibr" rid="B12">Dalai et al. (2018)</xref>.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>AUTO-TUNE distance thresholds and scores as a function of the minimal cluster size. Rows are sorted by the score at cluster size &#x2265;2. The maximum score (or scores) for each row are highlighted in boldface.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="3" align="left">References</th>
<th colspan="8" align="center">AUTO-TUNE minimum cluster size</th>
</tr>
<tr>
<th colspan="2" align="center">Size &#x2265;2</th>
<th colspan="2" align="center">Size &#x2265;3</th>
<th colspan="2" align="center">Size &#x2265;4</th>
<th colspan="2" align="center">Size &#x2265;5</th>
</tr>
<tr>
<th align="left">Threshold, %</th>
<th align="left">Score</th>
<th align="left">Threshold, %</th>
<th align="left">Score</th>
<th align="left">Threshold, %</th>
<th align="left">Score</th>
<th align="left">Threshold, %</th>
<th align="left">Score</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<xref ref-type="bibr" rid="B39">Li et al. (2022)</xref>
</td>
<td align="left">1.15</td>
<td align="left">
<bold>2.0</bold>
</td>
<td align="left">1.15</td>
<td align="left">1.97</td>
<td align="left">1.8</td>
<td align="left">1.4</td>
<td align="left">1.15</td>
<td align="left">
<bold>2.0</bold>
</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B10">Chato et al. (2020)</xref> TN</td>
<td align="left">1.87</td>
<td align="left">
<bold>2.0</bold>
</td>
<td align="left">1.87</td>
<td align="left">
<bold>2.0</bold>
</td>
<td align="left">1.87</td>
<td align="left">
<bold>2.0</bold>
</td>
<td align="left">1.87</td>
<td align="left">
<bold>2.0</bold>
</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B63">Rhee et al. (2019)</xref>
</td>
<td align="left">1.14</td>
<td align="left">
<bold>1.94</bold>
</td>
<td align="left">1.14</td>
<td align="left">1.17</td>
<td align="left">2.05</td>
<td align="left">1.12</td>
<td align="left">1.14</td>
<td align="left">1.11</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B3">Bbosa et al. (2020)</xref>
</td>
<td align="left">2.04</td>
<td align="left">
<bold>1.94</bold>
</td>
<td align="left">2.49</td>
<td align="left">1.01</td>
<td align="left">2.04</td>
<td align="left">1.00</td>
<td align="left">2.31</td>
<td align="left">1.03</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B12">Dalai et al. (2018)</xref>
</td>
<td align="left">1.94</td>
<td align="left">1.89</td>
<td align="left">1.50</td>
<td align="left">1.02</td>
<td align="left">1.95</td>
<td align="left">1.00</td>
<td align="left">1.95</td>
<td align="left">
<bold>2.0</bold>
</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B71">Temereanca et al. (2017)</xref>
</td>
<td align="left">0.19</td>
<td align="left">
<bold>1.78</bold>
</td>
<td align="left">0.27</td>
<td align="left">2.0</td>
<td align="left">2.79</td>
<td align="left">1.00</td>
<td align="left">2.79</td>
<td align="left">1.06</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B86">Yu et al. (2022)</xref>
</td>
<td align="left">1.18</td>
<td align="left">
<bold>1.76</bold>
</td>
<td align="left">1.22</td>
<td align="left">1.00</td>
<td align="left">2.67</td>
<td align="left">1.18</td>
<td align="left">2.67</td>
<td align="left">1.37</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B67">Sivay et al. (2018)</xref>
</td>
<td align="left">2.51</td>
<td align="left">
<bold>1.41</bold>
</td>
<td align="left">3.68</td>
<td align="left">1.07</td>
<td align="left">3.67</td>
<td align="left">1.07</td>
<td align="left">3.67</td>
<td align="left">1.07</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B87">Zai et al. (2020)</xref>
</td>
<td align="left">0.26</td>
<td align="left">
<bold>1.41</bold>
</td>
<td align="left">0.26</td>
<td align="left">1.41</td>
<td align="left">0.26</td>
<td align="left">1.19</td>
<td align="left">0.27</td>
<td align="left">1.18</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B40">Little et al. (2014)</xref>
</td>
<td align="left">2.35</td>
<td align="left">
<bold>1.58</bold>
</td>
<td align="left">2.35</td>
<td align="left">1.20</td>
<td align="left">1.76</td>
<td align="left">1.04</td>
<td align="left">2.10</td>
<td align="left">1.03</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B6">Brenner et al. (2021)</xref>
</td>
<td align="left">2.74</td>
<td align="left">1.22</td>
<td align="left">2.74</td>
<td align="left">1.22</td>
<td align="left">2.01</td>
<td align="left">
<bold>1.26</bold>
</td>
<td align="left">2.01</td>
<td align="left">1.26</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B69">Stecher et al. (2018)</xref>
</td>
<td align="left">3.06</td>
<td align="left">
<bold>1.20</bold>
</td>
<td align="left">3.46</td>
<td align="left">1.14</td>
<td align="left">3.49</td>
<td align="left">1.12</td>
<td align="left">3.88</td>
<td align="left">1.09</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B10">Chato et al. (2020)</xref> Seattle</td>
<td align="left">1.53</td>
<td align="left">
<bold>1.16</bold>
</td>
<td align="left">1.76</td>
<td align="left">1.11</td>
<td align="left">1.54</td>
<td align="left">1.16</td>
<td align="left">1.54</td>
<td align="left">1.09</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B4">Billings et al. (2019)</xref>
</td>
<td align="left">2.33</td>
<td align="left">1.16</td>
<td align="left">2.97</td>
<td align="left">1.12</td>
<td align="left">2.33</td>
<td align="left">1.17</td>
<td align="left">3.00</td>
<td align="left">
<bold>1.31</bold>
</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B83">Yan et al. (2021)</xref>
</td>
<td align="left">1.22</td>
<td align="left">
<bold>1.09</bold>
</td>
<td align="left">1.29</td>
<td align="left">1.01</td>
<td align="left">1.48</td>
<td align="left">1.00</td>
<td align="left">1.62</td>
<td align="left">1.01</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B11">Chen et al. (2023)</xref>
</td>
<td align="left">1.30</td>
<td align="left">1.11</td>
<td align="left">2.7</td>
<td align="left">1.31</td>
<td align="left">1.89</td>
<td align="left">1.08</td>
<td align="left">2.73</td>
<td align="left">
<bold>1.83</bold>
</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B37">Leal et al. (2020)</xref>
</td>
<td align="left">4.03</td>
<td align="left">1.11</td>
<td align="left">4.03</td>
<td align="left">1.12</td>
<td align="left">4.24</td>
<td align="left">1.19</td>
<td align="left">4.05</td>
<td align="left">
<bold>1.21</bold>
</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B55">P&#xe9;rez-Losada et al. (2017)</xref>
</td>
<td align="left">1.73</td>
<td align="left">
<bold>1.06</bold>
</td>
<td align="left">2.04</td>
<td align="left">1.04</td>
<td align="left">2.04</td>
<td align="left">1.03</td>
<td align="left">2.04</td>
<td align="left">1.04</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B41">Liu et al. (2020)</xref>
</td>
<td align="left">0.62</td>
<td align="left">
<bold>1.13</bold>
</td>
<td align="left">0.62</td>
<td align="left">
<bold>1.13</bold>
</td>
<td align="left">0.62</td>
<td align="left">1.13</td>
<td align="left">1.29</td>
<td align="left">1.00</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B10">Chato et al. (2020)</xref> Alberta</td>
<td align="left">1.20</td>
<td align="left">
<bold>1.03</bold>
</td>
<td align="left">3.87</td>
<td align="left">1.00</td>
<td align="left">3.87</td>
<td align="left">1</td>
<td align="left">1.15</td>
<td align="left">1.01</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B18">Fabeni et al. (2020)</xref>
</td>
<td align="left">1.77</td>
<td align="left">1.00</td>
<td align="left">3.05</td>
<td align="left">1.00</td>
<td align="left">3.97</td>
<td align="left">1.15</td>
<td align="left">3.97</td>
<td align="left">
<bold>1.45</bold>
</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-4">
<title>3.4 Comparisons with published non-HIV molecular epidemiology studies</title>
<p>While HIV-1 epidemiology is the predominant niche for distance-based molecular transmission analyses, other rapidly evolving viruses, especially HCV, have also been analyzed with these approaches (<xref ref-type="bibr" rid="B2">Bartlett et al., 2017</xref>). Unlike HIV-1, there is considerably less work on how to choose an appropriate distance threshold, further complicated by the use of different genes to build networks (see <xref ref-type="bibr" rid="B9">Chan et al. (2020)</xref> for a comprehensive summary). Two commonly seen methods exist: use some measure of intra-host variation (obtained by deep sequencing) as a lower bound for the threshold, or tune <italic>D</italic> to obtain some desired network property, e.g., the maximal number of clusters. Like with HIV-1, we searched the literature for relevant studies, and selected several with publicly available sequence data.</p>
<p>Most of the datasets are much smaller and less systematically sampled than those for HIV-1, and often combine highly divergent subtypes in the same collection, making a joint analysis challenging. As with HIV-1, AUTO-TUNE returns a wide range of scores and <italic>D</italic> thresholds. For example, effectively maximizing the number of clusters on rhinovirus sequences from <xref ref-type="bibr" rid="B46">Ng et al. (2022)</xref> yields a <italic>D</italic> estimate very similar to that obtained by the authors from intra-host variability (information not available to AUTO-TUNE). <xref ref-type="table" rid="T4">Table 4</xref>.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Comparison of AUTO-TUNE and published thresholds from prior studies using sequences from viruses other than HIV-1. &#x201c;N/A&#x201d;: no distance-based clustering analyses were done. Other notation is the same as in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left">References</th>
<th rowspan="2" align="left">Virus</th>
<th rowspan="2" align="left">Gene</th>
<th rowspan="2" align="left">
<italic>N</italic>
</th>
<th rowspan="2" align="left">
<italic>L</italic>
</th>
<th rowspan="2" align="left">
<italic>E</italic> [<italic>D</italic>] (%)</th>
<th rowspan="2" align="left">Scope</th>
<th rowspan="2" align="left">Location/Country</th>
<th rowspan="2" align="left">Timespan</th>
<th colspan="2" align="center">Distance threshold, subs/site</th>
</tr>
<tr>
<th align="left">Published</th>
<th align="left">AUTO-TUNE (score)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<xref ref-type="bibr" rid="B32">Jia et al. (2023)</xref>
</td>
<td align="left">HCV</td>
<td align="left">
<italic>NS5B</italic>
</td>
<td align="left">503</td>
<td align="left">315</td>
<td align="left">34.9</td>
<td align="left">Province</td>
<td align="left">Yunnan, China</td>
<td align="left">2008&#x2013;2018</td>
<td align="left">N/A</td>
<td align="left">1.933 (1.92)</td>
</tr>
<tr>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="left">97</td>
<td align="left"/>
<td align="left">8.0</td>
<td colspan="3" align="left">Genotype 1b only</td>
<td align="left">
<italic>&#xb6;</italic> 2.3</td>
<td align="left">1.944 (2.0)</td>
</tr>
<tr>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="left">53</td>
<td align="left"/>
<td align="left">7.4</td>
<td colspan="3" align="left">Genotype 2a only</td>
<td align="left">
<italic>&#xb6;</italic> 3.3</td>
<td align="left">3.3 (1.3)</td>
</tr>
<tr>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="left">110</td>
<td align="left"/>
<td align="left">5.4</td>
<td colspan="3" align="left">Genotype 3a only</td>
<td align="left">
<italic>&#xb6;</italic> 2.0</td>
<td align="left">3.6 (1.0)</td>
</tr>
<tr>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="left">189</td>
<td align="left"/>
<td align="left">5.5</td>
<td colspan="3" align="left">Genotype 3b only</td>
<td align="left">
<italic>&#xb6;</italic> 1.7</td>
<td align="left">1.6 (1.0)</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B45">Murphy et al. (2019b)</xref>
</td>
<td align="left">HCV</td>
<td align="left">
<italic>NS5B</italic>
</td>
<td align="left">119</td>
<td align="left">340&#x2013;850</td>
<td align="left">5.6</td>
<td align="left">Province</td>
<td align="left">Quebec, Canada</td>
<td align="left">2001&#x2013;2017</td>
<td align="left">N/A</td>
<td align="left">0.0251 (1.05)</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B51">Paraschiv et al. (2017)</xref>
</td>
<td align="left">HCV</td>
<td align="left">
<italic>NS5B</italic>
</td>
<td align="left">117</td>
<td align="left">
<inline-formula id="inf13">
<mml:math id="m15">
<mml:mo>&#x223c;</mml:mo>
<mml:mn>300</mml:mn>
</mml:math>
</inline-formula>
</td>
<td align="left">24.6</td>
<td align="left">Country</td>
<td align="left">Romania</td>
<td align="left">2011&#x2013;2014</td>
<td align="left">N/A</td>
<td align="left">1.394 (1.11)</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B84">Ye et al. (2023)</xref>
</td>
<td align="left">HCV</td>
<td align="left">
<italic>NS5B</italic>
</td>
<td align="left">1603/399 &#x2022;</td>
<td align="left">701</td>
<td align="left">27.6</td>
<td align="left">Country</td>
<td align="left">China</td>
<td align="left">1999&#x2013;2017</td>
<td align="left">
<italic>&#xb6;</italic> 0.010</td>
<td align="left">0.359 1)</td>
</tr>
<tr>
<td align="left"/>
<td align="left"/>
<td align="left">
<italic>C/E2</italic>
</td>
<td align="left">865/396 &#x2022;</td>
<td align="left">837</td>
<td align="left">37.3</td>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="left">
<italic>&#xb6;</italic> 0.0325</td>
<td align="left">1.572 (1.98)</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B46">Ng et al. (2022)</xref>
</td>
<td align="left">Rhinovirus</td>
<td align="left">
<italic>VP2/VP4</italic>
</td>
<td align="left">977</td>
<td align="left">388</td>
<td align="left">43.2</td>
<td align="left">City</td>
<td align="left">Kuala Lumpur, Malaysia</td>
<td align="left">2012&#x2013;2014</td>
<td align="left">
<italic>&#xb6;</italic> 0.005</td>
<td align="left">0.523 1)</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-5">
<title>3.5 Large-scale HIV-1 database analyses</title>
<sec id="s3-5-1">
<title>3.5.1 Markedly different thresholds for different subtypes</title>
<p>Following the spirit of the analysis performed by <xref ref-type="bibr" rid="B80">Wertheim et al. (2014)</xref>, we downloaded partial <italic>pol</italic> sequences (between HXB-2 coordinates 2253 and 3200, one sequence per patient) from the Los Alamos HIV-1 Database, split them by annotated subtype and applied AUTO-TUNE to individual subtypes with 1000 or more sequences.</p>
<p>Some (but not all) HIV-1 subtypes often act as strong correlates of regional and temporal distributions of sequences, and are expected to represent epidemics with different sampling rates and transmission dynamics. These differences are reflected in a wide range of mean pairwise distances and inferred AUTO-TUNE thresholds shown in <xref ref-type="table" rid="T5">Table 5</xref>. For example, the relatively young subtype A6, which is the most common subtype in the countries of the former Soviet Union (<xref ref-type="bibr" rid="B1">Abidi et al., 2021</xref>), has a low mean pairwise distance (0.046) and a low AUTO-TUNE threshold (0.0056). In contrast A1D recombinant sequences have high distance and threshold values (0.089 and 0.0323, respectively), because sequences of this &#x201c;subtype&#x201d; represent broadly circulating strains with complex backgrounds, and extensive histories of recombination (<xref ref-type="bibr" rid="B19">Foster et al., 2014</xref>; <xref ref-type="bibr" rid="B85">Yebra et al., 2015</xref>).</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>An application of AUTO-TUNE to subtype stratified HIV-1 pol sequences from the LANL database. Fraction clustered is the proportion of all sequences that are connected to at least one other sequence. Subtypes are sorted by the inferred threshold, lowest first. Other notation is the same as in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left">Subtype</th>
<th rowspan="2" align="left">
<italic>N</italic>
</th>
<th rowspan="2" align="left">
<italic>E</italic> [<italic>D</italic>]</th>
<th colspan="2" align="center">AUTO-TUNE</th>
<th align="left">Fraction</th>
<th align="left">Mean</th>
<th rowspan="2" align="left">
<italic>&#x3c1;</italic>
</th>
</tr>
<tr>
<th align="left">Threshold, %</th>
<th align="left">Score</th>
<th align="left">Clustered, %</th>
<th align="left">Degree</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">CRF55</td>
<td align="left">2237</td>
<td align="left">2.6</td>
<td align="left">0.187</td>
<td align="left">1.20</td>
<td align="left">29.6</td>
<td align="left">1.41</td>
<td align="left">2.41</td>
</tr>
<tr>
<td align="left">CRF07</td>
<td align="left">11682</td>
<td align="left">3.3</td>
<td align="left">0.26</td>
<td align="left">1.00</td>
<td align="left">26.9</td>
<td align="left">3.42</td>
<td align="left">1.87</td>
</tr>
<tr>
<td align="left">CRF63</td>
<td align="left">1649</td>
<td align="left">3.6</td>
<td align="left">0.505</td>
<td align="left">1.01</td>
<td align="left">22.1</td>
<td align="left">4.85</td>
<td align="left">1.8</td>
</tr>
<tr>
<td align="left">01B</td>
<td align="left">2237</td>
<td align="left">7.8</td>
<td align="left">0.518</td>
<td align="left">1.08</td>
<td align="left">22.2</td>
<td align="left">0.97</td>
<td align="left">5.05</td>
</tr>
<tr>
<td align="left">A6</td>
<td align="left">11991</td>
<td align="left">4.6</td>
<td align="left">0.558</td>
<td align="left">1.09</td>
<td align="left">18.6</td>
<td align="left">5.55</td>
<td align="left">1.6</td>
</tr>
<tr>
<td align="left">CRF08</td>
<td align="left">2538</td>
<td align="left">3.9</td>
<td align="left">0.82</td>
<td align="left">1.95</td>
<td align="left">25.6</td>
<td align="left">1.95</td>
<td align="left">2.12</td>
</tr>
<tr>
<td align="left">CRF01</td>
<td align="left">25689</td>
<td align="left">5.1</td>
<td align="left">0.875</td>
<td align="left">1.73</td>
<td align="left">47.0</td>
<td align="left">1.94</td>
<td align="left">5.54</td>
</tr>
<tr>
<td align="left">B</td>
<td align="left">106261</td>
<td align="left">6.4</td>
<td align="left">1.084</td>
<td align="left">2.00</td>
<td align="left">46.4</td>
<td align="left">4.77</td>
<td align="left">1.95</td>
</tr>
<tr>
<td align="left">D</td>
<td align="left">3561</td>
<td align="left">6.6</td>
<td align="left">1.133</td>
<td align="left">1.12</td>
<td align="left">20.8</td>
<td align="left">3.65</td>
<td align="left">0.79</td>
</tr>
<tr>
<td align="left">C</td>
<td align="left">30714</td>
<td align="left">6.7</td>
<td align="left">1.438</td>
<td align="left">2.00</td>
<td align="left">19.3</td>
<td align="left">1.26</td>
<td align="left">2.22</td>
</tr>
<tr>
<td align="left">A1</td>
<td align="left">7154</td>
<td align="left">7.0</td>
<td align="left">1.89</td>
<td align="left">2.00</td>
<td align="left">17.0</td>
<td align="left">5</td>
<td align="left">1.7</td>
</tr>
<tr>
<td align="left">CRF02</td>
<td align="left">7821</td>
<td align="left">6.3</td>
<td align="left">1.97</td>
<td align="left">1.01</td>
<td align="left">34.3</td>
<td align="left">10.44</td>
<td align="left">1.53</td>
</tr>
<tr>
<td align="left">BF1</td>
<td align="left">4825</td>
<td align="left">8.1</td>
<td align="left">2.046</td>
<td align="left">1.03</td>
<td align="left">25.1</td>
<td align="left">2.27</td>
<td align="left">1.95</td>
</tr>
<tr>
<td align="left">G</td>
<td align="left">2162</td>
<td align="left">7.3</td>
<td align="left">2.407</td>
<td align="left">1.03</td>
<td align="left">49.0</td>
<td align="left">9.1</td>
<td align="left">1.66</td>
</tr>
<tr>
<td align="left">F1</td>
<td align="left">3986</td>
<td align="left">7.6</td>
<td align="left">2.941</td>
<td align="left">1.34</td>
<td align="left">50.4</td>
<td align="left">15.03</td>
<td align="left">1.40</td>
</tr>
<tr>
<td align="left">A1D</td>
<td align="left">1284</td>
<td align="left">8.9</td>
<td align="left">3.23</td>
<td align="left">1.70</td>
<td align="left">27.5</td>
<td align="left">1</td>
<td align="left">4.3</td>
</tr>
<tr>
<td align="left">BC</td>
<td align="left">2724</td>
<td align="left">8.0</td>
<td align="left">3.54</td>
<td align="left">1.00</td>
<td align="left">81.4</td>
<td align="left">71.2</td>
<td align="left">1.32</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B80">Wertheim et al. (2014)</xref>
</td>
<td align="left">84527</td>
<td align="left"/>
<td align="left">1.0</td>
<td align="left">N/A</td>
<td align="left">15.6</td>
<td align="left">3.84</td>
<td align="left">1.74</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>There was extensive variability among subtypes in all high-level network statistics, including the mean degree, fractions of nodes that were in the network, and the characteristic exponent <italic>&#x3c1;</italic>, where <italic>&#x3c1;</italic> is inferred from by fitting the degree distribution to various network formation models, and with Prob (degree &#x3d; <italic>k</italic>) &#x223c; 1/<italic>k</italic>
<sup>
<italic>&#x3c1;</italic>
</sup> for large <italic>k</italic>.</p>
<p>For A1, B, C, and CRF08 networks there&#x2019;s very strong support for a single AUTO-TUNE threshold (score <inline-formula id="inf14">
<mml:math id="m16">
<mml:mo>&#x3e;</mml:mo>
<mml:mn>1.9</mml:mn>
</mml:math>
</inline-formula>), while for many other subtypes there is extreme ambiguity in which threshold to choose (score <inline-formula id="inf15">
<mml:math id="m17">
<mml:mo>&#x3c;</mml:mo>
<mml:mn>1.1</mml:mn>
</mml:math>
</inline-formula>). We suggest that networks where AUTO-TUNE fails to find a single threshold may comprise heterogeneous data which require multiple thresholds to resolve.</p>
</sec>
<sec id="s3-5-2">
<title>3.5.2 Congruence of networks inferred from different genes</title>
<p>Very few published studies of HIV-1 transmission networks use genes other than <italic>pol</italic>, and nearly all of the extrinsically motivated thresholds have been derived for this gene, the utility of other genes and the appropriate <italic>D</italic> values for them are unclear. Because of different rates of evolution in HIV-1 genes and, possibly, subtypes (<xref ref-type="bibr" rid="B54">Penn et al., 2008</xref>), one would expect <italic>D</italic> to be different for different subtypes and genes. As a simple exercise, we downloaded full-length HIV-1 genomes from the LANL database, stratified them by subtype, and conducted AUTO-TUNE inference using four genomic segments: protease &#x2b; reverse transcriptase, integrase, matrix (gag), and the less variable gp41 segment of the envelope gene.</p>
<p>Only three subtypes had <inline-formula id="inf16">
<mml:math id="m18">
<mml:mo>&#x2265;</mml:mo>
<mml:mn>500</mml:mn>
</mml:math>
</inline-formula> full-length sequences in the LANL HIV database (<xref ref-type="table" rid="T6">Table 6</xref>): B, C, and CRF01. As expected, the inferred thresholds differed by gene and subtype, with lower thresholds inferred for more slowly evolving segments (PR &#x2b; RT and INT), and similar numbers of clusters found in the resulting subtype-level networks. For all three subtypes, the level of agreement between the four networks on whether or not nodes were clustered or not (present/absent from the network), measured by Krippendorff&#x2019;s <italic>&#x3b1;</italic> (<xref ref-type="bibr" rid="B29">Hayes and Krippendorff, 2007</xref>), were substantially higher than expected by chance (<italic>&#x3b1;</italic> &#x3d; 0). All four networks also had between a quarter and a half of all the clusters in perfect agreement.</p>
<table-wrap id="T6" position="float">
<label>TABLE 6</label>
<caption>
<p>Distance thresholds and key network properties using four different HIV-1 genomic regions, stratified by subtype (minimum 500 sequences).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left">Subtype</th>
<th rowspan="2" align="left">N</th>
<th colspan="4" align="center">AUTO-TUNE <italic>D</italic>, <italic>subs</italic>/<italic>site</italic>
</th>
<th colspan="4" align="center">Number of clusters</th>
<th align="left">Full agreement</th>
<th rowspan="2" align="left">Krippendorff <italic>&#x3b1;</italic>
</th>
</tr>
<tr>
<th align="left">pr &#x2b; rt</th>
<th align="left">Integrase</th>
<th align="left">gag</th>
<th align="left">gp41</th>
<th align="left">pr &#x2b; rt</th>
<th align="left">Integrase</th>
<th align="left">gag</th>
<th align="left">gp41</th>
<th align="left">Clusters</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">B</td>
<td align="left">1843</td>
<td align="left">0.02081</td>
<td align="left">2.0005</td>
<td align="left">3.137</td>
<td align="left">5.095</td>
<td align="left">115</td>
<td align="left">128</td>
<td align="left">119</td>
<td align="left">144</td>
<td align="left">64</td>
<td align="left">0.723</td>
</tr>
<tr>
<td align="left">C</td>
<td align="left">877</td>
<td align="left">0.03266</td>
<td align="left">0.02</td>
<td align="left">4.754</td>
<td align="left">5.325</td>
<td align="left">44</td>
<td align="left">35</td>
<td align="left">47</td>
<td align="left">46</td>
<td align="left">21</td>
<td align="left">0.588</td>
</tr>
<tr>
<td align="left">CRF01/AE</td>
<td align="left">624</td>
<td align="left">0.01635</td>
<td align="left">0.818</td>
<td align="left">2.285</td>
<td align="left">2.037</td>
<td align="left">40</td>
<td align="left">30</td>
<td align="left">40</td>
<td align="left">41</td>
<td align="left">12</td>
<td align="left">0.610</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s3-6">
<title>3.6 Evaluating inferred networks using homophily</title>
<p>Non-random mixing and attribute-based homophily are intrinsic characteristics of human contact networks and can be expected to be present in transmission networks, particularly in the context of HIV transmission. People frequently engage in relationships with those who share similar attributes or behaviors, such as risk factors (e.g., PWID, MSM). Recent evidence suggests that race/ethnicity is also a strong predictor of homophily in HIV networks (<xref ref-type="bibr" rid="B58">Ragonnet-Cronin et al., 2021</xref>). The effect of these nonrandom mixing patterns on the genetic diversity of HIV-1 has not only been extensively explored via modeling and simulations (<xref ref-type="bibr" rid="B25">Goodreau, 2006</xref>), but the structure of sexual contact networks has been found to directly influence pathogen phylogenies (<xref ref-type="bibr" rid="B64">Robinson et al., 2013</xref>). Phylogenetic analysis of HIV type 1 sequences has revealed distinct grouping patterns based on risk behaviors (<xref ref-type="bibr" rid="B30">Holmes et al., 1995</xref>). The expectation of homophily is so strong, that its disruption, e.g., the presence of a self-reported heterosexual risk group individual in a cluster exclusively composed of MSMs has been used as a marker of undisclosed/incomplete risk factor reporting (<xref ref-type="bibr" rid="B61">Ragonnet-Cronin et al., 2018</xref>). Therefore, when subject-level attributes are available, homophily is an expected and desired feature of the network.</p>
<p>To assess the performance of an AUTO-TUNEd optimized threshold using degree-weighted homophily, we first evaluated a CRF07_BC network with national survey data from China (<xref ref-type="bibr" rid="B21">Ge et al., 2021</xref>). Each of the 8178 pol sequences was annotated with a transmission risk factor: heterosexual contact (Hetero), people who use injection drugs (PWID), or men who have sex with men (MSM), among other attributes.</p>
<p>When we analyze the dataset with AUTO-TUNE, local maxima of AUTO-TUNE scores were achieved with 0.0076 sub/site and 0.0019 sub/site thresholds, at scores 1.137 and 1.030, respectively. Notably, the DWH scores for PWID exhibited a significant surge at these thresholds, indicating a robust pattern of increased PWID homophily even when relatively low scoring. The close proximity of AUTO-TUNE scores and the consistent rise in PWID homophily at 0.0076 and 0.0019 thresholds suggest comparable performance at these thresholds compared to the default 0.015 threshold, suggesting that these thresholds might be more effective in representing homophilic relationships in this network. At each threshold&#x2014;0.015, 0.0076, and 0.0019&#x2014;all DWH scores for the risk groups (MSM, Hetero, and PWID) lie outside their respective panmictic ranges. This consistently indicates non-random mixing and attribute-based homophily across the network. Detailed results are in <xref ref-type="table" rid="T7">Tables 7</xref>, <xref ref-type="table" rid="T8">8</xref>.</p>
<table-wrap id="T7" position="float">
<label>TABLE 7</label>
<caption>
<p>CRF07_BC nodes count at different thresholds.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Threshold subs/site</th>
<th align="left">AUTO-TUNE score</th>
<th align="left">Nodes</th>
<th align="left">PWID</th>
<th align="left">MSM</th>
<th align="left">Hetero</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">0.015</td>
<td align="left">0.029</td>
<td align="left">5923</td>
<td align="left">559</td>
<td align="left">3371</td>
<td align="left">1993</td>
</tr>
<tr>
<td align="left">0.0076</td>
<td align="left">1.1369</td>
<td align="left">3537</td>
<td align="left">236</td>
<td align="left">2271</td>
<td align="left">1030</td>
</tr>
<tr>
<td align="left">0.0019</td>
<td align="left">1.0303</td>
<td align="left">1654</td>
<td align="left">151</td>
<td align="left">1075</td>
<td align="left">428</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T8" position="float">
<label>TABLE 8</label>
<caption>
<p>Panmictic ranges for CRF07_BC DWH at different thresholds.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Threshold subs/site</th>
<th align="left">Risk group</th>
<th align="left">DWH (panmictic range)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="3" align="left">0.015</td>
<td align="left">MSM</td>
<td align="left">0.211 (&#x2212;0.213, &#x2212;0.085)</td>
</tr>
<tr>
<td align="left">Hetero</td>
<td align="left">0.133 (&#x2212;0.190, &#x2212;0.087)</td>
</tr>
<tr>
<td align="left">PWID</td>
<td align="left">0.168 (&#x2212;0.091, 0.002)</td>
</tr>
<tr>
<td rowspan="3" align="left">0.0076</td>
<td align="left">MSM</td>
<td align="left">0.237 (&#x2212;0.240, &#x2212;0.120)</td>
</tr>
<tr>
<td align="left">Hetero</td>
<td align="left">0.185 (&#x2212;0.211, &#x2212;0.100)</td>
</tr>
<tr>
<td align="left">PWID</td>
<td align="left">0.401 (&#x2212;0.081, &#x2212;0.005)</td>
</tr>
<tr>
<td rowspan="3" align="left">0.0019</td>
<td align="left">MSM</td>
<td align="left">0.292 (&#x2212;0.280, &#x2212;0.146)</td>
</tr>
<tr>
<td align="left">Hetero</td>
<td align="left">0.250 (&#x2212;0.256, &#x2212;0.093)</td>
</tr>
<tr>
<td align="left">PWID</td>
<td align="left">0.445 (&#x2212;0.129, &#x2212;0.012)</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-7">
<title>3.7 Comparison with clustuneR</title>
<p>We benchmarked AUTO-TUNE <italic>versus</italic> <monospace>clustuneR</monospace> (<xref ref-type="bibr" rid="B10">Chato et al., 2020</xref>), which employs the recency of sample collection or diagnosis as individual-level weights in a predictive model to estimate the growth of HIV clusters. The thresholds deemed optimal by <monospace>clustuneR</monospace> were found by a grid-search for the minimum GAIC (generalized Akaike Information Criterion) across candidate distances between 0 and 0.04 in steps of 8 &#xd7; 10<sup>&#x2212;4</sup>. GAIC is the difference between a null model that is only influenced by cluster size, and a weighted model that includes individual-level attributes among known cases in the cluster. Using the minimum GAIC metric, it was found that 0.016 (&#xb1;0.5 &#xd7; 10<sup>&#x2212;4</sup>) was the optimal threshold for Tennessee and Seattle, and 0.0104 for Northern Alberta.</p>
<p>In contrast, AUTO-TUNE does not incorporate any attribute data in its scoring heuristic. Instead, it relies on clustering metrics constructed purely from pairwise distances between sequences. Using nearly same datasets analyzed by <monospace>clustuneR</monospace> (<xref ref-type="bibr" rid="B10">Chato et al., 2020</xref>), AUTO-TUNE found the thresholds with the highest scores to be 0.01872 for Middle Tennessee, 0.01538 for Seattle, and 0.01201 for Northern Alberta <xref ref-type="table" rid="T9">Table 9</xref>. We use the adjective &#x201c;nearly&#x201d; because we were not able to exactly match the number of sequences analyzed in <xref ref-type="bibr" rid="B10">Chato et al. (2020)</xref> by obtaining the referenced GenBank accession number and our best-effort intepretation of the filtering steps.</p>
<table-wrap id="T9" position="float">
<label>TABLE 9</label>
<caption>
<p>ClustuneR Comparison.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Dataset</th>
<th colspan="3" align="center">AUTO-TUNE</th>
<th colspan="2" align="center">clustuneR</th>
</tr>
<tr>
<th align="center">Threshold subs/site</th>
<th align="center">Avg. Homophily</th>
<th align="center">Max score</th>
<th align="center">Threshold</th>
<th align="center">Avg. Homophily</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Seattle</td>
<td align="center">0.01354</td>
<td align="center">0.0348</td>
<td align="center">1.53325</td>
<td align="center">0.0160</td>
<td align="center">0.0259</td>
</tr>
<tr>
<td align="center">Tennessee</td>
<td align="center">0.01431</td>
<td align="center">0.0147</td>
<td align="center">1.25807</td>
<td align="center">0.0160</td>
<td align="center">0.0079</td>
</tr>
<tr>
<td align="center">Canada</td>
<td align="center">0.01099</td>
<td align="center">&#x2212;0.0448</td>
<td align="center">1.01678</td>
<td align="center">0.0104</td>
<td align="center">&#x2212;0.0536</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Both methods agree that there is a qualitative relationship of Northern Alberta &#x3c; Seattle &#x223c;Tennessee for distance thresholds. AUTO-TUNE thresholds, while not optimal in the GAIC sense all yield improvements over the null model, hence they are qualitatively similar to <monospace>clustuneR</monospace> (<xref ref-type="fig" rid="F2">Figure 2</xref> in <xref ref-type="bibr" rid="B10">Chato et al. (2020)</xref>). AUTO-TUNE is notably faster in computation than clustuneR due to the fact that AUTO-TUNE only clusters based on pairwise distances rather than inferring a maximum-likelihood phylogeny. For example, the entire pipeline for the Seattle dataset took less than 16&#xa0;s on an Apple M1 Max. Alternatively, the tree inference step alone with <monospace>clustuneR</monospace> takes several hours to complete.</p>
<p>Because the methods optimize very different objectives and <monospace>clustuneR</monospace> makes use of additional data, broad agreement between the inferred thresholds is encouraging.</p>
</sec>
<sec id="s3-8">
<title>3.8 The effect of subsampling on optimal thresholds and AUTO-TUNE scores</title>
<p>To address the challenges of applying network inference algorithms to incompletely sampled datasets, this study includes a focused evaluation of AUTO-TUNE&#x2019;s performance across varying data densities. Given logistical limitations, obtaining a fully sampled HIV transmission network is often infeasible. Therefore, we label a dataset as &#x2018;full&#x2019; to serve as a closest approximation of a fully sampled network. Using the selected dataset as a benchmark, we assess AUTO-TUNE&#x2019;s adaptability and robustness when applied to sparser datasets, a prevalent issue in real-world settings. In this analysis there is no expectation that any specific fraction of undelying infections was sampled, but simply that the complete dataset acts as the upper bound for inference. This was done in <xref ref-type="bibr" rid="B13">Dasgupta et al. (2019)</xref>, for example,.</p>
<p>Since the <xref ref-type="bibr" rid="B63">Rhee et al. (2019)</xref> dataset exhibited a clear optimal peak, we used the dataset for analysis, and randomly sampled 10 times from the entire dataset at 25%, 50%, and 75% each. The original full dataset confidently determined 0.01699 (AUTO-TUNE score 1.9998).</p>
<p>Sampling at 25% yielded a mean top threshold of 0.021509, median at 0.019765, and standard deviation of 0.004388 (<xref ref-type="fig" rid="F3">Figure 3</xref>). 50% yielded 0.018581 and 0.01871 mean and median, respectively with a standard deviation of 0.001629. Finally, 75% calculated mean is approximately 0.017403, with a median of approximately 0.01699. The standard deviation was 0.000924.</p>
<p>As the dataset becomes sparser due to subsampling, the algorithm tends to select higher distance thresholds. This phenomenon can be understood by considering the effect of reduced sampling density on the network topology. Sparse datasets naturally result in less interconnected clusters. To capture a comparable level of network connectivity as in denser datasets, higher distance thresholds are necessary. This is evidenced by the observed mean thresholds: 0.021509&#xa0;at 25%, 0.018581&#xa0;at 50%, and 0.017403 at 75%. The standard deviations also narrow as the sampling density increases, corroborating the increased precision of the threshold selection in denser datasets.</p>
<p>As the proportion increased from 25% to 50% and 75%, observable shifts were also noted in the mean, median, and standard deviation of the AUTO-TUNE scores. At 25%, the mean and median scores were 1.5585 and 1.5014 respectively, with a standard deviation of 0.3568. At 50%, both mean and median scores significantly increased to 1.8171 and 1.9191 respectively, and the standard deviation dropped to 0.2482. Upon reaching an AUTO-TUNE of 75%, the mean and median scores rose further to 1.9870 and 1.9997 respectively, while the standard deviation shrank substantially to 0.0364, indicating higher consistency in scores.</p>
<p>Next to determine how well subsampled datasets aligned with the full dataset, we used two primary outcomes to gauge this concordance: the proportion of nodes that remained clustered after subsampling and the proportion of singletons from the original network that clustered in the subsampled networks.</p>
<p>We observed a consistent increase in the proportion of nodes that remained clustered from the 0.015 sub/site threshold to the AUTO-TUNE threshold for each respective subsampling proportion, with 25% subsampling being the most profound difference rising from a roughly 80%&#x2013;86% interquartile range (IQR) for 0.015 threshold to a 90% 96% IQR for AUTO-TUNE, which indicates that the AUTO-TUNE thresholds retain a higher degree of stability in the network&#x2019;s structure across sampling density (Please see <xref ref-type="fig" rid="F6">Figure 6</xref>, Panel A).</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Figure A and B present the effects of subsampling on network structure using different thresholds. Figure A illustrates the proportion of nodes subsampled that remained clustered in both the original and the subsampled networks, with an observable increase in nodes captured as the threshold transitions from 1.5.</p>
</caption>
<graphic xlink:href="fbinf-04-1400003-g006.tif"/>
</fig>
<p>Since the thresholds inferred by AUTO-TUNE for the subsampled networks were larger than the &#x201c;fully&#x201d; sampled network, we also measured the impact of thresholding on the network&#x2019;s nodes that were originally singletons. Across all variations in subsampling rates, the proportion of sampled singletons that clustered all maintained low IQRs (See <xref ref-type="fig" rid="F6">Figure 6</xref>, Panel B). This implies that while AUTO-TUNE is effective in maintaining the core structure of the network, it does not significantly alter the clustering of nodes that were singletons in the full dataset.</p>
<p>As the sample proportion increased, an upward trend was noted in average AUTO-TUNE scores. Additionally, the standard deviation reduced significantly when increasing sample proportion. This implies that as sampling becomes denser, AUTO-TUNE will become more confident in determining the optimal threshold for a particular dataset.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<p>AUTO-TUNE addresses the challenge of selecting an appropriate genetic distance threshold to construct HIV transmission networks by implementing a heuristic scoring system. This system is predicated on two key features of networks generated by candidate genetic distance thresholds: a high number of clusters and the absence of a giant component. Few small clusters indicate an excessively low threshold, while a giant cluster comprising numerous sequences signals an overly high threshold. The efficacy of AUTO-TUNE is evidenced by its ability to select thresholds that yield higher quality clustering, as demonstrated by improved Degree-Weighted Homophily (DWH) scores across various datasets, epidemic contexts, and risk groups. Furthermore, AUTO-TUNE thresholds not only matched but often outperformed those manually selected in prior studies, thus underlining the benefits of a more systematic, automated, and data-responsive approach.</p>
<p>For example, the results of our study suggest that AUTO-TUNE, which relies solely on clustering metrics from pairwise distances, could be an effective alternative to other distance-based methods, such as <monospace>clustuneR</monospace> while less time-consuming and possessing a gentle learning curve, which makes it easy to use by personnel not specialized in bioinformatics and computer science. Furthermore, the simplicity of the method without compromising results represents an advantage over phylogenetic methods where, in addition to the calculation of genetic distances, it must also determine a support/distance threshold where a rationale for the selection of these thresholds is rarely provided (<xref ref-type="bibr" rid="B34">Junqueira et al., 2019</xref>).</p>
<p>AUTO-TUNE generated thresholds for all three examined datasets (Middle Tennessee, Seattle, and Northern Alberta) that outperformed <monospace>clustuneR</monospace> using DWH on 3-year collection date windows across all three datasets. This indicates that even without incorporating attribute data, AUTO-TUNE&#x2019;s scoring heuristic could provide reliable thresholds for HIV clusters. However, for the determination of the optimal genetic distance threshold, time-related and context-specific factors might need to be considered if there is no significant score for any one candidate threshold, especially if there are multiple peaks. For example, during HIV outbreaks in injection drug users (that usually occur over several months), it may be more appropriate to use the shorter genetic distance threshold (<xref ref-type="bibr" rid="B56">Peters et al., 2016</xref>; <xref ref-type="bibr" rid="B7">Campbell et al., 2017</xref>) between multiple high-scoring thresholds. On the contrary, larger and more extended epidemics over time exhibit a tendency toward larger genetic distance thresholds in order to capture transmission than younger epidemics and less densely sampled epidemic investigations (<xref ref-type="bibr" rid="B38">Leung et al., 2019</xref>; <xref ref-type="bibr" rid="B14">Di Giallonardo et al., 2021</xref>; <xref ref-type="bibr" rid="B53">Patil et al., 2022</xref>).</p>
<p>Our review of publications citing HIV-TRACE revealed the largely qualitative determination of distance thresholds. This approach may result in less accurate or suboptimal thresholds due to a lack of systematic analysis. In contrast, AUTO-TUNE offers a more systematic and granular approach to threshold selection, with our findings demonstrating that even minor adjustments to the distance can drastically change the score. Therefore, using AUTO-TUNE could potentially improve the quality of HIV clustering and transmission network studies.</p>
<p>The Degree-Weighted Homophily (DWH) evaluation showed that AUTO-TUNE could improve network quality based on specific attributes, such as risk factor, which is an important part of HIV studies and informing prevention measures (<xref ref-type="bibr" rid="B57">Potterat et al., 2002</xref>; <xref ref-type="bibr" rid="B20">Fujimoto et al., 2021</xref>). For example, the use of AUTO-TUNE resulted in an increased DWH among the MSM, Hetero, and PWID groups when analyzing a CRF07_BC network. Additionally, the results from the Rhee et al. dataset also demonstrated AUTO-TUNE&#x2019;s ability to improve DWH geographically, enhancing the network&#x2019;s ability to accurately reflect transmission dynamics. However, in contexts with overlapping risk factors, the interpretation of these improvements requires caution. The complexities of risk group interactions mean that applying AUTO-TUNE&#x2019;s thresholds should be tailored to the specific epidemiological setting to ensure accurate modeling of HIV transmission networks. More broadly, the ultimate impact of AUTO-TUNE on network quality and interpretation will be case specific, strongly dependant on how clusters are ultimately used in the study, and what types of data in addition to sequences alone are available.</p>
<p>Our analysis of AUTO-TUNE&#x2019;s performance on subsamples of a dataset revealed its sensitivity to sample size. The results indicated a correlation between increased sample size and higher average AUTO-TUNE scores, as well as lower score variability. This suggests that denser sampling could enhance AUTO-TUNE&#x2019;s ability to determine the optimal threshold for a dataset. Further studies might be needed to establish the minimum sample size required for reliable threshold determination.</p>
<sec id="s4-1">
<title>4.1 When a score is below 1.9</title>
<p>In some cases, multiple scores at different thresholds could suggest the presence of inherently different scales in the network. For instance, if a network combines both global and local transmission patterns, AUTO-TUNE may produce more than one high score, reflecting these different scales. This was observed in a study on HIV-1 CRF07_BC transmission networks in China, where two distinct clusters, 07BC_N and 07BC_O, showed different transmission routes and geographic concentrations (<xref ref-type="bibr" rid="B15">Ding et al., 2022</xref>). Such network complexities could mean that different thresholds might offer more accurate insights into subpopulations or transmission dynamics.</p>
<p>The use of AUTO-TUNE, while offering a method for automated threshold selection, may not always provide a single, decisive score that unambiguously determines the optimal threshold. In certain situations, such as datasets with lower sampling densities or those reflecting heterogenous dynamics within an epidemic, several candidate thresholds may yield similar AUTO-TUNE scores, making it difficult to single out one as the clear-cut &#x2018;optimal&#x2019; threshold. In these scenarios, the process of threshold selection becomes more nuanced and requires a deeper analysis. The plot of AUTO-TUNE scores across candidate thresholds can serve as a valuable tool in these cases. For instance, researchers could identify a range of thresholds that all produce similar scores, suggesting that the specific choice of threshold within this range may not significantly impact the resulting network. Moreover, combining AUTO-TUNE with the DWH measure can enhance the interpretation of such plots. By considering how assortativity changes across the range of candidates, researchers can make more informed decisions about the appropriate choice. If there is a certain threshold at which the DWH measure noticeably changes for an attribute of interest, this could suggest a meaningful shift in the network structure that would be worth considering when selecting a threshold. The symbiotic approach of combining AUTO-TUNE scores, DWH measure, and visual analysis of score plots provides a more nuanced method for threshold selection when no clear optimal threshold emerges from the AUTO-TUNE scores alone.</p>
<p>The AUTO-TUNE methodology has several limitations. First, even though it provides the advantage of operating without the need for metadata, the size and the subgenomic region analyzed may affect the accuracy of transmission inference (<xref ref-type="bibr" rid="B34">Junqueira et al., 2019</xref>). Second, our analysis of AUTO-TUNE&#x2019;s performance on subsamples of a dataset revealed its sensitivity to sample size, as the performance of the method can be affected by sampling density, improving the reliability of the test as the sampling density increases. However, our results were consistent with previous studies, which have suggested an optimal sampling density of 50&#x2013;70% for HIV-1 cluster analysis (<xref ref-type="bibr" rid="B47">Novitsky et al., 2014</xref>), and the significant drop-off in power to detect clusters using a fixed distance threshold as the sampling fraction was reduced (<xref ref-type="bibr" rid="B13">Dasgupta et al., 2019</xref>). Third, even when it provides an insight of the optimal threshold to analyze a network, the supplied information might still need validation by experts, especially when no clear threshold is identified. In this case, it has been recommended to combine genetic data with clinical and sociodemographic information for a better characterization of the network structure. Finally, the performance of the method needs to be assessed in pathogens different from HIV, leading to opportunities for future research.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>AUTO-TUNE operates solely utilizing genetic sequence data to ascertain a decisive threshold. It employs a scoring heuristic, which is based on the number of clusters produced by a pairwise distance threshold and the ratio of the largest cluster to the second largest across a range of possible thresholds using sliding windows.</p>
<p>A key advantage of this approach is its autonomy from supplementary data. When a patient receives an HIV diagnosis, data collection protocols can greatly vary, and additional data are not always available or consistent. However, by leveraging only genetic sequence data, AUTO-TUNE eliminates the need for such information in some cases, and at minimum serves as a preliminary assessment of candidate thresholds.</p>
<p>Consequently, AUTO-TUNE&#x2019;s performance is consistently controlled, irrespective of the fluctuations seen in data collection protocols after an HIV diagnosis. This level of adaptability demonstrates its suitability for integration into various contexts related to HIV, and possibly other viral cluster detection and response protocols. This versatility underscores the strong methodological foundation of AUTO-TUNE and its potential utility.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://github.com/veg/autotune-paper">https://github.com/veg/autotune-paper</ext-link>.</p>
</sec>
<sec id="s7">
<title>Author contributions</title>
<p>SW: Writing&#x2013;review and editing, Writing&#x2013;original draft, Visualization, Supervision, Software, Project administration, Methodology, Formal Analysis, Data curation, Conceptualization. VD: Writing&#x2013;review and editing, Writing&#x2013;original draft, Data curation. DJ: Writing&#x2013;review and editing, Software. HV: Writing&#x2013;review and editing, Writing&#x2013;original draft, Visualization. S&#x00c1;-R: Writing&#x2013;review and editing. AL: Conceptualization, Writing&#x2013;review and editing. JW: Writing&#x2013;review and editing, Conceptualization. SK: Writing&#x2013;review and editing, Writing&#x2013;original draft, Visualization, Validation, Supervision, Software, Resources, Project administration, Methodology, Funding acquisition, Data curation, Conceptualization.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research, authorship, and/or publication of this article. SK and SW were supported in part by grant funding from the NIH, grants AI134384, AI140970, GM144468, GM110749, GM151683. JOW was supported in part by AI135992.</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abidi</surname>
<given-names>S. H.</given-names>
</name>
<name>
<surname>Aibekova</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Davlidova</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Amangeldiyeva</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Foley</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Ali</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Origin and evolution of HIV-1 subtype A6</article-title>. <source>PLoS One</source> <volume>16</volume>, <fpage>e0260604</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0260604</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bartlett</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Wertheim</surname>
<given-names>J. O.</given-names>
</name>
<name>
<surname>Bull</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Matthews</surname>
<given-names>G. V.</given-names>
</name>
<name>
<surname>Lamoury</surname>
<given-names>F. M.</given-names>
</name>
<name>
<surname>Scheffler</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>A molecular transmission network of recent hepatitis c infection in people with and without hiv: implications for targeted treatment strategies</article-title>. <source>J. viral Hepat.</source> <volume>24</volume>, <fpage>404</fpage>&#x2013;<lpage>411</lpage>. <pub-id pub-id-type="doi">10.1111/jvh.12652</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bbosa</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Ssemwanga</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Kaleebu</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Short communication: choosing the right program for the identification of HIV-1 transmission networks from nucleotide sequences sampled from different populations</article-title>. <source>AIDS Res. Hum. retroviruses</source> <volume>36</volume>, <fpage>948</fpage>&#x2013;<lpage>951</lpage>. <pub-id pub-id-type="doi">10.1089/AID.2020.0033</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Billings</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Kijak</surname>
<given-names>G. H.</given-names>
</name>
<name>
<surname>Sanders-Buell</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Ndembi</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>O&#x2019;Sullivan</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Adebajo</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>New subtype b containing hiv-1 circulating recombinant of sub-saharan africa origin in nigerian men who have sex with men</article-title>. <source>J. Acquir Immune Defic. Syndr.</source> <volume>81</volume>, <fpage>578</fpage>&#x2013;<lpage>584</lpage>. <pub-id pub-id-type="doi">10.1097/QAI.0000000000002076</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Boender</surname>
<given-names>T. S.</given-names>
</name>
<name>
<surname>Smit</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Sighem</surname>
<given-names>A. v.</given-names>
</name>
<name>
<surname>Bezemer</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ester</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Zaheri</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>AIDS therapy evaluation in The Netherlands (ATHENA) national observational HIV cohort: cohort profile</article-title>. <source>BMJ Open</source> <volume>8</volume>, <fpage>e022516</fpage>. <pub-id pub-id-type="doi">10.1136/bmjopen-2018-022516</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brenner</surname>
<given-names>B. G.</given-names>
</name>
<name>
<surname>Ibanescu</surname>
<given-names>R.-I.</given-names>
</name>
<name>
<surname>Osman</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Cuadra-Foy</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Oliveira</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chaillon</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>The role of phylogenetics in unravelling patterns of HIV transmission towards epidemic control: the quebec experience (2002-2020)</article-title>. <source>Viruses</source> <volume>13</volume>, <fpage>1643</fpage>. <pub-id pub-id-type="doi">10.3390/v13081643</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Campbell</surname>
<given-names>E. M.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Shankar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hanson</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Masciotra</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Detailed transmission network analysis of a large opiate-driven outbreak of HIV infection in the United States</article-title>. <source>J. Infect. Dis.</source> <volume>216</volume>, <fpage>1053</fpage>&#x2013;<lpage>1062</lpage>. <pub-id pub-id-type="doi">10.1093/infdis/jix307</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Campigotto</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chris</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Orkin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lau</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Marshall</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bitnun</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Utility of SARS-CoV-2 genomic sequencing for understanding transmission and school outbreaks</article-title>. <source>Pediatr. Infect. Dis. J.</source> <volume>42</volume>, <fpage>324</fpage>&#x2013;<lpage>331</lpage>. <pub-id pub-id-type="doi">10.1097/INF.0000000000003834</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chan</surname>
<given-names>C. P.</given-names>
</name>
<name>
<surname>Uemura</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Kwan</surname>
<given-names>T. H.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>N. S.</given-names>
</name>
<name>
<surname>Oka</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>D. P. C.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Review on the molecular epidemiology of sexually acquired hepatitis c virus infection in the asia-pacific region</article-title>. <source>J. Int. AIDS Soc.</source> <volume>23</volume>, <fpage>e25618</fpage>. <pub-id pub-id-type="doi">10.1002/jia2.25618</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chato</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Kalish</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>Poon</surname>
<given-names>A. F. Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Public health in genetic spaces: a statistical framework to optimize cluster-based outbreak detection</article-title>. <source>Virus Evol.</source> <volume>6</volume>, <fpage>veaa011</fpage>. <pub-id pub-id-type="doi">10.1093/ve/veaa011</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lan</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ruan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>McNeil</surname>
<given-names>E. B.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Inferring potential non-disclosed men who have sex with men among self-reported heterosexual men with hiv in southwest China: a genetic network study</article-title>. <source>PLoS One</source> <volume>18</volume>, <fpage>e0283031</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0283031</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dalai</surname>
<given-names>S. C.</given-names>
</name>
<name>
<surname>Junqueira</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Wilkinson</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Mehra</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Kosakovsky Pond</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Levy</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Combining phylogenetic and network approaches to identify HIV-1 transmission links in san mateo county, California</article-title>. <source>Front. Microbiol.</source> <volume>9</volume>, <fpage>2799</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2018.02799</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dasgupta</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>France</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Brandt</surname>
<given-names>M.-G.</given-names>
</name>
<name>
<surname>Reuer</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Panneer</surname>
<given-names>N.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Estimating effects of HIV sequencing data completeness on transmission network patterns and detection of growing HIV transmission clusters</article-title>. <source>AIDS Res. Hum. Retroviruses</source> <volume>35</volume>, <fpage>368</fpage>&#x2013;<lpage>375</lpage>. <pub-id pub-id-type="doi">10.1089/AID.2018.0181</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Di Giallonardo</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Pinto</surname>
<given-names>A. N.</given-names>
</name>
<name>
<surname>Keen</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Shaik</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Carrera</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Salem</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Subtype-specific differences in transmission cluster dynamics of HIV-1 B and CRF01_ae in New South Wales, Australia</article-title>. <source>J. Int. AIDS Soc.</source> <volume>24</volume>, <fpage>e25655</fpage>. <pub-id pub-id-type="doi">10.1002/jia2.25655</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ding</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chaillon</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhong</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Characterizing genetic transmission networks among newly diagnosed HIV-1 infected individuals in eastern China: 2012&#x2013;2016</article-title>. <source>PLOS ONE</source> <volume>17</volume>, <fpage>e0269973</fpage>. <comment>Publisher: Public Library of Science</comment>. <pub-id pub-id-type="doi">10.1371/journal.pone.0269973</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dunn</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Pillay</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>UK HIV drug resistance database: background and recent outputs</article-title>. <source>J. HIV Ther.</source> <volume>12</volume>, <fpage>97</fpage>&#x2013;<lpage>98</lpage>.</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Erly</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Naismith</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kerani</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Buskin</surname>
<given-names>S. E.</given-names>
</name>
<name>
<surname>Reuer</surname>
<given-names>J. R.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Predictive value of time-space clusters for HIV transmission in Washington state, 2017-2019</article-title>. <source>J. Acquir. Immune Defic. Syndromes</source> <volume>87</volume>, <fpage>912</fpage>&#x2013;<lpage>917</lpage>. <pub-id pub-id-type="doi">10.1097/QAI.0000000000002675</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fabeni</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Santoro</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Lorenzini</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Rusconi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gianotti</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Costantini</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Evaluation of hiv transmission clusters among natives and foreigners living in Italy</article-title>. <source>Viruses</source> <volume>12</volume>, <fpage>791</fpage>. <pub-id pub-id-type="doi">10.3390/v12080791</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Foster</surname>
<given-names>G. M.</given-names>
</name>
<name>
<surname>Ambrose</surname>
<given-names>J. C.</given-names>
</name>
<name>
<surname>Hu&#xe9;</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Delpech</surname>
<given-names>V. C.</given-names>
</name>
<name>
<surname>Fearnhill</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Abecasis</surname>
<given-names>A. B.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Novel HIV-1 recombinants spreading across multiple risk groups in the United Kingdom: the identification and phylogeography of circulating recombinant form (crf) 50_a1d</article-title>. <source>PLoS One</source> <volume>9</volume>, <fpage>e83337</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0083337</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fujimoto</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Bahl</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wertheim</surname>
<given-names>J. O.</given-names>
</name>
<name>
<surname>Del Vecchio</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Hicks</surname>
<given-names>J. T.</given-names>
</name>
<name>
<surname>Damodaran</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Methodological synthesis of Bayesian phylodynamics, HIV-TRACE, and GEE: HIV-1 transmission epidemiology in a racially/ethnically diverse Southern U.S. context</article-title>. <source>Sci. Rep.</source> <volume>11</volume>, <fpage>3325</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-021-82673-8</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ge</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Rashid</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zaongo</surname>
<given-names>S. D.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>HIV-1 CRF07_BC transmission dynamics in China: two decades of national molecular surveillance</article-title>. <source>Emerg. Microbes Infect.</source> <volume>10</volume>, <fpage>1919</fpage>&#x2013;<lpage>1930</lpage>. <pub-id pub-id-type="doi">10.1080/22221751.2021.1978822</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Golub</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Jackson</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Network structure and the speed of learning measuring homophily based on its consequences</article-title>. <source>Ann. Econ. Statistics</source>, <fpage>33</fpage>&#x2013;<lpage>48</lpage>. <pub-id pub-id-type="doi">10.2307/23646571</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goodreau</surname>
<given-names>S. M.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Assessing the effects of human mixing patterns on human immunodeficiency virus-1 interhost phylogenetics through social network simulation</article-title>. <source>Genetics</source> <volume>172</volume>, <fpage>2033</fpage>&#x2013;<lpage>2045</lpage>. <pub-id pub-id-type="doi">10.1534/genetics.103.024612</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gore</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Schueler</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Ramani</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Uvin</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Phillips</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>McNulty</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>HIV response interventions that integrate HIV molecular cluster and social network analysis: a systematic review</article-title>. <source>AIDS Behav.</source> <volume>26</volume>, <fpage>1750</fpage>&#x2013;<lpage>1792</lpage>. <pub-id pub-id-type="doi">10.1007/s10461-021-03525-0</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Grabowski</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Herbeck</surname>
<given-names>J. T.</given-names>
</name>
<name>
<surname>Poon</surname>
<given-names>A. F. Y.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Genetic cluster analysis for hiv prevention</article-title>. <source>Curr. HIV/AIDS Rep.</source> <volume>15</volume>, <fpage>182</fpage>&#x2013;<lpage>189</lpage>. <pub-id pub-id-type="doi">10.1007/s11904-018-0384-1</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hayes</surname>
<given-names>A. F.</given-names>
</name>
<name>
<surname>Krippendorff</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Answering the call for a standard reliability measure for coding data</article-title>. <source>Commun. Methods Meas.</source> <volume>1</volume>, <fpage>77</fpage>&#x2013;<lpage>89</lpage>. <pub-id pub-id-type="doi">10.1080/19312450709336664</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Holmes</surname>
<given-names>E. C.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L. Q.</given-names>
</name>
<name>
<surname>Robertson</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Cleland</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Harvey</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Simmonds</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>1995</year>). <article-title>The molecular epidemiology of human immunodeficiency virus type 1 in Edinburgh</article-title>. <source>J. Infect. Dis.</source> <volume>171</volume>, <fpage>45</fpage>&#x2013;<lpage>53</lpage>. <pub-id pub-id-type="doi">10.1093/infdis/171.1.45</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Inzaule</surname>
<given-names>S. C.</given-names>
</name>
<name>
<surname>Siedner</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Little</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Avila-Rios</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ayitewala</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bosch</surname>
<given-names>R. J.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Recommendations on data sharing in hiv drug resistance research</article-title>. <source>PLoS Med.</source> <volume>20</volume>, <fpage>e1004293</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pmed.1004293</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jia</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yue</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yue</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>The distribution of hepatitis C viral genotypes shifted among chronic hepatitis c patients in yunnan, China, between 2008-2018</article-title>. <source>Front. Cell Infect. Microbiol.</source> <volume>13</volume>, <fpage>1092936</fpage>. <pub-id pub-id-type="doi">10.3389/fcimb.2023.1092936</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jombart</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Eggo</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Dodd</surname>
<given-names>P. J.</given-names>
</name>
<name>
<surname>Balloux</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Reconstructing disease outbreaks from genetic data: a graph approach</article-title>. <source>Heredity</source> <volume>106</volume>, <fpage>383</fpage>&#x2013;<lpage>390</lpage>. <pub-id pub-id-type="doi">10.1038/hdy.2010.78</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Junqueira</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Sibisi</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wilkinson</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>de Oliveira</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Factors influencing HIV-1 phylogenetic clustering</article-title>. <source>Curr. Opin. HIV AIDS</source> <volume>14</volume>, <fpage>161</fpage>&#x2013;<lpage>172</lpage>. <pub-id pub-id-type="doi">10.1097/COH.0000000000000540</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kosakovsky Pond</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Posada</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Stawiski</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Chappey</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Poon</surname>
<given-names>A. F.</given-names>
</name>
<name>
<surname>Hughes</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>An evolutionary model-based algorithm for accurate phylogenetic breakpoint mapping and subtype prediction in hiv-1</article-title>. <source>PLoS Comput. Biol.</source> <volume>5</volume>, <fpage>e1000581</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1000581</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kosakovsky Pond</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Weaver</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Leigh Brown</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Wertheim</surname>
<given-names>J. O.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>HIV-TRACE (TRAnsmission cluster engine): a tool for large scale molecular epidemiology of HIV-1 and other rapidly evolving pathogens</article-title>. <source>Mol. Biol. Evol.</source> <volume>35</volume>, <fpage>1812</fpage>&#x2013;<lpage>1819</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msy016</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Leal</surname>
<given-names>&#xc9;.</given-names>
</name>
<name>
<surname>Arrais</surname>
<given-names>C. R.</given-names>
</name>
<name>
<surname>Barreiros</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Farias Rodrigues</surname>
<given-names>J. K.</given-names>
</name>
<name>
<surname>Silva Sousa</surname>
<given-names>N. P.</given-names>
</name>
<name>
<surname>Duarte Costa</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Characterization of hiv-1 genetic diversity and antiretroviral resistance in the state of maranh&#xe3;o, northeast Brazil</article-title>. <source>PLoS One</source> <volume>15</volume>, <fpage>e0230878</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0230878</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Leung</surname>
<given-names>K. S.-S.</given-names>
</name>
<name>
<surname>To</surname>
<given-names>S. W.-C.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J. H.-K.</given-names>
</name>
<name>
<surname>Siu</surname>
<given-names>G. K.-H.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>K. C.-W.</given-names>
</name>
<name>
<surname>Yam</surname>
<given-names>W.-C.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Molecular characterization of HIV-1 minority subtypes in Hong Kong: a recent epidemic of CRF07_bc among the men who have sex with men population</article-title>. <source>Curr. HIV Res.</source> <volume>17</volume>, <fpage>53</fpage>&#x2013;<lpage>64</lpage>. <pub-id pub-id-type="doi">10.2174/1570162X17666190530081355</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Hiv-1 pretreatment drug resistance and genetic transmission network in the southwest border region of China</article-title>. <source>BMC Infect. Dis.</source> <volume>22</volume>, <fpage>741</fpage>. <pub-id pub-id-type="doi">10.1186/s12879-022-07734-3</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Little</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Kosakovsky Pond</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Anderson</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Young</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Wertheim</surname>
<given-names>J. O.</given-names>
</name>
<name>
<surname>Mehta</surname>
<given-names>S. R.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Using hiv networks to inform real time prevention interventions</article-title>. <source>PLoS One</source> <volume>9</volume>, <fpage>e98443</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0098443</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>An</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Dynamics of HIV-1 molecular networks reveal effective control of large transmission clusters in an area affected by an epidemic of multiple HIV subtypes</article-title>. <source>Front. Microbiol.</source> <volume>11</volume>, <fpage>604993</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2020.604993</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mai</surname>
<given-names>T. Q.</given-names>
</name>
<name>
<surname>Martinez</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Menon</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Van Anh</surname>
<given-names>N. T.</given-names>
</name>
<name>
<surname>Hien</surname>
<given-names>N. T.</given-names>
</name>
<name>
<surname>Marais</surname>
<given-names>B. J.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>
<italic>Mycobacterium tuberculosis</italic> drug resistance and transmission among human immunodeficiency virus&#x2013;infected patients in Ho chi minh city, vietnam</article-title>. <source>Am. J. Trop. Med. Hyg.</source> <volume>99</volume>, <fpage>1397</fpage>&#x2013;<lpage>1406</lpage>. <pub-id pub-id-type="doi">10.4269/ajtmh.18-0185</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Molloy</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Reed</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>1995</year>). <article-title>A critical point for random graphs with a given degree sequence</article-title>. <source>Random Struct. Algorithms</source> <volume>6</volume>, <fpage>161</fpage>&#x2013;<lpage>180</lpage>. <pub-id pub-id-type="doi">10.1002/rsa.3240060204</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Murphy</surname>
<given-names>D. G.</given-names>
</name>
<name>
<surname>Dion</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Simard</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Vachon</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>Martel-Laferri&#xe8;re</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Serhir</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2019a</year>). <article-title>Molecular surveillance of hepatitis C virus genotypes identifies the emergence of a genotype 4d lineage among men in Quebec, 2001-2017</article-title>. <source>Can. Commun. Dis. Rep. &#x3d; Releve Des. Mal. Transm. Au Can.</source> <volume>45</volume>, <fpage>230</fpage>&#x2013;<lpage>237</lpage>. <pub-id pub-id-type="doi">10.14745/ccdr.v45i09a02</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Murphy</surname>
<given-names>D. G.</given-names>
</name>
<name>
<surname>Dion</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Simard</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Vachon</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>Martel-Laferri&#xe8;re</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Serhir</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2019b</year>). <article-title>Molecular surveillance of hepatitis c virus genotypes identifies the emergence of a genotype 4d lineage among men in quebec, 2001-2017</article-title>. <source>Can. Commun. Dis. Rep.</source> <volume>45</volume>, <fpage>230</fpage>&#x2013;<lpage>237</lpage>. <pub-id pub-id-type="doi">10.14745/ccdr.v45i09a02</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ng</surname>
<given-names>K. T.</given-names>
</name>
<name>
<surname>Ng</surname>
<given-names>L. J.</given-names>
</name>
<name>
<surname>Oong</surname>
<given-names>X. Y.</given-names>
</name>
<name>
<surname>Chook</surname>
<given-names>J. B.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>K. G.</given-names>
</name>
<name>
<surname>Takebe</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Application of a vp4/vp2-inferred transmission clusters in estimating the impact of interventions on rhinovirus transmission</article-title>. <source>Virol. J.</source> <volume>19</volume>, <fpage>36</fpage>. <pub-id pub-id-type="doi">10.1186/s12985-022-01762-w</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Novitsky</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Moyo</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lei</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>DeGruttola</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Essex</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Impact of sampling density on the extent of HIV clustering</article-title>. <source>AIDS Res. Hum. retroviruses</source> <volume>30</volume>, <fpage>1226</fpage>&#x2013;<lpage>1235</lpage>. <pub-id pub-id-type="doi">10.1089/aid.2014.0173</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Novitsky</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Steingrimsson</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Howison</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Gillani</surname>
<given-names>F. S.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Manne</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Empirical comparison of analytical approaches for identifying molecular HIV-1 clusters</article-title>. <source>Sci. Rep.</source> <volume>10</volume>, <fpage>18547</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-020-75560-1</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Oster</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>France</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Panneer</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Ba&#xf1;ez Ocfemia</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Campbell</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Dasgupta</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Identifying clusters of recent and rapid HIV transmission through analysis of molecular surveillance data</article-title>. <source>J. Acquir. Immune Defic. Syndromes</source> <volume>79</volume>, <fpage>543</fpage>&#x2013;<lpage>550</lpage>. <pub-id pub-id-type="doi">10.1097/QAI.0000000000001856</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Oster</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Lyss</surname>
<given-names>S. B.</given-names>
</name>
<name>
<surname>McClung</surname>
<given-names>R. P.</given-names>
</name>
<name>
<surname>Watson</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Panneer</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Hernandez</surname>
<given-names>A. L.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>HIV cluster and outbreak detection and response: the science and experience</article-title>. <source>Am. J. Prev. Med.</source> <volume>61</volume>, <fpage>S130</fpage>&#x2013;<lpage>S142</lpage>. <pub-id pub-id-type="doi">10.1016/j.amepre.2021.05.029</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paraschiv</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Banica</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Nicolae</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Niculescu</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Abagiu</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Jipa</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Epidemic dispersion of HIV and HCV in a population of co-infected Romanian injecting drug users</article-title>. <source>PLoS One</source> <volume>12</volume>, <fpage>e0185866</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0185866</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paraskevis</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Nikolopoulos</surname>
<given-names>G. K.</given-names>
</name>
<name>
<surname>Magiorkinis</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Hodges-Mameletzis</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Hatzakis</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>The application of HIV molecular epidemiology to public health</article-title>. <source>Infect. Genet. Evol. J. Mol. Epidemiol. Evol. Genet. Infect. Dis.</source> <volume>46</volume>, <fpage>159</fpage>&#x2013;<lpage>168</lpage>. <pub-id pub-id-type="doi">10.1016/j.meegid.2016.06.021</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Patil</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Patil</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rao</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gadhe</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kurle</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Panda</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Exploring the evolutionary history and phylodynamics of human immunodeficiency virus type 1 outbreak from unnao, India using phylogenetic approach</article-title>. <source>Front. Microbiol.</source> <volume>13</volume>, <fpage>848250</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2022.848250</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Penn</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Stern</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Rubinstein</surname>
<given-names>N. D.</given-names>
</name>
<name>
<surname>Dutheil</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Bacharach</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Galtier</surname>
<given-names>N.</given-names>
</name>
<etal/>
</person-group> (<year>2008</year>). <article-title>Evolutionary modeling of rate shifts reveals specificity determinants in hiv-1 subtypes</article-title>. <source>PLoS Comput. Biol.</source> <volume>4</volume>, <fpage>e1000214</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1000214</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>P&#xe9;rez-Losada</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Castel</surname>
<given-names>A. D.</given-names>
</name>
<name>
<surname>Lewis</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Kharfen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Cartwright</surname>
<given-names>C. P.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Characterization of HIV diversity, phylodynamics and drug resistance in Washington, DC</article-title>. <source>PLoS One</source> <volume>12</volume>, <fpage>e0185644</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0185644</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peters</surname>
<given-names>P. J.</given-names>
</name>
<name>
<surname>Pontones</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Hoover</surname>
<given-names>K. W.</given-names>
</name>
<name>
<surname>Patel</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Galang</surname>
<given-names>R. R.</given-names>
</name>
<name>
<surname>Shields</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>HIV infection linked to injection use of oxymorphone in Indiana, 2014-2015</article-title>. <source>N. Engl. J. Med.</source> <volume>375</volume>, <fpage>229</fpage>&#x2013;<lpage>239</lpage>. <pub-id pub-id-type="doi">10.1056/NEJMoa1515195</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Potterat</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Phillips-Plummer</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Muth</surname>
<given-names>S. Q.</given-names>
</name>
<name>
<surname>Rothenberg</surname>
<given-names>R. B.</given-names>
</name>
<name>
<surname>Woodhouse</surname>
<given-names>D. E.</given-names>
</name>
<name>
<surname>Maldonado-Long</surname>
<given-names>T. S.</given-names>
</name>
<etal/>
</person-group> (<year>2002</year>). <article-title>Risk network structure in the early epidemic phase of HIV transmission in Colorado Springs</article-title>. <source>Sex. Transm. Infect.</source> <volume>78</volume>, <fpage>i159</fpage>&#x2013;<lpage>i163</lpage>. <pub-id pub-id-type="doi">10.1136/sti.78.suppl_1.i159</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ragonnet-Cronin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Benbow</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Hayford</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Poortinga</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Forgione</surname>
<given-names>L. A.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Sorting by race/ethnicity across hiv genetic transmission networks in three major metropolitan areas in the United States</article-title>. <source>AIDS Res. Hum. retroviruses</source> <volume>37</volume>, <fpage>784</fpage>&#x2013;<lpage>792</lpage>. <pub-id pub-id-type="doi">10.1089/aid.2020.0145</pub-id>
</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ragonnet-Cronin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hayford</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>D&#x2019;Aquila</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Ward</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Benbow</surname>
<given-names>N.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Forecasting HIV-1 genetic cluster growth in Illinois,United States</article-title>. <source>J. Acquir. Immune Defic. Syndromes</source> <volume>89</volume>, <fpage>49</fpage>&#x2013;<lpage>55</lpage>. <pub-id pub-id-type="doi">10.1097/QAI.0000000000002821</pub-id>
</citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ragonnet-Cronin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hodcroft</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Hu&#xe9;</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Fearnhill</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Delpech</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Brown</surname>
<given-names>A. J. L.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Automated analysis of phylogenetic clusters</article-title>. <source>BMC Bioinforma.</source> <volume>14</volume>, <fpage>317</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-14-317</pub-id>
</citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ragonnet-Cronin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hu&#xe9;</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hodcroft</surname>
<given-names>E. B.</given-names>
</name>
<name>
<surname>Tostevin</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dunn</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Fawcett</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Non-disclosed men who have sex with men in UK HIV transmission networks: phylogenetic analysis of surveillance data</article-title>. <source>Lancet HIV</source> <volume>5</volume>, <fpage>e309</fpage>&#x2013;<lpage>e316</lpage>. <pub-id pub-id-type="doi">10.1016/S2352-3018(18)30062-6</pub-id>
</citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rhee</surname>
<given-names>S.-Y.</given-names>
</name>
<name>
<surname>Magalis</surname>
<given-names>B. R.</given-names>
</name>
<name>
<surname>Hurley</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Silverberg</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Marcus</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Slome</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>National and international dimensions of human immunodeficiency virus-1 sequence clusters in a northern California clinical cohort</article-title>. <source>Open Forum Infect. Dis.</source> <volume>6</volume>, <fpage>ofz135</fpage>. <pub-id pub-id-type="doi">10.1093/ofid/ofz135</pub-id>
</citation>
</ref>
<ref id="B64">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Robinson</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Fyson</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Cohen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Fraser</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Colijn</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>How the dynamics and structure of sexual contact networks shape pathogen phylogenies</article-title>. <source>PLOS Comput. Biol.</source> <volume>9</volume>, <fpage>e1003105</fpage>. <comment>Publisher: Public Library of Science</comment>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1003105</pub-id>
</citation>
</ref>
<ref id="B65">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rose</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Cross</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lamers</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Astemborski</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kirk</surname>
<given-names>G. D.</given-names>
</name>
<name>
<surname>Mehta</surname>
<given-names>S. H.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Persistence of HIV transmission clusters among people who inject drugs</article-title>. <source>AIDS Lond. Engl.</source> <volume>34</volume>, <fpage>2037</fpage>&#x2013;<lpage>2044</lpage>. <pub-id pub-id-type="doi">10.1097/QAD.0000000000002662</pub-id>
</citation>
</ref>
<ref id="B66">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Scherrer</surname>
<given-names>A. U.</given-names>
</name>
<name>
<surname>Traytel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Braun</surname>
<given-names>D. L.</given-names>
</name>
<name>
<surname>Calmy</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Battegay</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Cavassini</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Cohort profile update: the Swiss HIV cohort study (SHCS)</article-title>. <source>Int. J. Epidemiol.</source> <volume>51</volume>, <fpage>33</fpage>&#x2013;<lpage>34j</lpage>. <pub-id pub-id-type="doi">10.1093/ije/dyab141</pub-id>
</citation>
</ref>
<ref id="B67">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sivay</surname>
<given-names>M. V.</given-names>
</name>
<name>
<surname>Hudelson</surname>
<given-names>S. E.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Agyei</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hamilton</surname>
<given-names>E. L.</given-names>
</name>
<name>
<surname>Selin</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>HIV-1 diversity among young women in rural South Africa: HPTN 068</article-title>. <source>PloS One</source> <volume>13</volume>, <fpage>e0198999</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0198999</pub-id>
</citation>
</ref>
<ref id="B68">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sizemore</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Fill</surname>
<given-names>M.-M.</given-names>
</name>
<name>
<surname>Mathieson</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Black</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Brantley</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Cooper</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Using an established outbreak response plan and molecular epidemiology methods in an HIV transmission cluster investigation, Tennessee, january-june 2017</article-title>. <source>Public Health Rep. Wash. D.C. 1974</source> <volume>135</volume>, <fpage>329</fpage>&#x2013;<lpage>333</lpage>. <pub-id pub-id-type="doi">10.1177/0033354920915445</pub-id>
</citation>
</ref>
<ref id="B69">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stecher</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chaillon</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Eberle</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Behrens</surname>
<given-names>G. M. N.</given-names>
</name>
<name>
<surname>Eis-H&#xfc;binger</surname>
<given-names>A.-M.</given-names>
</name>
<name>
<surname>Lehmann</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Molecular epidemiology of the hiv epidemic in three German metropolitan regions - cologne/bonn, munich and hannover, 1999-2016</article-title>. <source>Sci. Rep.</source> <volume>8</volume>, <fpage>6799</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-018-25004-8</pub-id>
</citation>
</ref>
<ref id="B70">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tamura</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Nei</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>1993</year>). <article-title>Estimation of the number of nucleotide substitutions in the control region of mitochondrial DNA in humans and chimpanzees</article-title>. <source>Mol. Biol. Evol.</source> <volume>10</volume>, <fpage>512</fpage>&#x2013;<lpage>526</lpage>. <pub-id pub-id-type="doi">10.1093/oxfordjournals.molbev.a040023</pub-id>
</citation>
</ref>
<ref id="B71">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Temereanca</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Oprea</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wertheim</surname>
<given-names>J. O.</given-names>
</name>
<name>
<surname>Ianache</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Ceausu</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Cernescu</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Hiv transmission clusters among injecting drug users in Romania</article-title>. <source>Rom. Biotechnol. Lett.</source> <volume>22</volume>, <fpage>12307</fpage>&#x2013;<lpage>12315</lpage>.</citation>
</ref>
<ref id="B72">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thoma</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Seneghini</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Seiffert</surname>
<given-names>S. N.</given-names>
</name>
<name>
<surname>Vuichard Gysin</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Scanferla</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Haller</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>The challenge of preventing and containing outbreaks of multidrug-resistant organisms and Candida auris during the coronavirus disease 2019 pandemic: report of a carbapenem-resistant Acinetobacter baumannii outbreak and a systematic review of the literature</article-title>. <source>Antimicrob. Resist. Infect. Control</source> <volume>11</volume>, <fpage>12</fpage>. <pub-id pub-id-type="doi">10.1186/s13756-022-01052-8</pub-id>
</citation>
</ref>
<ref id="B73">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tookes</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Bartholomew</surname>
<given-names>T. S.</given-names>
</name>
<name>
<surname>Geary</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Matthias</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Poschman</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Blackmore</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Rapid identification and investigation of an HIV risk network among people who inject drugs -miami, FL, 2018</article-title>. <source>AIDS Behav.</source> <volume>24</volume>, <fpage>246</fpage>&#x2013;<lpage>256</lpage>. <pub-id pub-id-type="doi">10.1007/s10461-019-02680-9</pub-id>
</citation>
</ref>
<ref id="B74">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tumpney</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>John</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Panneer</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>McClung</surname>
<given-names>R. P.</given-names>
</name>
<name>
<surname>Campbell</surname>
<given-names>E. M.</given-names>
</name>
<name>
<surname>Roosevelt</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Human immunodeficiency virus (HIV) outbreak investigation among persons who inject drugs in Massachusetts enhanced by HIV sequence data</article-title>. <source>J. Infect. Dis.</source> <volume>222</volume>, <fpage>S259</fpage>&#x2013;<lpage>S267</lpage>. <pub-id pub-id-type="doi">10.1093/infdis/jiaa053</pub-id>
</citation>
</ref>
<ref id="B75">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Volz</surname>
<given-names>E. M.</given-names>
</name>
<name>
<surname>Ndembi</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Nowak</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Kijak</surname>
<given-names>G. H.</given-names>
</name>
<name>
<surname>Idoko</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Dakum</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Phylodynamic analysis to inform prevention efforts in mixed hiv epidemics</article-title>. <source>Virus Evol.</source> <volume>3</volume>, <fpage>vex014</fpage>. <pub-id pub-id-type="doi">10.1093/ve/vex014</pub-id>
</citation>
</ref>
<ref id="B76">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>von Rotz</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kuehl</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Durovic</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zingg</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Apitz</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wegner</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>A systematic outbreak investigation of SARS-CoV-2 transmission clusters in a tertiary academic care center</article-title>. <source>Antimicrob. Resist. Infect. Control</source> <volume>12</volume>, <fpage>38</fpage>. <pub-id pub-id-type="doi">10.1186/s13756-023-01242-y</pub-id>
</citation>
</ref>
<ref id="B77">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vrancken</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Adachi</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Benedet</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Read</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Shafran</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>The multi-faceted dynamics of HIV-1 transmission in Northern Alberta: a combined analysis of virus genetic and public health data</article-title>. <source>Infect. Genet. Evol. J. Mol. Epidemiol. Evol. Genet. Infect. Dis.</source> <volume>52</volume>, <fpage>100</fpage>&#x2013;<lpage>105</lpage>. <pub-id pub-id-type="doi">10.1016/j.meegid.2017.04.005</pub-id>
</citation>
</ref>
<ref id="B78">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Targeting HIV prevention based on molecular epidemiology among deeply sampled subnetworks of men who have sex with men</article-title>. <source>Clin. Infect. Dis. Official Publ. Infect. Dis. Soc. Am.</source> <volume>61</volume>, <fpage>1462</fpage>&#x2013;<lpage>1468</lpage>. <pub-id pub-id-type="doi">10.1093/cid/civ526</pub-id>
</citation>
</ref>
<ref id="B79">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Weaver</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shank</surname>
<given-names>S. D.</given-names>
</name>
<name>
<surname>Spielman</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Muse</surname>
<given-names>S. V.</given-names>
</name>
<name>
<surname>Kosakovsky Pond</surname>
<given-names>S. L.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Datamonkey 2.0: a modern web application for characterizing selective and other evolutionary processes</article-title>. <source>Mol. Biol. Evol.</source> <volume>35</volume>, <fpage>773</fpage>&#x2013;<lpage>777</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msx335</pub-id>
</citation>
</ref>
<ref id="B80">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wertheim</surname>
<given-names>J. O.</given-names>
</name>
<name>
<surname>Leigh Brown</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Hepler</surname>
<given-names>N. L.</given-names>
</name>
<name>
<surname>Mehta</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Richman</surname>
<given-names>D. D.</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>D. M.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>The global transmission network of HIV-1</article-title>. <source>J. Infect. Dis.</source> <volume>209</volume>, <fpage>304</fpage>&#x2013;<lpage>313</lpage>. <pub-id pub-id-type="doi">10.1093/infdis/jit524</pub-id>
</citation>
</ref>
<ref id="B81">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wolf</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Herbeck</surname>
<given-names>J. T.</given-names>
</name>
<name>
<surname>Van Rompaey</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kitahata</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Thomas</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Pepper</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Short communication: phylogenetic evidence of HIV-1 transmission between adult and adolescent men who have sex with men</article-title>. <source>AIDS Res. Hum. retroviruses</source> <volume>33</volume>, <fpage>318</fpage>&#x2013;<lpage>322</lpage>. <pub-id pub-id-type="doi">10.1089/AID.2016.0061</pub-id>
</citation>
</ref>
<ref id="B82">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Q.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>The central role of nondisclosed men who have sex with men in human immunodeficiency virus-1 transmission networks in guangzhou, China</article-title>. <source>Open Forum Infect. Dis.</source> <volume>7</volume>, <fpage>ofaa154</fpage>. <pub-id pub-id-type="doi">10.1093/ofid/ofaa154</pub-id>
</citation>
</ref>
<ref id="B83">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Q.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Acquisition and transmission of hiv-1 among migrants and Chinese in guangzhou, China from 2008 to 2012: phylogenetic analysis of surveillance data</article-title>. <source>Infect. Genet. Evol.</source> <volume>92</volume>, <fpage>104870</fpage>. <pub-id pub-id-type="doi">10.1016/j.meegid.2021.104870</pub-id>
</citation>
</ref>
<ref id="B84">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ye</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Distribution pattern, molecular transmission networks, and photodynamic of hepatitis c virus in China</article-title>. <source>PLoS One</source> <volume>18</volume>, <fpage>e0296053</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0296053</pub-id>
</citation>
</ref>
<ref id="B85">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yebra</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Ragonnet-Cronin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ssemwanga</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Parry</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Logue</surname>
<given-names>C. H.</given-names>
</name>
<name>
<surname>Cane</surname>
<given-names>P. A.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Analysis of the history and spread of HIV-1 in Uganda using phylodynamics</article-title>. <source>J. Gen. Virol.</source> <volume>96</volume>, <fpage>1890</fpage>&#x2013;<lpage>1898</lpage>. <pub-id pub-id-type="doi">10.1099/vir.0.000107</pub-id>
</citation>
</ref>
<ref id="B86">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Prevalence of drug resistance and genetic transmission networks among human immunodeficiency virus/acquired immunodeficiency syndrome patients with antiretroviral therapy failure in guangxi, China</article-title>. <source>AIDS Res. Hum. Retroviruses</source> <volume>38</volume>, <fpage>822</fpage>&#x2013;<lpage>830</lpage>. <pub-id pub-id-type="doi">10.1089/AID.2021.0181</pub-id>
</citation>
</ref>
<ref id="B87">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zai</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chaillon</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Tracing the transmission dynamics of hiv-1 crf55_01b</article-title>. <source>Sci. Rep.</source> <volume>10</volume>, <fpage>5098</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-020-61870-x</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>