<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Malar.</journal-id>
<journal-title>Frontiers in Malaria</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Malar.</abbrev-journal-title>
<issn pub-type="epub">2813-7396</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmala.2024.1363981</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Malaria</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A non-parametric approach to estimate multiplicity of infection and pathogen haplotype frequencies</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Kayanula</surname>
<given-names>Loyce</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2011072"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Schneider</surname>
<given-names>Kristan Alexander</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1566421"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Applied Computer and Biosciences, University of Applied Sciences Mittweida</institution>, <addr-line>Mittweida</addr-line>, <country>Germany</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Center for Global Health, Department of Internal Medicine, School of Medicine, University of New Mexico</institution>, <addr-line>Albuquerque, NM</addr-line>, <country>United States</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Translational Informatics Division, Department of Internal Medicine, School of Medicine, University of New Mexico</institution>, <addr-line>Albuquerque, NM</addr-line>, <country>United States</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Clinical and Translational Science Center, Health Science Center, University of New Mexico</institution>, <addr-line>Albuquerque, NM</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Sarah Reece, University of Edinburgh, United Kingdom</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Megan Greischar, Cornell University, United States</p>
<p>Lucy Okell, Imperial College London, United Kingdom</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Loyce Kayanula, <email xlink:href="mailto:lkayanul@hs-mittweida.de">lkayanul@hs-mittweida.de</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>18</day>
<month>07</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>2</volume>
<elocation-id>1363981</elocation-id>
<history>
<date date-type="received">
<day>31</day>
<month>12</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>07</day>
<month>05</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Kayanula and Schneider</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Kayanula and Schneider</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>The presence of multiple genetically distinct variants (lineages) within an infection (multiplicity of infection, MOI) is common in infectious diseases such as malaria. MOI is considered an epidemiologically and clinically relevant quantity that scales with transmission intensity and potentially impacts the clinical pathogenesis of the disease. Several statistical methods to estimate MOI assume that the number of infectious events per person follows a Poisson distribution. However, this has been criticized since empirical evidence suggests that the number of mosquito bites per person is over-dispersed compared to the Poisson distribution. </p>
</sec>
<sec>
<title>Methods</title>
<p>We introduce a statistical model that does not assume that MOI follows a parametric distribution, i.e., the most flexible possible approach. The method is designed to estimate the distribution of MOI and allele frequency distributions from a single molecular marker. We derive the likelihood function and propose a maximum likelihood approach to estimate the desired parameters. The expectation maximization algorithm (EM algorithm) is used to numerically calculate the maximum likelihood estimate. </p>
</sec>
<sec>
<title>Results</title>
<p>By numerical simulations, we evaluate the performance of the proposed method in comparison to an established method that assumes a Poisson distribution for MOI. Our results suggest that the Poisson model performs sufficiently well if MOI is not highly over-dispersed. Hence, any model extension will not greatly improve the estimation of MOI. However, if MOI is highly over-dispersed, the method is less biased. We exemplify the method by analyzing three empirical evidence in <italic>P. falciparum</italic> data sets from drug resistance studies in Venezuela, Cameroon, and Kenya. Based on the allele frequency estimates, we estimate the heterozygosity and the average MOI for the respective microsatellite markers. </p>
</sec>
<sec>
<title>Discussion</title>
<p>In conclusion, the proposed non-parametric method to estimate the distribution of MOI is appropriate when the transmission intensities in the population are heterogeneous, yielding an over-dispersed distribution. If MOI is not highly over-dispersed, the Poisson model is sufficiently accurate and cannot be improved by other methods. The EM algorithm provides a numerically stable method to derive MOI estimates and is made available as an R script.</p>
</sec>
</abstract>
<kwd-group>
<kwd>malaria</kwd>
<kwd>complexity of infection</kwd>
<kwd>molecular surveillance</kwd>
<kwd>drug resistance</kwd>
<kwd>prevalence</kwd>
<kwd>transmission intensities</kwd>
<kwd>superinfection</kwd>
<kwd>co-infection</kwd>
</kwd-group>
<contract-num rid="cn001">Project-ID 57417782</contract-num>
<contract-sponsor id="cn001">German Academic Exchange Service<named-content content-type="fundref-id">10.13039/100021828</named-content>
</contract-sponsor>
<contract-sponsor id="cn002">Bundesministerium f&#xfc;r Bildung und Forschung<named-content content-type="fundref-id">10.13039/501100002347</named-content>
</contract-sponsor>
<counts>
<fig-count count="10"/>
<table-count count="2"/>
<equation-count count="19"/>
<ref-count count="48"/>
<page-count count="17"/>
<word-count count="8930"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Antimalarial Drug Resistance</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Malaria and similar infectious diseases often exhibit a complex landscape of pathogen genetic diversity within individual infections. In malaria, the presence of multiple genetically distinct pathogen variants within a single infection is typically referred to as complexity of infection (COI) or multiplicity of infection (MOI) (<xref ref-type="bibr" rid="B35">Read and Taylor, 2001</xref>; <xref ref-type="bibr" rid="B4">Chang et&#xa0;al., 2017</xref>; <xref ref-type="bibr" rid="B37">Schneider, 2018</xref>). MOI is commonly reported in the context of malaria molecular surveillance, which has become increasingly popular in the last years as molecular assays become more affordable in endemic settings and due to WHO recommendations (<xref ref-type="bibr" rid="B47">World Health Organization, 2022</xref>; <xref ref-type="bibr" rid="B43">Sinha et&#xa0;al., 2023</xref>).</p>
<p>Epidemiologically, MOI is an important parameter as it scales with transmission intensities, however not necessarily in a linear way (<xref ref-type="bibr" rid="B30">Pacheco et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B43">Sinha et&#xa0;al., 2023</xref>). The effect of interactions of different pathogen variants within infections (intra-host dynamics) has been considered important in several theoretical models (e.g., <xref ref-type="bibr" rid="B15">Hastings and Watkins, 2005</xref>; <xref ref-type="bibr" rid="B11">Gurarie and McKenzie, 2006</xref>). However, so far, empirical evidence on the impact of MOI on the clinical pathogenesis of malaria remains inconclusive (<xref ref-type="bibr" rid="B31">Pacheco et&#xa0;al., 2016</xref>). Nevertheless, the distribution of MOI influences the evolutionary dynamics of malaria (<xref ref-type="bibr" rid="B38">Schneider, 2021</xref>; <xref ref-type="bibr" rid="B41">Schneider and Salas, 2022</xref>) and mediates evolutionary&#x2013;genetic patterns and pathogen genetic diversity (e.g., patterns of genetic hitchhiking and linkage disequilibrium) as it affects the effective rate of recombination as illustrated in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref> (<xref ref-type="bibr" rid="B40">Schneider and Kim, 2010</xref>; <xref ref-type="bibr" rid="B2">Alizon et&#xa0;al., 2013</xref>). Moreover, there is an important link between the frequency distribution of pathogen variants, their occurrence within infections (i.e., prevalence), and MOI. Specifically, given the frequency of a certain pathogen variant, its prevalence increases with MOI. This is particularly relevant in the context of anti-malarial drug resistance and seasonal malaria with varying transmission intensities (<xref ref-type="bibr" rid="B9">Geiger et&#xa0;al., 2014</xref>; <xref ref-type="bibr" rid="B38">Schneider, 2021</xref>; <xref ref-type="bibr" rid="B42">Schneider et&#xa0;al., 2022</xref>).</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Illustration of MOI: Illustrated are two infections with MOI = 1 (single infection) shown on the top and MOI = 4 (multiple infection) shown on the bottom. Single infections lead effectively to no recombination. Note that the host on the bottom is infected twice with the same lineage. Because there are three different lineages in the infection, recombination can happen during host&#x2013;vector transmission.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmala-02-1363981-g001.tif"/>
</fig>
<p>Importantly, MOI or COI is not unambiguously defined. Several formulations of MOI exist, with discrepancies between verbal and formal definitions (which typically underlay statistical models) as discussed in detail in <xref ref-type="bibr" rid="B42">Schneider et al. (2022)</xref>. Here we follow the suggested definition in <xref ref-type="bibr" rid="B42">Schneider et al. (2022)</xref>, which is used in most theoretical frameworks. Particularly, MOI is not defined as the number of distinct pathogen variants within an infection but as the number of &#x201c;super-infections&#x201d; during one disease episode. More precisely, MOI is the number of independent infectious events, assuming that exactly one pathogen variant (lineage) is transmitted per event (<xref ref-type="bibr" rid="B16">Hill and Babiker, 1995</xref>; <xref ref-type="bibr" rid="B39">Schneider and Escalante, 2014</xref>). This implies that MOI is an unobservable quantity because a host can be infected multiple times with the same pathogen variant (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>), and these infectious events cannot be reconstructed from molecular assays (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>). Notably, as pointed out in <xref ref-type="bibr" rid="B42">Schneider et al. (2022)</xref>, if the distribution of MOI in the population is known or estimated, the distribution of different pathogen variants within infections can be derived (but not <italic>vice versa</italic>). The definition of MOI used here only approximately accounts for &#x201c;co-infections&#x201d;, i.e., the co-transmission of several pathogenic variants during an infective episode (<xref ref-type="bibr" rid="B38">Schneider, 2021</xref>). Focusing on super-infections (and only approximately covering co-infections) has the pragmatic advantage that no explicit model of vector&#x2013;host transmission has to be specified (<xref ref-type="bibr" rid="B38">Schneider, 2021</xref>) considering that a co-infection is particularly relevant if one aims to determine the genetic relatedness of pathogen variants within an infection. To obtain enough resolution to study genetic relatedness, these approaches require high-quality genomic data. Although such approaches have become increasingly popular in the last years (cf. <xref ref-type="bibr" rid="B25">Nkhoma et&#xa0;al., 2012</xref>; <xref ref-type="bibr" rid="B45">Wong et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B48">Zhu et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B26">Nkhoma et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B6">Dia and Cheeseman, 2021</xref>; <xref ref-type="bibr" rid="B23">Neafsey et&#xa0;al., 2021</xref>), these approaches are inappropriate if only a handful of genetic markers are available.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Infections and observations: Illustrated are three infections from four lineages circulating in the pathogen population. The first infection has MOI = 2 and contains two distinct pathogen variants. The second infection has MOI = 3, but only with two distinct pathogen variants, and the third has MOI = 4 and contains three distinct pathogen variants. Only the presence or absence of variants can be observed by molecular assays. The number of times that each lineage was transmitted cannot be reconstructed from a sample; generally, MOI is unobservable.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmala-02-1363981-g002.tif"/>
</fig>
<p>Heuristic methods to estimate the distribution of MOI are typically biased; methods based on a solid statistical framework are preferable (<xref ref-type="bibr" rid="B38">Schneider, 2021</xref>). Several such methods (which are essentially based on the same statistical framework) have been proposed to estimate the MOI and lineage frequencies for various assumptions concerning the genetic architecture of the underlying molecular data. All these methods make certain assumptions on the distribution of MOI in the population. The maximum likelihood-based method of <xref ref-type="bibr" rid="B16">Hill and Babiker (1995)</xref> assumes that MOI follows either a (conditional) Poisson or negative binomial distribution and is based on one or two genetic markers. In the case of MOI following a positive Poisson distribution, this method was refined by applying a bias correction in the case of a single molecular marker by <xref ref-type="bibr" rid="B12">Hashemi and Schneider (2021)</xref> and to an arbitrary number of biallelic molecular markers (<xref ref-type="bibr" rid="B44">Tsoungui Obama and Schneider, 2022</xref>). Moreover, the approach of <xref ref-type="bibr" rid="B18">Li et al. (2007)</xref> assumes that MOI follows a (conditional) Poisson distribution. The Bayesian method in the program COIL (<xref ref-type="bibr" rid="B8">Galinsky et&#xa0;al., 2015</xref>) and its generalization THE REAL McCOIL (<xref ref-type="bibr" rid="B4">Chang et&#xa0;al., 2017</xref>) does not explicitly specify a specific distribution for MOI <italic>per se</italic>, but the implementation imposes a uniform distribution, which constrains the resulting posterior distribution of MOI.</p>
<p>Under the assumption that infective mosquito bites are rare and independent in a population with homogeneous exposure, MOI follows a Poisson distribution. This renders this parametric choice an important null model. However, if mosquito biting rates are heterogeneous in the population, the distribution of MOI will more likely follow a mixture of Poisson or a negative binomial distribution (<xref ref-type="bibr" rid="B42">Schneider et&#xa0;al., 2022</xref>). In fact, empirical evidence indicates that mosquito biting patterns are heterogeneous, with certain individuals experiencing more bites than others. This is influenced by confounding factors such as environmental conditions or individual attractiveness to mosquitoes (<xref ref-type="bibr" rid="B27">Noor et&#xa0;al., 2014</xref>; <xref ref-type="bibr" rid="B10">Guelbeogo et&#xa0;al., 2018</xref>). Consequently, the number of mosquito bites per person tends to be over-dispersed compared to the Poisson distribution (<xref ref-type="bibr" rid="B17">Irvine et&#xa0;al., 2018</xref>). (Note, however, that an over-dispersed mosquito biting rate does not imply that the MOI distribution is over-dispersed, as it is concerned only with the number of infective bites.) In any case, significant deviations from the Poisson assumptions suggest that the negative binomial distribution might be more suitable (<xref ref-type="bibr" rid="B19">Lloyd-Smith, 2007</xref>). However, maximum likelihood estimation of the negative binomial distribution in general is problematic. As mentioned in <xref ref-type="bibr" rid="B1">Adamidis (1999)</xref>, if the empirical variance is not larger than the mean of count data, the maximum likelihood estimates of the parameters of the negative binomial distribution are degenerate. A way to overcome the problem that estimation of both parameters can compromise the stability and interpretability of a negative binomial model (<xref ref-type="bibr" rid="B3">Bandara et&#xa0;al., 2019</xref>) is to estimate only one parameter, while fixing the other, as suggested in <xref ref-type="bibr" rid="B32">Piegorsch (1990)</xref>, <xref ref-type="bibr" rid="B36">Saha and Paul (2005)</xref>, and <xref ref-type="bibr" rid="B19">Lloyd-Smith (2007)</xref>. In practice, this means that there needs to be prior information on one parameter. However, this is typically not feasible in the context of MOI because it would require additional information such as mosquito biting rates or host exposure to mosquitoes. Such information is typically beyond the scope of malaria molecular surveillance.</p>
<p>If there is a prior belief that MOI does not follow a Poisson distribution, rather than assuming that MOI falls into a different class of parametric distributions, such as the negative binomial distribution, no particular class of distributions has to be imposed. Such a non-parametric approach offers a valid alternative if the MOI is completely unknown because of its flexibility. In fact, a non-parametric approach is the most flexible approach in this context.</p>
<p>Here we introduce a non-parametric statistical model to estimate the distribution of MOI and pathogen lineage (allele) frequencies from a single molecular marker by maximum likelihood. Non-parametric refers to the fact that the MOI distribution is not assumed to fall into a class of parametric distributions. The statistical model is first introduced in &#x201c;Materials and methods&#x201d;. Because the resulting likelihood function is too complex to have a closed-form solution, we derive the expectation maximization algorithm (EM algorithm) to derive the maximum likelihood estimate (MLE) numerically (<xref ref-type="bibr" rid="B5">Couvreur, 1997</xref>; <xref ref-type="bibr" rid="B24">Ng et&#xa0;al., 2012</xref>). The EM algorithm provides a numerically stable iteration to derive the MLE. By numerical simulations, we further investigate the performance of the non-parametric estimator in terms of bias and variance and contrast it to MOI estimates based on the assumption of an underlying Poisson distribution. The method proposed here is further applied to three data sets from Cameroon, Kenya, and Venezuela as an illustration. The method is implemented as an R script available in the <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material</bold>
</xref> and at <ext-link ext-link-type="uri" xlink:href="https://github.com/Maths-against-Malaria/Non-parametric-MOI-estimation">https://github.com/Maths-against-Malaria/Non-parametric-MOI-estimation</ext-link>
<ext-link ext-link-type="uri" xlink:href="https://github.com/Maths-against-Malaria/Non-parametric-MOI-estimation">.</ext-link>
</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<p>In the following mathematical notation, we use oblique letters, e.g., <bold>
<italic>m, p, x</italic>
</bold>, to indicate vectors, and italic fonts, e.g., <italic>m, p, x</italic>, to refer to integers or scalars.</p>
<p>We consider a pathogen population with lineages <italic>A</italic>
<sub>1</sub>
<italic>, &#x2026;</italic>, <italic>A<sub>n</sub>
</italic>, detected at a single marker locus. Each lineage (or allele) <italic>A<sub>i</sub>
</italic> has relative frequency <italic>p<sub>i</sub>
</italic> in the pathogen population, jointly denoted by the vector <bold>
<italic>p</italic>
</bold> = (<italic>p</italic>
<sub>1</sub>, <italic>&#x2026;</italic>, <italic>p<sub>n</sub>
</italic>). We assume that at each infective event, the mosquito vector transmits exactly one lineage to the host. This corresponds to randomly sampling one lineage from the pathogen population. Hence, co-infections (cf. <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref> in <xref ref-type="bibr" rid="B42">Schneider et&#xa0;al., 2022</xref>), i.e., the co-transmission of several distinct lineages during one infective event, are ignored. However, hosts can be infected multiple times by different mosquitoes (super-infections) during one disease episode. It is assumed that super-infections occur during relatively short time periods, e.g., a few days, so that all infecting variants reach detectable concentrations. Infective events, in which the variants do not reach detectable frequencies, do not count as super-infection as these (i) are irrelevant for the clinical pathogenesis of the disease and (ii) are undetectable by molecular assays. Following <xref ref-type="bibr" rid="B42">Schneider et al. (2022)</xref>, we refer to the number of infective events during one disease episode as multiplicity of infection (MOI). Importantly, the hosts might be infected multiple times with the same lineage. Formally, if <italic>m<sub>i</sub>
</italic> represents the number of times an individual was infected with lineage <italic>A<sub>i</sub>
</italic>, then <italic>m<sub>i</sub>
</italic> = 0 if the host was not infected with lineage <italic>A<sub>i</sub>
</italic>. Summing over all <italic>m<sub>i</sub>
</italic> yields MOI <italic>m</italic>, i.e., MOI <italic>m</italic> is defined by:</p>
<disp-formula>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>:</mml:mo>
<mml:mo>=</mml:mo>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mtext mathvariant="bold-italic">m</mml:mtext>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <bold>
<italic>m</italic>
</bold> = (<italic>m</italic>
<sub>1</sub>, <italic>&#x2026;</italic>, <italic>m<sub>n</sub>
</italic>). The <italic>m</italic> lineages infecting a host are randomly sampled (with replacement) from the pathogen population. Therefore, within an infection, the configuration of pathogen lineages <bold>
<italic>m</italic>
</bold> follows a multinomial distribution with parameters <italic>m</italic> = |<bold>
<italic>m</italic>
</bold>| and <bold>
<italic>p</italic>
</bold>, i.e., <bold>
<italic>m</italic>
</bold> &#x223c; Multi (<italic>m,<bold>p</bold>
</italic>). Hence, given that a host has MOI <italic>m</italic>, the probability of configuration <bold>
<italic>m</italic>
</bold>, i.e., of being infected <italic>m<sub>i</sub>
</italic> times with lineage <italic>A<sub>i</sub>
</italic> (<italic>i</italic> = 1,<italic>&#x2026;</italic>, <italic>n</italic>), is:</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo stretchy="false">[</mml:mo>
<mml:mtext mathvariant="bold-italic">m</mml:mtext>
<mml:mo>|</mml:mo>
<mml:mtext>MOI</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo stretchy="false">]</mml:mo>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>!</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>!</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
<mml:mo>!</mml:mo>
</mml:mrow>
</mml:mfrac>
<mml:msubsup>
<mml:mi>p</mml:mi>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2026;</mml:mo>
<mml:msubsup>
<mml:mi>p</mml:mi>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mover>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:msup>
<mml:mi>p</mml:mi>
<mml:mi>m</mml:mi>
</mml:msup>
</mml:mstyle>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mover>
<mml:mi mathvariant="bold-italic">m</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>:</mml:mo>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>!</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>!</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
<mml:mo>!</mml:mo>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula> in <xref ref-type="disp-formula" rid="eq1">Equation 1</xref> is the multinomial coefficient and <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:mstyle mathvariant="bold-italic">
<mml:msup>
<mml:mi>p</mml:mi>
<mml:mi>m</mml:mi>  </mml:msup>
</mml:mstyle>
<mml:mo>:</mml:mo>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mi>p</mml:mi>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2026;</mml:mo>
<mml:msubsup>
<mml:mi>p</mml:mi>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>The configuration of infecting pathogen lineages (<bold>
<italic>m</italic>
</bold>) and even MOI (<italic>m</italic> = |<bold>
<italic>m</italic>
</bold>|) is unobservable. Specifically, from a blood sample of an infected person, the observation is limited to the absence/presence of the infecting lineages (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>). Formally, we represent an observation as the vector <bold>
<italic>x</italic>
</bold> = (<italic>x<sub>i</sub>
</italic>)<italic>, i</italic> = 1,<italic>&#x2026;</italic>, <italic>n</italic>, such that the entries <italic>x<sub>i</sub>
</italic> are 0 or 1 (formally, <italic>x<sub>i</sub> </italic>&#x2208; {0,1}), where 0 denotes the absence and 1 denotes the presence of the lineages in the infection (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>). Thus, <italic>x<sub>i</sub>
</italic> is the sign of <italic>m<sub>i</sub>
</italic>, i.e., <italic>x<sub>i</sub>
</italic> = 0 if <italic>m<sub>i</sub>
</italic> = 0 and <italic>x<sub>i</sub>
</italic> = 1 if <italic>m<sub>i</sub>
</italic>&#x2265; 1 (formally, <italic>x<sub>i</sub>
</italic> := sign <italic>m<sub>i</sub>
</italic>, or in compact notation <bold>
<italic>x</italic>
</bold> = sign <bold>
<italic>m</italic>
</bold>).</p>
<p>Because we consider only disease-positive samples, the observations <bold>
<italic>x</italic>
</bold> correspond to vectors of length <italic>n</italic> with entries 0 and 1 excluding the vector that contains only zeros, <bold>0</bold>, which corresponds to a disease-negative sample. In mathematical notation, <bold>
<italic>x</italic>
</bold> are elements of the set &#x1d4aa; := {0,1}<italic>
<sup>n</sup>
</italic> \ {<bold>0</bold>}.</p>
<p>The underlying assumption is that molecular/genetic methods are not quantifying the concentration of lineages but rather detect their presence. Here it is ignored that lineages remain undetected. While this can be included in a statistical model (see, e.g., <xref ref-type="bibr" rid="B14">Hashemi and Schneider, 2024</xref>), here it is ignored. For more discussion on undetected or erroneously detected variants, see <xref ref-type="bibr" rid="B42">Schneider et&#xa0;al. (2022)</xref>.</p>
<p>To express the probability of <bold>
<italic>x</italic>
</bold>, further notation is needed. We call an observation <bold>
<italic>y</italic>
</bold> &#x2208; &#x1d4aa; a sub-observation of <bold>
<italic>x</italic>
</bold> (denoted <bold>
<italic>y</italic>
</bold> &#x227c; <bold>
<italic>x</italic>
</bold>); all lineages observed in <bold>
<italic>y</italic>
</bold> are also observed in <bold>
<italic>x</italic>
</bold>, i.e., if <italic>y<sub>i</sub>
</italic> &#x2264; <italic>x<sub>i</sub>
</italic> for <italic>i</italic> = 1,<italic>&#x2026;</italic>, <italic>n</italic>. The set of all sub-observations of <bold>
<italic>x</italic>
</bold> is denoted by:</p>
<disp-formula>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">A</mml:mi>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:msub>
<mml:mo>:</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo>{</mml:mo>
<mml:mi mathvariant="bold-italic">y</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">O</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo>|</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi mathvariant="bold-italic">y</mml:mi>
<mml:mo>&#x227c;</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mo>}</mml:mo>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>We define <italic>&#x3ba;<sub>m</sub>
</italic> := <italic>P</italic>[MOI = <italic>m</italic>] as the probability that a host is infected exactly <italic>m</italic> times (MOI = <italic>m</italic>) and collectively denote the MOI distribution by <bold>
<italic>&#x3ba;</italic>
</bold> : = (<italic>&#x3ba;</italic>
<sub>1</sub>
<italic>,&#x3ba;</italic>
<sub>2</sub>,<italic>&#x2026;</italic>). Furthermore, the probability generating function (PGF) of the MOI distribution evaluated at a point <italic>z</italic> is denoted by <italic>G</italic>(<italic>z</italic>) (see &#x201c;Probability distribution of observations&#x201d; in the <xref ref-type="supplementary-material" rid="SM1">
<bold>Appendix</bold>
</xref>); the probability of observation <bold>
<italic>x</italic>
</bold> is derived to be as:</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo stretchy="false">[</mml:mo>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
<mml:mo>|</mml:mo>
<mml:mtext mathvariant="bold-italic">&#x3b8;</mml:mtext>
<mml:mo stretchy="false">]</mml:mo>
<mml:mo>:</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mtext mathvariant="bold-italic">y</mml:mtext>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">A</mml:mi>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mtext mathvariant="bold-italic">y</mml:mtext>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mo stretchy="true">(</mml:mo>
<mml:mrow>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="true">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where we jointly denote the model parameters by <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:mtext mathvariant="bold-italic">&#x3b8;</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mtext mathvariant="bold-italic">&#x3ba;</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext mathvariant="bold-italic">p</mml:mtext>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo stretchy="false">[</mml:mo>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
<mml:mo>|</mml:mo>
<mml:mtext mathvariant="bold-italic">&#x3b8;</mml:mtext>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> for <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> to emphasize the dependency on the model parameters whenever necessary. Clearly, the probability in (<xref ref-type="disp-formula" rid="eq2">Equation 2</xref>) depends on the model parameters <bold>
<italic>p</italic>
</bold> and <bold>
<italic>&#x3ba;</italic>
</bold> (through the PGF).</p>
<p>Note that while <italic>p<sub>i</sub>
</italic>is the frequency of lineage <italic>A<sub>i</sub>
</italic>, as shown in <xref ref-type="bibr" rid="B42">Schneider et al. (2022)</xref>, its prevalence, i.e., the probability that this lineage occurs in an infection, is given by</p>
<disp-formula>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:msub>
<mml:mi>q</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>i.e., the PGF of the MOI distribution links frequency and prevalence.</p>
<p>The model parameters <bold>
<italic>&#x3b8;</italic>
</bold>, i.e., the distribution of MOI and the lineage frequency distribution, can be estimated from the probabilistic model (<xref ref-type="disp-formula" rid="eq2">Equation 2</xref>). We proceed with maximum likelihood (ML) estimation.</p>
<sec id="s2_1">
<label>2.1</label>
<title>Likelihood function</title>
<p>Considering <italic>N</italic> independent observations <bold>
<italic>x</italic>
</bold>
<sup>(1)</sup>, <italic>&#x2026;</italic>, <bold>
<italic>x</italic>
</bold>
<sup>(</sup>
<italic>
<sup>N</sup>
</italic>
<sup>)</sup> from disease-positive hosts, collectively denoted as <inline-formula>
<mml:math display="inline" id="im1a">
<mml:mrow>
<mml:mi mathvariant="script">X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, the likelihood function <inline-formula>
<mml:math display="inline" id="im1b">
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> (<italic>
<bold>&#x3b8;</bold>
</italic>; <inline-formula>
<mml:math display="inline" id="im1c">
<mml:mrow>
<mml:mi mathvariant="script">X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) is given by</p>
<disp-formula>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mtext mathvariant="bold-italic">&#x3b8;</mml:mtext>
<mml:mo>;</mml:mo>
<mml:mi mathvariant="script">X</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:munderover>
<mml:mo>&#x220f;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:mi>P</mml:mi>
<mml:mo stretchy="false">[</mml:mo>
<mml:msup>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
<mml:mo>|</mml:mo>
<mml:mtext mathvariant="bold-italic">&#x3b8;</mml:mtext>
<mml:mo stretchy="false">]</mml:mo>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In practice, the same allele configuration <bold>
<italic>x</italic>
</bold> can be observed in several hosts. Let <italic>n</italic>
<bold>
<italic>
<sub>x</sub>
</italic>
</bold> be the number of times observation <bold>
<italic>x</italic>
</bold> occurs in the data. (Clearly, the total sample size <italic>N</italic> must be the sum over all <inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, i.e., <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:munder>
<mml:mo>&#x3a3;</mml:mo>
<mml:mrow>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">O</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, where the sum runs over all possible observations.) With this notation, the likelihood function can be rewritten as</p>
<disp-formula>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mtext mathvariant="bold-italic">&#x3b8;</mml:mtext>
<mml:mo>;</mml:mo>
<mml:mi mathvariant="script">X</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:munder>
<mml:mo>&#x220f;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">O</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mi>P</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
<mml:mo>|</mml:mo>
<mml:mtext mathvariant="bold-italic">&#x3b8;</mml:mtext>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>and the log-likelihood function becomes</p>
<disp-formula>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mtext mathvariant="bold-italic">&#x3b8;</mml:mtext>
<mml:mo>;</mml:mo>
<mml:mi mathvariant="script">X</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">O</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
</mml:msub>
<mml:mtext>log&#xa0;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mo stretchy="false">[</mml:mo>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
<mml:mo>|</mml:mo>
<mml:mtext mathvariant="bold-italic">&#x3b8;</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
<mml:mrow>
<mml:mo>=</mml:mo>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">O</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
</mml:msub>
<mml:mtext>log&#xa0;</mml:mtext>
<mml:mo stretchy="true">[</mml:mo>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mtext mathvariant="bold-italic">y</mml:mtext>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">A</mml:mi>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:munder>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mtext mathvariant="bold-italic">y</mml:mtext>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mo stretchy="true">(</mml:mo>
<mml:mrow>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="true">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="true">]</mml:mo>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The maximum likelihood estimate (MLE) of the true unknown parameter <inline-formula>
<mml:math display="inline" id="im8">
<mml:mover accent="true">
<mml:mtext mathvariant="bold-italic">&#x3b8;</mml:mtext>
<mml:mo>^</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> is the parameter vector that maximizes the likelihood or, equivalently, the log-likelihood function, i.e.,</p>
<disp-formula>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:mover accent="true">
<mml:mtext mathvariant="bold-italic">&#x3b8;</mml:mtext>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:munder>
<mml:mrow>
<mml:mi>arg</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:mtext mathvariant="bold-italic">&#x3b8;</mml:mtext>
</mml:munder>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mtext mathvariant="bold-italic">&#x3b8;</mml:mtext>
<mml:mo>;</mml:mo>
<mml:mi mathvariant="script">X</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:munder>
<mml:mrow>
<mml:mi>arg</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:mtext mathvariant="bold-italic">&#x3b8;</mml:mtext>
</mml:munder>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mtext mathvariant="bold-italic">&#x3b8;</mml:mtext>
<mml:mo>;</mml:mo>
<mml:mi mathvariant="script">X</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Maximizing the log-likelihood function is infeasible without further restrictions because the model parameters <bold>
<italic>&#x3b8;</italic>
</bold> are infinite-dimensional. The reason is that the distribution of MOI (<italic>&#x3ba;<sub>m</sub>
</italic>) is infinite-dimensional. However, there are several meaningful strategies to restrict oneself to a finite-dimensional parameter space. A standard strategy is to assume that the MOI distribution falls into a parametric family and is hence characterized by finitely many model parameters&#x2014;for instance, the simplest case is to assume that MOI follows a positive Poisson distribution and is hence characterized by a single parameter (cf. <xref ref-type="bibr" rid="B16">Hill and Babiker, 1995</xref>; <xref ref-type="bibr" rid="B39">Schneider and Escalante, 2014</xref>; <xref ref-type="bibr" rid="B37">Schneider, 2018</xref>; <xref ref-type="bibr" rid="B12">Hashemi and Schneider, 2021</xref>; <xref ref-type="bibr" rid="B38">Schneider, 2021</xref>; <xref ref-type="bibr" rid="B42">Schneider et al., 2022</xref>; <xref ref-type="bibr" rid="B44">Tsoungui Obama and Schneider, 2022</xref>). This, however, requires the additional assumption that infectious bites are rare and independent. A similar assumption is that MOI follows a positive negative binomial distribution and is hence characterized by two parameters (cf. <xref ref-type="bibr" rid="B16">Hill and Babiker, 1995</xref>; <xref ref-type="bibr" rid="B42">Schneider et al., 2022</xref>). The negative binomial distribution allows modeling over-dispersion in the number of infectious bites. However, since the observations will tend to look under-dispersed (because only absence/presence rather than MOI is observed), one needs to estimate the amount of over-dispersion from an additional data source. In principle, any other parametric distribution can be used.</p>
<p>In case there is no empirical argument that justifies the use of a specific parametric distribution, the MOI distribution can be just truncated by assuming a maximum MOI value <italic>M</italic>, i.e., <italic>&#x3ba;<sub>m</sub>
</italic> = 0 for <italic>m &gt; M</italic>. This is a reasonable assumption since <italic>&#x3ba;<sub>m</sub>
</italic> will be negligible for large <italic>m</italic> anyway. We denote the MOI distribution by <bold>
<italic>&#x3ba;</italic>
</bold> = (<italic>&#x3ba;</italic>
<sub>1</sub>
<italic>, &#x2026;, &#x3ba;<sub>M</sub>
</italic>). In the following, we will pursue this non-parametric approach by restricting the admissible parameter space to the set of all possible MOI distributions with MOI values 1,2<italic>, &#x2026; , M</italic> and all possible lineage frequency distributions, i.e.,</p>
<disp-formula>
<mml:math display="block" id="M10">
<mml:mrow>
<mml:mtext>&#x398;</mml:mtext>
<mml:mo>:</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3ba;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">p</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>|</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mn>0</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mi>&#x3ba;</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mtext>&#xa0;for&#xa0;</mml:mtext>
<mml:mi>m</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>M</mml:mi>
<mml:mo>,</mml:mo>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>M</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mi>&#x3ba;</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mtext>&#xa0;for&#xa0;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;and&#xa0;</mml:mtext>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">}</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>M</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>S<sub>M</sub>
</italic> and <italic>S<sub>n</sub>
</italic> denote the (<italic>M</italic>&#x2212;1)&#x2212; <italic>and</italic> (<italic>n&#x2212;1</italic>)&#x2212;dimensional simplices, respectively.</p>
<p>If there was no restriction on the maximum MOI value, the probabilistic model would be correct. The restriction renders the model to be only approximately correct&#x2014;for instance, if the true MOI distribution follows a Poisson or negative binomial distribution, MOI can be any integer. Such a distribution can only be approximated by the probabilistic model above restricted to the parameter space &#x398;. However, by choosing the maximum MOI value <italic>M</italic> that is sufficiently large, any distribution can be approximated to any level of accuracy (cf. also <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>).</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Illustration of simulated data: Illustrated is the simulation scheme for the numerical investigations. A sample of size <italic>N</italic> (<italic>N</italic> = 3 in the illustration) is created as follows: For a given MOI distribution, e.g., Poisson distribution (illustrated in red) or the negative binomial distribution (illustrated in blue), characterized by parameters, first <italic>N</italic> MOI values are randomly drawn (left-hand side for Poisson distribution and right-hand side for negative binomial distribution). For each of the MOI values, <italic>N</italic> infections are drawn. For MOI = <italic>m</italic>, exactly <italic>m</italic> lineages are drawn randomly from the lineage distribution, with replacement (multinomial distribution). The data of sample size <italic>N</italic> is then obtained by retaining only the information of absence and presence of lineages in the infection. From the resulting data, the model parameters are estimated. First, they are estimated by the non-parametric method, which is approximately correct, as it imposes a maximum MOI value <italic>M</italic>, which neither the Poisson nor the negative binomial distribution does. Therefore, these true underlying distributions are only approximated by the model. Second, the model parameters are estimated by the Poisson model, which estimates the lineage frequency distribution and the MOI parameter <italic>&#x3bb;</italic>. This is the correct model if the true underlying distribution is actually a Poisson distribution (left-hand side), but it is incorrect if it is not (right-hand side). In the illustration, the negative binomial distribution is &#x201c;approximated&#x201d; by a Poisson distribution. The true (in practice unknown) parameter of the Poisson distribution was chosen to be <italic>&#x3bb;</italic> = 0.5, and the parameters of the negative binomial distribution (<italic>&#xb5;</italic> = 0.67, <italic>&#x3c9;</italic> = 2.08) were chosen; thus, the distribution is over-dispersed by 50%.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmala-02-1363981-g003.tif"/>
</fig>
<p>Unfortunately, the complexity of the log-likelihood function does not allow for an explicit solution of the MLE. The reason is that the derivatives of the log-likelihood function are polynomials in the lineage frequencies of degree up to <italic>M</italic>. Hence, the MLE must be derived numerically. We will further make use of the EM algorithm for this purpose.</p>
<p>The EM algorithm is derived in &#x201c;Derivation of the <italic>Q</italic>-function&#x201d; in the <xref ref-type="supplementary-material" rid="SM1">
<bold>Appendix</bold>
</xref>. In the present case, the algorithm estimates each probability <italic>&#x3ba;</italic>
<sub>1</sub> = <italic>P</italic>[MOI = 1]<italic>, &#x3ba;</italic>
<sub>2</sub> = <italic>P</italic>[MOI = 2]<italic>,&#x2026;, &#x3ba;<sub>M</sub>
</italic> = <italic>P</italic>[MOI = <italic>M</italic>] separately as well as the lineage frequencies <italic>p</italic>
<sub>1</sub>
<italic>,&#x2026;, p<sub>n</sub>
</italic>. The method is only approximately correct as we impose that <italic>M</italic> is the maximum possible MOI, i.e., <italic>&#x3ba;<sub>m</sub>
</italic> = 0 for <italic>m &gt; M</italic>.</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Numerical investigations</title>
<p>Since there is no closed form for the MLE, we investigate the performance of the ML estimator by numerical simulations for a representative set of parameters. We further compare the performance of the ML estimator with that of <xref ref-type="bibr" rid="B39">Schneider and Escalante (2014)</xref>, which assumes that the MOI follows a conditional Poisson distribution. <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref> illustrates how a data set is simulated, assuming that the true (in practice unknown) MOI distribution is either conditionally Poisson or negative binomially distributed.</p>
<p>For each choice of model parameters <bold>
<italic>&#x3b8;</italic>
</bold> = (<bold>
<italic>&#x3ba;</italic>
</bold>, <bold>
<italic>p</italic>
</bold>), we constructed <italic>K</italic> = 25, 000 data sets of sample size <italic>N</italic> = 50, 100, 200, and 300 according to the probabilistic model (<xref ref-type="disp-formula" rid="eq2">Equation 2</xref>) (see <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref> for the construction of a data set of sample size <italic>N</italic> = 3, assuming that the underlying MOI distribution is either conditionally Poisson or negative binomially distributed). More precisely, for each of the <italic>N</italic> samples, first, the MOI value <italic>m</italic> was chosen randomly according to the distribution <bold>
<italic>&#x3ba;</italic>
</bold>. For each MOI value <italic>m</italic> the MOI vector <bold>
<italic>m</italic>
</bold> was chosen randomly from a multinomial distribution with parameters <italic>m</italic> and <bold>
<italic>p</italic>
</bold>, and the corresponding observation was derived as <bold>
<italic>x</italic>
</bold> = sign <bold>
<italic>m</italic>
</bold>. This procedure was repeated <italic>K</italic> times. For a given set of parameters (<italic>N</italic>, <bold>
<italic>&#x3ba;</italic>
</bold>, <bold>
<italic>p</italic>
</bold>), this resulted in <italic>K</italic> data sets <inline-formula>
<mml:math display="inline" id="im1d">
<mml:mrow>
<mml:mi mathvariant="script">X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> <sup>(1)</sup>, &#x2026;, <inline-formula>
<mml:math display="inline" id="im1e">
<mml:mrow>
<mml:mi mathvariant="script">X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> <sup>(</sup>
<italic>
<sup>K</sup>
</italic>
<sup>)</sup>. For data set <inline-formula>
<mml:math display="inline" id="im1f">
<mml:mrow>
<mml:mi mathvariant="script">X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
<sup>(</sup>
<italic>
<sup>k</sup>
</italic>
<sup>)</sup>, we calculated the MLE <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mover accent="true">
<mml:mtext mathvariant="bold-italic">&#x3ba;</mml:mtext>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mover accent="true">
<mml:mtext mathvariant="bold-italic">p</mml:mtext>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> from the non-parametric model according to Result 1, assuming a maximum MOI of <italic>M</italic> = 6 and the MLE <inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mover accent="true">
<mml:mtext mathvariant="bold-italic">&#x3ba;</mml:mtext>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mover accent="true">
<mml:mtext mathvariant="bold-italic">p</mml:mtext>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> according to the parametric model (Poisson model) of <xref ref-type="bibr" rid="B39">Schneider and Escalante (2014)</xref> using the implementation of <xref ref-type="bibr" rid="B37">Schneider (2018)</xref>.</p>
<p>Let <italic>&#x3b8;</italic> denote a component of the parameter vector <bold>
<italic>&#x3b8;</italic>
</bold>. The relative bias of <italic>&#x3b8;</italic>, defined by <inline-formula>
<mml:math display="inline" id="im11">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="double-struck">E</mml:mi>
<mml:mo stretchy="false">[</mml:mo>
<mml:mover accent="true">
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mo stretchy="false">]</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula>, was approximated by</p>
<disp-formula id="eq3a">
<label>(3A)</label>
<mml:math display="block" id="M11">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where</p>
<disp-formula id="eq3b">
<label>(3B)</label>
<mml:math display="block" id="M12">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>K</mml:mi>
</mml:mfrac>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>K</mml:mi>
</mml:munderover>
<mml:msup>
<mml:mover accent="true">
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>i.e., the expectation <inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:mi mathvariant="double-struck">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mover accent="true">
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> which cannot be calculated because the MLE that has no closed form was approximated by the empirical mean over the <italic>K-</italic>simulated data sets.</p>
<p>Similarly, the variability of the estimator relative to the true parameters was assessed by the coefficient of variation (CV), i.e., as</p>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M13">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>K</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mover accent="true">
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mfrac>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The bias and variance for the parametric estimator were calculated in the same way with the necessary modifications.</p>
<p>The bias and variance of rare lineages might be substantial. However, the estimates of rare lineages that will be unlikely observed in practice have limited relevance. Therefore, we focus on reporting the bias and variance of the predominant lineage, i.e., of the largest lineage frequency, which is in practice an important quantity. Similarly, concerning the distribution of MOI <bold>
<italic>&#x3ba;</italic>
</bold>, reporting bias and variance of small <italic>&#x3ba;<sub>m</sub>
</italic> are not meaningful. However, also reporting on the MOI value with the highest frequency is not meaningful. Summary statistics such as the average MOI <inline-formula>
<mml:math display="inline" id="im13">
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
<mml:mo>=</mml:mo>
<mml:munderover>
<mml:mo>&#x3a3;</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>M</mml:mi>
</mml:munderover>
<mml:mi>m</mml:mi>
<mml:msub>
<mml:mi>&#x3ba;</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are of practical interest. The average MOI is not a model parameter but can be readily estimated by the plug-in estimator.</p>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M14">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x3c8;</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>M</mml:mi>
</mml:munderover>
<mml:mi>m</mml:mi>
<mml:msub>
<mml:mover accent="true">
<mml:mi>&#x3ba;</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In the case of the parametric model, conditional Poisson model, the average MOI is estimated as <inline-formula>
<mml:math display="inline" id="im14">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x3c8;</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mover accent="true">
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula>. Here we report on the bias and variance of the average MOI estimated by the respective plug-in estimators.</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Parameter choice</title>
<p>Besides the choices of <italic>K</italic> = 25, 000, sample sizes <italic>N</italic> = 50, 100, 200, and 300, as well as maximum MOI <italic>M</italic> = 6, we chose the model parameters <bold>
<italic>p</italic>
</bold> and <bold>
<italic>&#x3ba;</italic>
</bold> as follows.</p>
<p>Concerning the lineage frequencies for <italic>n</italic> = 4 and <italic>n</italic> = 5 lineages, we chose the balanced and unbalanced frequency distributions (reported in <xref ref-type="table" rid="T1">
<bold>Tables&#xa0;1</bold>
</xref>, <xref ref-type="table" rid="T2">
<bold>2</bold>
</xref>). By balanced frequency distributions, we refer to instances in which each lineage has approximately the same frequency, whereas we refer to unbalanced distributions if there are one or more dominating lineages and one or more rare lineages.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Choice of <italic>n</italic> = 4 lineages and their corresponding balanced and unbalanced frequency distributions.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="bottom" align="left">Lineages</th>
<th valign="bottom" align="center">
<italic>A</italic>
<sub>1</sub>
</th>
<th valign="bottom" align="center">
<italic>A</italic>
<sub>2</sub>
</th>
<th valign="bottom" align="center">
<italic>A</italic>
<sub>3</sub>
</th>
<th valign="bottom" align="center">
<italic>A</italic>
<sub>4</sub>
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="bottom" align="left">Balanced distribution</td>
<td valign="bottom" align="left">0.25</td>
<td valign="bottom" align="left">0.25</td>
<td valign="bottom" align="left">0.25</td>
<td valign="bottom" align="left">0.25</td>
</tr>
<tr>
<td valign="bottom" align="left">Unbalanced distribution</td>
<td valign="bottom" align="left">0.70</td>
<td valign="bottom" align="left">0.20</td>
<td valign="bottom" align="left">0.07</td>
<td valign="bottom" align="left">0.03</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Choice of <italic>n</italic> = 5 lineages and their corresponding balanced and unbalanced frequency distributions.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="bottom" align="left">Lineages</th>
<th valign="bottom" align="center">
<italic>A</italic>
<sub>1</sub>
</th>
<th valign="bottom" align="center">
<italic>A</italic>
<sub>2</sub>
</th>
<th valign="bottom" align="center">
<italic>A</italic>
<sub>3</sub>
</th>
<th valign="bottom" align="center">
<italic>A</italic>
<sub>4</sub>
</th>
<th valign="bottom" align="center">
<italic>A</italic>
<sub>5</sub>
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="bottom" align="left">Balanced distribution</td>
<td valign="bottom" align="left">0.20</td>
<td valign="bottom" align="left">0.20</td>
<td valign="bottom" align="left">0.20</td>
<td valign="bottom" align="left">0.20</td>
<td valign="bottom" align="left">0.20</td>
</tr>
<tr>
<td valign="bottom" align="left">Unbalanced distribution</td>
<td valign="bottom" align="left">0.60</td>
<td valign="bottom" align="left">0.20</td>
<td valign="bottom" align="left">0.14</td>
<td valign="bottom" align="left">0.05</td>
<td valign="bottom" align="left">0.01</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Concerning the distribution of MOI (<bold>
<italic>&#x3ba;</italic>
</bold>), we assumed either a conditional Poisson distribution with parameter <italic>&#x3bb;</italic> ranging from 0.1 to 2.9 in steps of 0.1 or a conditional negative binomial distribution with different degrees of over-dispersion. The parameters of the conditional negative binomial distribution were chosen such that the mean MOI matched those of the conditional Poisson distributions. For the Poisson distribution, the mean equals the variance (this is no longer true for the conditional Poisson distribution). For the negative binomial distribution with a scale parameter (<italic>&#x3c9;</italic>) and shape parameter (<italic>&#xb5;</italic>), the variance is larger than the mean, i.e., it is over-dispersed. The amount of over-dispersion is</p>
<disp-formula>
<mml:math display="block" id="M15">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mi>&#x3bc;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo stretchy="false">/</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>&#x3bc;</mml:mi>
</mml:mfrac>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Regarding the parameter choices, we first chose <inline-formula>
<mml:math display="inline" id="im15">
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>&#x3bc;</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula> as 1.05, 1.50, and 2, corresponding to 5%, 50%, and 100% over-dispersion. Then, we numerically matched the scale parameters such that the mean MOI of the corresponding conditional negative binomial distribution equals that of the conditional Poisson distribution. As an example, the conditional Poisson distribution in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref> is <italic>&#x3bb;</italic> = 0.9, and the corresponding negative binomial distribution is over-dispersed by 50%. Notably, it is impossible to find matching distributions if the mean MOI is too small. In other words, over-dispersion requires a sufficiently large mean MOI.</p>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Model implementation</title>
<p>The statistical model is implemented in R (<xref ref-type="bibr" rid="B34">R Core Team, 2023</xref>) and available in the <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Materials</bold>
</xref> and on GitHub <ext-link ext-link-type="uri" xlink:href="https://github.com/Maths-against-Malaria/Non-parametric-MOI-estimation">https://github.com/Maths-against-Malaria/Non-parametric-MOI-estimation</ext-link>
<ext-link ext-link-type="uri" xlink:href="https://github.com/Maths-against-Malaria/Non-parametric-MOI-estimation">
</ext-link>, alongside a user-friendly documentation.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results</title>
<p>First, it is shown how the maximum likelihood estimate (MLE) is derived. This is followed by results on the estimator&#x2019;s performance in terms of bias and variance.</p>
<sec id="s3_1">
<label>3.1</label>
<title>Deriving the MLE</title>
<p>The specific form of the likelihood function does not allow obtaining an explicit solution for the MLE. Specifically, assuming a maximum MOI of <italic>M</italic>, the derivatives of the likelihood function are polynomials in the model parameters of degree <italic>M</italic> &#x2212; 1, for which no general solution of the roots exists. However, the MLE can be easily calculated numerically from the EM algorithm, which is derived in the <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Materials</bold>
</xref> [expectation maximization (EM) algorithm]. The EM algorithm provides a numerically stable and efficient iteration to calculate the MLE.</p>
<p>RESULT 1. <italic>Assume molecular information from N samples</italic> <bold>
<italic>x</italic>
</bold>
<sup>(1)</sup>, <italic>&#x2026;</italic>, <bold>
<italic>x</italic>
</bold>
<sup>(</sup>
<italic>
<sup>N</sup>
</italic>
<sup>)</sup>
<italic>. Furthermore, assume a maximum MOI value M. The MLE of the lineage frequency distribution</italic> <bold>
<italic>p</italic>
</bold> = (<italic>p</italic>
<sub>1</sub>
<italic>,&#x2026;,p<sub>n</sub>
</italic>) <italic>and the distribution of MOI</italic> <bold>
<italic>&#x3ba;</italic>
</bold> = (<italic>&#x3ba;</italic>
<sub>1</sub>,<italic>&#x2026;</italic>, <italic>&#x3ba;<sub>M</sub>
</italic>)<italic>, is calculated by the EM algorithm by performing the following steps:</italic>
</p>
<p>1. <italic>Choose arbitrary initial conditions</italic> <bold>
<italic>p</italic>
</bold>
<sup>(0)</sup> <italic>and</italic> <bold>
<italic>&#x3ba;</italic>
</bold>
<sup>(0)</sup>
<italic>;</italic>
</p>
<p>2. <italic>In step t</italic> + 1<italic>, update the parameter choice</italic> <bold>
<italic>p</italic>
</bold>
<sup>(</sup>
<italic>
<sup>t</sup>
</italic>
<sup>)</sup> <italic>and</italic> <bold>
<italic>&#x3ba;</italic>
</bold>
<sup>(</sup>
<italic>
<sup>t</sup>
</italic>
<sup>)</sup> <italic>by</italic>
</p>
<disp-formula>
<mml:math display="block" id="M16">
<mml:mrow>
<mml:msubsup>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mi>T</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msubsup>
<mml:mi>T</mml:mi>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>
<italic>and</italic>
</p>
<disp-formula>
<mml:math display="block" id="M17">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3ba;</mml:mi>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mi>R</mml:mi>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>M</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msubsup>
<mml:mi>R</mml:mi>
<mml:mi>u</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>
<italic>where</italic>
</p>
<disp-formula>
<mml:math display="block" id="M18">
<mml:mrow>
<mml:msubsup>
<mml:mi>T</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">O</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo stretchy="false">[</mml:mo>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
<mml:mo>|</mml:mo>
<mml:msup>
<mml:mtext mathvariant="bold-italic">&#x3b8;</mml:mtext>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mfrac>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mtext mathvariant="bold-italic">y</mml:mtext>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">A</mml:mi>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mtext mathvariant="bold-italic">y</mml:mtext>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
<mml:msup>
<mml:msup>
<mml:mi>G</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mo stretchy="true">(</mml:mo>
<mml:mrow>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munder>
<mml:msubsup>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="true">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>
<italic>and</italic>
</p>
<disp-formula>
<mml:math display="block" id="M19">
<mml:mrow>
<mml:msubsup>
<mml:mi>R</mml:mi>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mi>&#x3ba;</mml:mi>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">O</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo stretchy="false">[</mml:mo>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
<mml:mo>|</mml:mo>
<mml:msup>
<mml:mtext mathvariant="bold-italic">&#x3b8;</mml:mtext>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mfrac>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mtext mathvariant="bold-italic">y</mml:mtext>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">A</mml:mi>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mtext mathvariant="bold-italic">x</mml:mtext>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mtext mathvariant="bold-italic">y</mml:mtext>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="true">(</mml:mo>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munder>
<mml:msubsup>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo stretchy="true">)</mml:mo>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>
<italic>with P</italic>[<bold>
<italic>x</italic>
</bold>|<bold>
<italic>&#x3b8;</italic>
</bold>
<sup>(</sup>
<italic>
<sup>t</sup>
</italic>
<sup>)</sup>] <italic>given by</italic> <xref ref-type="disp-formula" rid="eq2">Equation 2</xref> <italic>and G</italic>
<sup>&#x2032;</sup> <italic>being the derivative of the PGF given by</italic> (A5).</p>
<p>3. <italic>Repeat step 2 until numerical convergence, e.g.</italic>, &#x2225;<bold>
<italic>&#x3b8;</italic>
</bold>
<sup>(</sup>
<italic>
<sup>t</sup>
</italic>
<sup>+1)</sup> &#x2212; <bold>
<italic>&#x3b8;</italic>
</bold>
<sup>(</sup>
<italic>
<sup>t</sup>
</italic>
<sup>)</sup>&#x2225;&lt; <italic>&#x3f5; for some specified error threshold &#x3b5;.</italic>
</p>
<p>The EM algorithm converges fast in practice and is implemented as an R script.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Performance of the estimator</title>
<p>Only results for <italic>n</italic> = 4 lineages are presented here. The results for <italic>n</italic> = 5 lineages are similar and presented in the <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material</bold>
</xref> (<xref ref-type="supplementary-material" rid="SM1">
<bold>Appendix</bold>
</xref>, Additional results; <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figures S2&#x2013;S6</bold>
</xref>).</p>
<sec id="s3_2_1">
<label>3.2.1</label>
<title>Bias of lineage frequencies</title>
<p>The MLE of the lineage frequencies has very little bias (<xref ref-type="disp-formula" rid="eq3a">Equation 3</xref>) (<xref ref-type="fig" rid="f4">
<bold>Figures&#xa0;4A, B</bold>
</xref>, <xref ref-type="fig" rid="f5">
<bold>5A&#x2013;C</bold>
</xref>, and <xref ref-type="fig" rid="f6">
<bold>6A&#x2013;C</bold>
</xref>). Shown is only the bias of the dominant lineage (lineage 1). Bias is typically small for small average MOI, while the dominant lineage frequency tends to be overestimated if the true average MOI is large. However, bias vanishes with increasing sample size <italic>N</italic>. Bias tends to be larger for unbalanced lineage frequency distributions (cf. <xref ref-type="fig" rid="f4">
<bold>Figures&#xa0;4A, B</bold>
</xref>, <xref ref-type="fig" rid="f5">
<bold>5A&#x2013;C</bold>
</xref>, and <xref ref-type="fig" rid="f6">
<bold>6A&#x2013;C</bold>
</xref>). This is not surprising since for balanced frequency distributions all lineages are equivalent and should be present in equal amounts throughout the data. Importantly, the results are relatively robust with respect to the underlying true MOI distribution. While the bias is lowest from data generated from a conditional Poisson distribution (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>), the bias remains similar if the data is generated from a conditional negative binomial distribution (<xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5</bold>
</xref> and <xref ref-type="fig" rid="f6">
<bold>6</bold>
</xref>). However, the bias increases with increasing over-dispersion for unbalanced frequency distributions.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Relative bias and variation of MLE for pathogen lineage frequencies if MOI follows a Poisson distribution: Assumed are four lineages following the distributions in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>. <bold>(A, C)</bold> Balanced frequency distribution. <bold>(B, D)</bold> Unbalanced lineage frequency distribution. The dominant lineage frequency is shown at the top of the plot panels. The true MOI distribution in all panels follows a Poisson distribution with varying average MOI (x-axis). The panels show the relative bias <bold>(A, B)</bold> and CV <bold>(C, D)</bold> of the ML estimators of the dominant lineage frequency, based either on the non-parametric model (NP; solid lines) or the conditional Poisson model (CP; dashed lines) as functions of the true average MOI (cf. <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>) (note that the non-parametric model is only approximately correct in this case because a maximum MOI of <italic>M</italic> = 6 is assumed, while the Poisson model is correct). Colors correspond to different sample sizes.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmala-02-1363981-g004.tif"/>
</fig>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Relative bias and variation of MLE for balanced pathogen lineage frequencies if MOI follows a negative binomial distribution: similar as in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>. However, a balanced lineage frequency distribution is assumed here in each panel, and the true MOI distribution is a conditional negative binomial distribution with 5% <bold>(A, D)</bold>, 50% <bold>(B, E)</bold>, and 100% <bold>(C, F)</bold> over-dispersion. The gray-shaded areas indicate the parameter range which is impossible for a negative binomial distribution with the respective amount of over-dispersion. Note here that the non-parametric model (NP; solid lines) is still approximately correct, while the Poisson model (CP; dashed lines) is incorrect.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmala-02-1363981-g005.tif"/>
</fig>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Relative bias and variation of MLE for unbalanced pathogen lineage frequencies if MOI follows a negative binomial distribution <bold>(A&#x2013;F)</bold>: see <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>, but for an unbalanced pathogen lineage distribution.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmala-02-1363981-g006.tif"/>
</fig>
</sec>
<sec id="s3_2_2">
<label>3.2.2</label>
<title>Variation of lineage frequencies</title>
<p>Not surprisingly, the variance of the MLE for the dominating lineage frequency&#x2014;measured by the coefficient of variation (CV) (<xref ref-type="disp-formula" rid="eq4">Equation 4</xref>)&#x2014;decreases substantially with increasing sample size (<xref ref-type="fig" rid="f4">
<bold>Figures&#xa0;4C, D</bold>
</xref>, <xref ref-type="fig" rid="f5">
<bold>5D, F</bold>
</xref>, and <xref ref-type="fig" rid="f6">
<bold>6D&#x2013;F</bold>
</xref>). For higher MOI, the CV tends to decrease slightly, which is not surprising since the data contains more information on the lineages. The CV tends to be smaller for unbalanced frequency distributions (cf. <xref ref-type="fig" rid="f4">
<bold>Figures&#xa0;4C, D</bold>
</xref>, <xref ref-type="fig" rid="f5">
<bold>5D&#x2013;F</bold>
</xref>, and <xref ref-type="fig" rid="f6">
<bold>6D&#x2013;F</bold>
</xref>) because the dominating lineage is present in more samples, increasing the information about its true frequency. The results seem to be robust with respect to the underlying true model, i.e., conditional Poisson and negative binomial model (<xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5</bold>
</xref> and <xref ref-type="fig" rid="f6">
<bold>6</bold>
</xref>).</p>
</sec>
<sec id="s3_2_3">
<label>3.2.3</label>
<title>Frequency estimates by the non-parametric vs. conditional Poisson model</title>
<p>Given that MOI in infections follows a conditional Poisson distribution, the non-parametric model introduced here performs almost as good as the conditional Poisson model (<xref ref-type="bibr" rid="B39">Schneider and Escalante, 2014</xref>) (the correct model in this case). If the lineage frequency distributions are balanced, the two estimators perform equally well (<xref ref-type="fig" rid="f4">
<bold>Figures&#xa0;4A, C</bold>
</xref>). For unbalanced distributions, the correct model has only a slightly lower coefficient of variation and only for high average MOI. However, for a small sample size and high average MOI, the bias of the non-parametric model is higher (but still small) (<xref ref-type="fig" rid="f4">
<bold>Figures&#xa0;4B, D</bold>
</xref>).</p>
<p>Not surprisingly, if the true MOI distribution is over-dispersed compared to the Poisson distribution, the non-parametric model performs similarly to the conditional Poisson model if the lineage frequency distribution is balanced (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>). However, the model outperforms the conditional Poisson model if the lineage frequency distributions are unbalanced. While the variances of the two estimators are comparable, the non-parametric estimates are less biased. This is intuitive because it is not constrained to an incorrect model. In fact, the Poisson model underestimates the dominant allele frequency (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>; <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S7</bold>
</xref> for a comparison of the absolute bias). Importantly, while the relative bias decreases for the non-parametric model with increasing sample size, the opposite is observed for the conditional Poisson model (<xref ref-type="fig" rid="f6">
<bold>Figures&#xa0;6B, C</bold>
</xref>)</p>
</sec>
<sec id="s3_2_4">
<label>3.2.4</label>
<title>Bias of the estimates of average MOI</title>
<p>The bias of the average MOI (<italic>&#x3c8;</italic>) estimated by the non-parametric model is generally small (<xref ref-type="fig" rid="f7">
<bold>Figures&#xa0;7A, B</bold>
</xref>, <xref ref-type="fig" rid="f8">
<bold>8A&#x2013;C</bold>
</xref>, and <xref ref-type="fig" rid="f9">
<bold>9A&#x2013;C</bold>
</xref>) if the true average MOI is low to intermediate. There is a tendency for the average MOI to be overestimated for most of the parameters explored, with the bias decreasing with increasing sample size. Only for large average MOI does the true parameter tend to be underestimated. This underestimation is more pronounced for more over-dispersed MOI distributions (cf. <xref ref-type="fig" rid="f7">
<bold>Figures&#xa0;7A, B</bold>
</xref>, <xref ref-type="fig" rid="f8">
<bold>8A</bold>
</xref>, and <xref ref-type="fig" rid="f9">
<bold>9A</bold>
</xref> with <xref ref-type="fig" rid="f8">
<bold>Figures&#xa0;8B, C</bold>
</xref> and <xref ref-type="fig" rid="f9">
<bold>9B, C</bold>
</xref>). More precisely, the average MOI parameter is not underestimated if the MOI follows a conditional Poisson distribution for (almost) the whole range of parameters simulated, and the more over-dispersed the MOI distribution, the lower the threshold for which the average MOI is underestimated.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Relative bias and variation of the average MOI, if MOI follows a Poisson distribution: Assumed are four lineages with the distributions shown at the top of each panel [<bold>(A, C)</bold>&#x2014;balanced; <bold>(B, D)</bold>&#x2014;unbalanced distributions]. The true MOI distribution in all panels follows a Poisson distribution with varying average MOI (x-axis). The panels show the relative bias <bold>(A, B)</bold> and CV <bold>(C, D)</bold> of the ML estimators of the average MOI based either on the non-parametric model (NP; solid lines) or the conditional Poisson model (CP; dashed lines) as functions of the true average MOI (cf. <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>) (note that the non-parametric model is only approximately correct in this case because a maximum MOI of <italic>M</italic> = 6 is assumed, while the Poisson model is correct). Colors correspond to different sample sizes.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmala-02-1363981-g007.tif"/>
</fig>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Relative bias and variation of the average MOI for balanced lineage frequencies if MOI follows a negative binomial distribution: similar as in <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>, but the true MOI distribution follows a negative binomial distribution with 5% <bold>(A, D)</bold>, 50% <bold>(B, E)</bold>, and 100% <bold>(C, F)</bold> over-dispersion and average MOI given on the x-axes. The corresponding model parameter <italic>&#xb5;</italic> is given at the top of each panel. All panels assume the same true balanced lineage frequency distribution (top of each panel). Note that the non-parametric model is approximately correct, while the Poisson model is incorrect.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmala-02-1363981-g008.tif"/>
</fig>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Relative bias and variation of the average MOI for unbalanced lineage frequencies if MOI follows a negative binomial distribution <bold>(A&#x2013;F)</bold>: see <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref> but for an unbalanced lineage frequency distribution (top of each panel).</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmala-02-1363981-g009.tif"/>
</fig>
</sec>
<sec id="s3_2_5">
<label>3.2.5</label>
<title>Variation of the estimates of average MOI</title>
<p>The variation of the estimator of the average MOI (<italic>&#x3c8;</italic>) (<xref ref-type="disp-formula" rid="eq5">Equation 5</xref>) measured by the CV increases with increasing true average MOI (<xref ref-type="fig" rid="f7">
<bold>Figures&#xa0;7C, D</bold>
</xref>, <xref ref-type="fig" rid="f8">
<bold>8D&#x2013;F</bold>
</xref>, and <xref ref-type="fig" rid="f9">
<bold>9D&#x2013;F</bold>
</xref>). The reason is that the variance of the underlying true MOI distribution is increasing. Moreover, the CV decreases substantially for larger sample sizes (<italic>N</italic>).</p>
<p>Notably, the CV tends to be smaller for a balanced lineage frequency distribution (cf. <xref ref-type="fig" rid="f7">
<bold>Figures&#xa0;7C</bold>
</xref>, <xref ref-type="fig" rid="f8">
<bold>8D&#x2013;F</bold>
</xref>, and <xref ref-type="fig" rid="f9">
<bold>9D&#x2013;F</bold>
</xref>). This is not surprising since the occurrence of different lineages within a sample is more likely for a balanced lineage frequency distribution. Hence, the data tends to harbor more accurate information on the MOI distribution. Particularly, each lineage tends to be represented similarly in the data, thereby reducing the variability of the data. The CV is insensitive to the amount of over-dispersion in the MOI distribution (cf. <xref ref-type="fig" rid="f7">
<bold>Figures&#xa0;7C</bold>
</xref> and <xref ref-type="fig" rid="f8">
<bold>8D&#x2013;F</bold>
</xref> as well as <xref ref-type="fig" rid="f7">
<bold>Figures&#xa0;7D</bold>
</xref> and <xref ref-type="fig" rid="f9">
<bold>9D&#x2013;F</bold>
</xref>).</p>
</sec>
<sec id="s3_2_6">
<label>3.2.6</label>
<title>Average MOI estimated by the non-parametric model vs. the conditional Poisson model</title>
<p>Assuming that MOI follows a conditional Poisson distribution, the non-parametric model performs similarly as the conditional Poisson model in terms of bias and variance (i.e., the correct model in this case; cf. <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>). However, the variance of the conditional Poisson model tends to be slightly lower than that of the non-parametric model, particularly for unbalanced frequency distributions. However, the differences vanish with increasing sample size. For small and large true average MOI values, the bias of the estimates is lower for the non-parametric model (<xref ref-type="fig" rid="f7">
<bold>Figures&#xa0;7A, B</bold>
</xref>). The same holds true if the true MOI distribution is slightly over-dispersed (<xref ref-type="fig" rid="f8">
<bold>Figures&#xa0;8A</bold>
</xref>, <xref ref-type="fig" rid="f9">
<bold>9A</bold>
</xref>). For highly over-dispersed MOI distributions, the non-parametric model still has a similar variance as the Poisson model, but bias behaves differently. For intermediate average MOI, the Poisson model tends to underestimate the true parameter, with the undesirable property of higher bias for larger sample sizes. For larger average MOI, the Poisson model tends to overestimate the true parameter by roughly the same amount by which the non-parametric model underestimates this parameter (<xref ref-type="fig" rid="f8">
<bold>Figures&#xa0;8B, C</bold>
</xref> and <xref ref-type="fig" rid="f9">
<bold>9B, C</bold>
</xref>).</p>
</sec>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Data application</title>
<p>In this section, we apply the non-parametric model introduced here and the alternative Poisson model to three empirical data sets from Cameroon, Kenya, and Venezuela collected during drug resistance studies. The data from Cameroon is described in <xref ref-type="bibr" rid="B21">McCollum et al. (2007)</xref> and consists of <italic>N</italic> = 166 <italic>P. falciparum</italic> samples collected in Yaound&#xe9; between 2001 and 2002. The samples were collected randomly from patients older than 12 years of age at the Nlongkak Catholic missionary dispensary in Yaound&#xe9;, Cameroon. The data contains information on 14 neutral microsatellite markers on chromosomes 2 and 3 (for details, see <xref ref-type="bibr" rid="B20">McCollum et&#xa0;al., 2008</xref>). At that time, the sampling location was an area of intense and perennial transmission. Although the data consists of random samples, the sampling point and inclusion criteria render the population rather homogeneous. Given the results from the numerical investigations, a good agreement between the two alternative statistical models is expected.</p>
<p>The data set from Kenya is described in <xref ref-type="bibr" rid="B22">McCollum et al. (2012)</xref>. It contains molecular information from nine neutral microsatellite markers at chromosomes 2 and 3 from <italic>N</italic> = 43 <italic>P. falciparum</italic> positive samples collected in Asembo Bay, Kenya, a holoendemic <italic>P. falciparum</italic> transmission region across 15 villages between April 1992 and March 1993 (see <xref ref-type="bibr" rid="B22">McCollum et&#xa0;al., 2012</xref>). Although we would expect transmission to be heterogeneous in this setting because of the small sample size, it is expected that the Poisson model performs similar or even slightly better than the non-parametric model.</p>
<p>The data from Venezuela consists of <italic>N</italic> = 97 samples collected from 2003 to 2004 in Sifontes municipality in Bolivar State, Venezuela and is described in <xref ref-type="bibr" rid="B21">McCollum et al. (2007)</xref>. The study area has a population size below 40,000 then and was the epicenter of multi-drug resistance in Venezuela, which accounted for a large proportion of malaria infections in Venezuela (<xref ref-type="bibr" rid="B21">McCollum et&#xa0;al., 2007</xref>). However, at the time the samples were collected, the study area was an area of low transmission. Here we included five microsatellite markers, namely, those from the original data that showed evidence of super-infections in at least one sample. Given the low transmission intensity in combination with the sample size (cf. <xref ref-type="fig" rid="f7">
<bold>Figures&#xa0;7A, B</bold>
</xref>), we expect the non-parametric model and the Poisson model to give very similar results.</p>
<p>The maximum likelihood estimate for the MOI distribution and allele frequency distributions were calculated for each molecular marker separately with the non-parametric model and the conditional Poisson model of (<xref ref-type="bibr" rid="B16">Hill and Babiker, 1995</xref>; <xref ref-type="bibr" rid="B39">Schneider and Escalante, 2014</xref>). These estimates were used as plug-ins to calculate estimates for the average MOI and heterozygosity. These estimates are reported in <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref> alongside 95% bootstrap confidence intervals (see <xref ref-type="bibr" rid="B7">Efron and Tibshirani, 1994</xref>).</p>
<fig id="f10" position="float">
<label>Figure&#xa0;10</label>
<caption>
<p>Data application: Shown are the estimated heterozygosity <bold>(A, B)</bold> and average MOI <bold>(C, D)</bold> by the non-parametric model (NP) and the conditional Poisson model (CP) with 95% bootstrap CIs for each molecular marker in the data set from Cameroon <bold>(A, D)</bold>, Kenya <bold>(B, E)</bold>, and Venezuela <bold>(C, F)</bold>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmala-02-1363981-g010.tif"/>
</fig>
<p>As expected from the above-mentioned considerations and seen from <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>, both methods yield very similar results. There is a tendency for the non-parametric model to have slightly lower point estimates of average MOI, particularly for the data set from Kenya, which has a small sample size. This observation is not surprising given the numerical investigations reported above. There are hardly any differences in the estimates of heterozygosity.</p>
<p>Note that the bootstrap confidence intervals do not have satisfying properties, particularly for the data from Venezuela (which hardly has indications of super-infections) and marker u7 in the Kenya data (which also hardly has indications of super-infections). Specifically, the lower confidence points are at 1, which is the minimum possible MOI value. The reason is that the bootstrap frequently repeats samples that have no signs of super-infections, such that the average MOI is estimated to be 1 by both methods. Hence, the nominal coverage of the lower confidence point does not coincide with the actual coverage. This can be resolved by using bias-corrected and accelerated bootstrap confidence intervals or profile-likelihood confidence intervals (cf. <xref ref-type="bibr" rid="B39">Schneider and Escalante, 2014</xref>).</p>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<label>4</label>
<title>Discussion</title>
<p>Estimating the multiplicity of infection (MOI) became popular in malaria molecular surveillance. <italic>Ad hoc</italic> methods are generally biased and have undesirable statistical properties. Methods based on probabilistic models are often based on similar assumptions. A popular assumption in many models is that MOI follows a Poisson distribution. This corresponds to the assumption that infectious events are rare and independent. (Remember, MOI is defined here as the number of super-infections within one disease episode.) Because there is empirical evidence that mosquito biting rates are over-dispersed, the Poisson assumption is challenged.</p>
<p>Here we explored the most flexible alternative, namely, the situation where no parametric assumption about the distribution of MOI is made (except that a maximum MOI exists). Note that the statistical model itself makes a number of simplifying assumptions. First, the duration of how long individuals are infected is not considered. Rather, it is assumed that super-infections occur within a short period of time, such that all variants which were successfully transmitted to a host, reach sufficient frequencies in the infection to be detectable by PCR. Thus, super-infections which do not contribute to the clinical pathogenesis of an infection are ignored. Furthermore, it is assumed that molecular assays have perfect sensitivity and specificity for all relevant lineages in an infection. Notably, the statistical model can be extended to account for missing information (imperfect sensitivity) as in <xref ref-type="bibr" rid="B14">Hashemi and Schneider (2024)</xref> (for the Poisson model); however, if all samples contain molecular information (i.e., no sample with missing data), the extended model reduces to the present one. More alternatives can be found in <xref ref-type="bibr" rid="B29">Okell et al. (2017)</xref>. Extending the model to include imperfect specificity, i.e., including erroneously detected variants, is more challenging. A model that yields appropriate estimates is the one given by <xref ref-type="bibr" rid="B33">Plucinski et al. (2015)</xref>. However, it is designed to distinguish recrudescence of reinfections and hence requires paired samples (i.e., two or more sample points for at least some patients). Furthermore, it is assumed that the molecular assays used can only provide absence/presence data but cannot quantify the relative abundance of lineages within an infection.</p>
<p>Given this framework, the resulting non-parametric model is more complicated than the corresponding Poisson model (cf. <xref ref-type="bibr" rid="B39">Schneider and Escalante, 2014</xref>), which falls into the class of exponential families (<xref ref-type="bibr" rid="B13">Hashemi and Schneider, 2024</xref>). This implies the usual desirable properties of maximum likelihood estimators for the Poisson model (existence and uniqueness of the MLE, efficiency, and consistency, cf. <xref ref-type="bibr" rid="B13">Hashemi and Schneider, 2024</xref>). Unfortunately, the non-parametric model is no longer within an exponential family and there is no proof for the same desirable theoretical properties. However, our numerical investigations suggest that they hold. In the case of the Poisson model, the MLE has an intuitive interpretation. Namely, the MLE is the parameter choice for which the empirical prevalences coincide with the expected prevalences. Moreover, the empirical prevalences form a sufficient statistic for the parameter estimation. For the non-parametric model, this interpretation is lost and the prevalences no longer form a sufficient statistic. In fact, the model used more information, namely not just how often lineages occur in infections but also in which configuration they are observed. As for the Poisson model, there exists no closed solution for the MLE of the non-parametric model, and it has to be derived numerically. For this purpose, the EM algorithm was employed here. It is a numerically stable and efficient algorithm to derive the MLE.</p>
<p>Notably, the non-parametric model also has certain advantages compared to the Poisson model. If the data lies on the boundary of the closed convex hull of the admissible sample space (when rewritten as a natural exponential family), the MLE of the Poisson model is degenerate (or does not exist; <xref ref-type="bibr" rid="B13">Hashemi and Schneider, 2023</xref>). This corresponds to the cases where only a single lineage is detected in every sample (i.e., no evidence of super-infections), or if one lineage is observed in all samples. In such a situation, an MLE is still numerically found for the non-parametric model.</p>
<p>The performance of the non-parametric estimator is comparably good to that of the Poisson model even if the true MOI distribution is well approximated by a Poisson distribution. Despite the Poisson model being the true model for low true MOI, the non-parametric model yields less biased estimates of this parameter. This is also true for high average MOI, at least for a sufficiently large sample size. Moreover, if the true MOI distribution is highly over-dispersed, the non-parametric estimator is preferential. Irrespective of the differences, the results obtained by both models for the empirical examples from Cameroon, Kenya, and Venezuela are very similar. This and the fact that the differences between the estimators are small suggests that there is no need to drop the Poisson assumption, except for extreme parameters or in cases, in which there is clear evidence that the Poisson assumption is unjustified. Therefore, as a recommendation, it seems appropriate to use the non-parametric model in areas of low transmission, particularly, if transmission is highly heterogeneous. In such situations, the Poisson model is highly sensitive to outliers (<xref ref-type="bibr" rid="B37">Schneider, 2018</xref>). Areas of low transmission are becoming increasingly important with the ongoing malaria eradication efforts to decrease the global malaria burden by 90% and eradicate the disease in at least 35 countries by 2023 (<xref ref-type="bibr" rid="B46">World Health Organization, 2021</xref>). However, in the empirical example from Venezuela, at a time of low transmission, both methods yielded comparable results. In the example, this was due to the relatively large sample size (<italic>N</italic> = 97). In practice, molecular surveillance data might not always appropriately reflect the heterogeneity in transmission but is rather conducted in homogeneous populations. This renders the Poisson model to be sufficiently accurate.</p>
<p>The results here suggest that the non-parametric approach to estimating MOI is comparably good as the Poisson model. A particularly convenient property of the Poisson distribution is that it is characterized by a single parameter. This is no longer true for the non-parametric model, where the complete probability mass function of MOI needs to be estimated. Although there are many other parametric alternatives to the Poisson model, every other meaningful distribution is characterized by more than a single parameter. Moreover, from the structure outlined in <xref ref-type="bibr" rid="B44">Tsoungui Obama and Schneider (2022)</xref>, such alternatives also do not retain the desirable properties of the Poisson model.</p>
<p>Given the data, i.e., the absence and presence of alleles in samples, the non-parametric model, is the most flexible assumption for MOI, since it does not require a certain distribution. In principle, the model should be able to approximate any parametric distribution, such as the conditional Poisson or conditional negative binomial distributions. Given that the non-parametric model does not clearly outperform the Poisson model implies that the latter already utilizes the information of data effectively, and any other model assumptions regarding over-dispersion (e.g., assuming a negative binomial distribution) will not outperform the Poisson model either. However, the statistical model can be extended to include further information in addition to the original data&#x2014;for instance, one could extend the non-parametric model to account for patient-specific risks (e.g., one could group patients into different risk strata based on patient queries determining the risk of exposure, such as the use of bed nets, indoor residual spraying, window screens, etc., or recent malaria cases within the household). In the case of the Poisson model, it could be extended to estimate different MOI parameters for each stratum (risk group). Notably, such information has to be collected in addition to molecular data. In any case, our results suggest that, unless transmission is very heterogeneous, it is not necessary to extend the non-parametric approach to the case in which multiple genetic markers are studies at the same time as it was done for the Poisson model (e.g. <xref ref-type="bibr" rid="B44">Tsoungui Obama and Schneider, 2022</xref>; <xref ref-type="bibr" rid="B28">Obama and Schneider, 2023</xref>).</p>
<p>In summary, our results suggest that the non-parametric model to estimate MOI and lineage frequencies from single molecular markers (e.g., microsatellites, SNPs, and micro-haplotypes) introduced here is a valid approach. An implementation of the method as an R script is available in the <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Materials</bold>
</xref> and at GitHub <ext-link ext-link-type="uri" xlink:href="https://github.com/Maths-against-Malaria/Non-parametric-MOI-estimation">https://github.com/Maths-against-Malaria/Non-parametric-MOI-estimation</ext-link>
<ext-link ext-link-type="uri" xlink:href="https://github.com/Maths-against-Malaria/Non-parametric-MOI-estimation">.</ext-link> Clearly, it is possible to extend the non-parametric model to more complex genetic architectures than a single molecular marker following the methodology in <xref ref-type="bibr" rid="B44">Tsoungui Obama and Schneider (2022)</xref>.</p>
</sec>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material</bold>
</xref>. Further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s6" sec-type="author-contributions">
<title>Author contributions</title>
<p>LK: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Software, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. KS: Conceptualization, Formal analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Software, Supervision, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing.</p>
</sec>
</body>
<back>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This work was also supported by grants of the German Academic Exchange (DAAD; <ext-link ext-link-type="uri" xlink:href="https://www.daad.de/de/">https://www.daad.de/de/</ext-link>
<ext-link ext-link-type="uri" xlink:href="https://www.daad.de/de/">;</ext-link> Project-ID 57417782, Project-ID: 57599539), and the Federal Ministry of Education and Research (BMBF) and the DLR (Project-ID 01DQ20002; <ext-link ext-link-type="uri" xlink:href="https://www.bmbf.de/">https://www.bmbf.de/</ext-link>
<ext-link ext-link-type="uri" xlink:href="https://www.bmbf.de/">;</ext-link> <ext-link ext-link-type="uri" xlink:href="https://www.dlr.de/">https://www.dlr.de/</ext-link>
<ext-link ext-link-type="uri" xlink:href="https://www.dlr.de/">
</ext-link>).</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>The authors want to thank Andrea M. McCollum, Dr. Ananias Escalante, and Dr. Venkatachalam (Kumar) Udhayakumar for sharing the data sets from Cameroon, Kenya, and Venezuela. The constructive comments of two reviewers are gratefully acknowledged.</p>
</ack>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s9" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s10" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fmala.2024.1363981/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fmala.2024.1363981/full#supplementary-material</ext-link>.</p>
<supplementary-material xlink:href="DataSheet_1.zip" id="SM1" mimetype="application/zip"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Adamidis</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>1999</year>). <article-title>Theory &amp; methods: An em algorithm for estimating negative binomial parameters</article-title>. <source>Aust. New Z. J. Stat.</source> <volume>41</volume>, <fpage>213</fpage>&#x2013;<lpage>221</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/1467-842X.00075</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alizon</surname> <given-names>S.</given-names>
</name>
<name>
<surname>de Roode</surname> <given-names>J. C.</given-names>
</name>
<name>
<surname>Michalakis</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Multiple infections and the evolution of virulence</article-title>. <source>Ecol. Lett.</source> <volume>16</volume>, <fpage>556</fpage>&#x2013;<lpage>567</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/ele.12076</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bandara</surname> <given-names>U.</given-names>
</name>
<name>
<surname>Gill</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Mitra</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>On computing maximum likelihood estimates for the negative binomial distribution</article-title>. <source>Stat Probability Lett.</source> <volume>148</volume>, <fpage>54</fpage>&#x2013;<lpage>58</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.spl.2019.01.009</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chang</surname> <given-names>H. H.</given-names>
</name>
<name>
<surname>Worby</surname> <given-names>C. J.</given-names>
</name>
<name>
<surname>Yeka</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Nankabirwa</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Kamya</surname> <given-names>M. R.</given-names>
</name>
<name>
<surname>Staedke</surname> <given-names>S. G.</given-names>
</name>
<etal/>
</person-group>. (<year>2017</year>). <article-title>THE REAL McCOIL: A method for the concurrent estimation of the complexity of infection and SNP allele frequency for malaria parasites</article-title>. <source>PloS Comput. Biol.</source> <volume>13</volume>, <fpage>1</fpage>&#x2013;<lpage>18</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pcbi.1005348</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Couvreur</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>1997</year>). &#x201c;<article-title>The em algorithm: A guided tour</article-title>,&#x201d; in <person-group person-group-type="editor">
<name>
<surname>K&#xe1;rn&#xfd;</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Warwick</surname> <given-names>K.</given-names>
</name>
</person-group> (eds) <source>Computer intensive methods in control and signal processing</source> (<publisher-loc>Birkh&#xe4;user, Boston, MA</publisher-loc>). doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-1-4612-1996-5_12</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dia</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Cheeseman</surname> <given-names>I. H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Single-cell genome sequencing of protozoan parasites</article-title>. <source>Trends Parasitol.</source> <volume>37</volume>, <fpage>803</fpage>&#x2013;<lpage>814</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.pt.2021.05.013</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Efron</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Tibshirani</surname> <given-names>R. J.</given-names>
</name>
</person-group> (<year>1994</year>). <source>An introduction to the bootstrap</source> (1st ed.) (<publisher-name>Chapman and Hall/CRC</publisher-name>). doi:&#xa0;<pub-id pub-id-type="doi">10.1201/9780429246593</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Galinsky</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Valim</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Salmier</surname> <given-names>A.</given-names>
</name>
<name>
<surname>de Thoisy</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Musset</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Legrand</surname> <given-names>E.</given-names>
</name>
<etal/>
</person-group>. (<year>2015</year>). <article-title>COIL: a methodology for evaluating malarial complexity of infection using likelihood from single nucleotide polymorphism data</article-title>. <source>Malaria J.</source> <volume>14</volume>, <elocation-id>4</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/1475-2875-14-4</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Geiger</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Compaore</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Coulibaly</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Sie</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Dittmer</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Sanchez</surname> <given-names>C.</given-names>
</name>
<etal/>
</person-group>. (<year>2014</year>). <article-title>Substantial increase in mutations in the genes pfdhfr and pfdhps puts sulphadoxine&#x2013;pyrimethamine-based intermittent preventive treatment for malaria at risk in Burkina Faso</article-title>. <source>Trop. Med. Int. Health.</source> <volume>19</volume>, <fpage>690</fpage>&#x2013;<lpage>697</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/tmi.12305</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guelbeogo</surname> <given-names>W. M.</given-names>
</name>
<name>
<surname>Gonc&#xb8;alves</surname> <given-names>B. P.</given-names>
</name>
<name>
<surname>Grignard</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Bradley</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Serme</surname> <given-names>S. S.</given-names>
</name>
<name>
<surname>Hellewell</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2018</year>). <article-title>Variation in natural exposure to anopheles mosquitoes and its effects on malaria transmission</article-title>. <source>Elife</source> <volume>7</volume>, <elocation-id>e32625</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.7554/eLife.32625</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gurarie</surname> <given-names>D.</given-names>
</name>
<name>
<surname>McKenzie</surname> <given-names>F. E.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Dynamics of immune response and drug resistance in malaria infection</article-title>. <source>Malaria J.</source> <volume>5</volume>, <elocation-id>86</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/1475-2875-5-86</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hashemi</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Schneider</surname> <given-names>K. A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Bias-corrected maximum-likelihood estimation of multiplicity of infection and lineage frequencies</article-title>. <source>PloS One</source>. <volume>16</volume>, <elocation-id>e0261889</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0261889</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hashemi</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Schneider</surname> <given-names>K. A.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Estimating multiplicity of infection, allele frequencies, and prevalences accounting for incomplete data</article-title>. <source>bioRxiv</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.1101/2023.06.01.543300</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hashemi</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Schneider</surname> <given-names>K. A.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Estimating multiplicity of infection, allele frequencies, and prevalences accounting for incomplete data</article-title>. <source>PloS One.</source> <volume>19</volume>, <fpage>1</fpage>&#x2013;<lpage>35</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0287161</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hastings</surname> <given-names>I. M.</given-names>
</name>
<name>
<surname>Watkins</surname> <given-names>W. M.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Intensity of malaria transmission and the evolution of drug resistance</article-title>. <source>Acta tropica</source> <volume>94</volume>, <fpage>218</fpage>&#x2013;<lpage>229</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.actatropica.2005.04.003</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hill</surname> <given-names>W. G.</given-names>
</name>
<name>
<surname>Babiker</surname> <given-names>H. A.</given-names>
</name>
</person-group> (<year>1995</year>). <article-title>Estimation of numbers of malaria clones in blood samples</article-title>. <source>Proc. R. Soc. London Ser. B: Biol. Sci.</source> <volume>262</volume>, <fpage>249</fpage>&#x2013;<lpage>257</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1098/rspb.1995.0203</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Irvine</surname> <given-names>M. A.</given-names>
</name>
<name>
<surname>Kazura</surname> <given-names>J. W.</given-names>
</name>
<name>
<surname>Hollingsworth</surname> <given-names>T. D.</given-names>
</name>
<name>
<surname>Reimer</surname> <given-names>L. J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Understanding heterogeneities in mosquitobite exposure and infection distributions for the elimination of lymphatic filariasis</article-title>. <source>Proc. R. Soc. B: Biol. Sci.</source> <volume>285</volume>, <fpage>20172253</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1098/rspb.2017.2253</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Foulkes</surname> <given-names>A. S.</given-names>
</name>
<name>
<surname>Yucel</surname> <given-names>R. M.</given-names>
</name>
<name>
<surname>Rich</surname> <given-names>S. M.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>An expectation maximization approach to estimate malaria haplotype frequencies in multiply infected children</article-title>. <source>Stat. Appl. Genet. Mol. Biol.</source> <volume>6</volume> Article33, <page-range>1-19</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.2202/1544-6115.1321</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lloyd-Smith</surname> <given-names>J. O.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Maximum likelihood estimation of the negative binomial dispersion parameter for highly overdispersed data, with applications to infectious diseases</article-title>. <source>PloS One.</source> <volume>2</volume>, <elocation-id>e180</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0000180</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McCollum</surname> <given-names>A. M.</given-names>
</name>
<name>
<surname>Basco</surname> <given-names>L. K.</given-names>
</name>
<name>
<surname>Tahar</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Udhayakumar</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Escalante</surname> <given-names>A. A.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Hitchhiking and selective sweeps of plasmodium falciparum sulfadoxine and pyrimethamine resistance alleles in a population from central africa</article-title>. <source>Antimicrob. Agents Chemother.</source> <volume>52</volume>, <fpage>4089</fpage>&#x2013;<lpage>4097</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1128/AAC.00623-08</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McCollum</surname> <given-names>A. M.</given-names>
</name>
<name>
<surname>Mueller</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Villegas</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Udhayakumar</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Escalante</surname> <given-names>A. A.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Common origin and fixation of Plasmodium falciparum dhfr and dhps mutations associated with sulfadoxine-pyrimethamine resistance in a low-transmission area in South America</article-title>. <source>Antimicrobial Agents chemotherapy</source> <volume>51</volume>, <fpage>2085</fpage>&#x2013;<lpage>2091</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1128/AAC.01228-06</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McCollum</surname> <given-names>A. M.</given-names>
</name>
<name>
<surname>Schneider</surname> <given-names>K. A.</given-names>
</name>
<name>
<surname>Griffing</surname> <given-names>S. M.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Kariuki</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Ter-Kuile</surname> <given-names>F.</given-names>
</name>
<etal/>
</person-group>. (<year>2012</year>). <article-title>Differences in selective pressure on dhps and dhfr drug resistant mutations in western Kenya</article-title>. <source>Malaria J.</source> <volume>11</volume>, <fpage>1</fpage>&#x2013;<lpage>14</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/1475-2875-11-77</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Neafsey</surname> <given-names>D. E.</given-names>
</name>
<name>
<surname>Taylor</surname> <given-names>A. R.</given-names>
</name>
<name>
<surname>MacInnis</surname> <given-names>B. L.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Advances and opportunities in malaria population genomics</article-title>. <source>Nat. Rev. Genet.</source> <volume>22</volume>, <fpage>502</fpage>&#x2013;<lpage>517</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41576-021-00349-5</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ng</surname> <given-names>S. K.</given-names>
</name>
<name>
<surname>Krishnan</surname> <given-names>T.</given-names>
</name>
<name>
<surname>McLachlan</surname> <given-names>G. J.</given-names>
</name>
</person-group> (<year>2012</year>). &#x201c;<article-title>The em algorithm</article-title>,&#x201d; <source>Handbook of computational statistics: concepts and methods</source>, <fpage>139</fpage>&#x2013;<lpage>172</lpage>. edited by <person-group person-group-type="author">
<name>
<surname>Gentle</surname> <given-names>J. E.</given-names>
</name>
<name>
<surname>Hardle</surname> <given-names>W. K.</given-names>
</name>
<name>
<surname>Mori</surname> <given-names>Y.</given-names>
</name>
</person-group> <publisher-loc>Berlin &amp; New York</publisher-loc>: <publisher-name>Springer</publisher-name>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-642-21551-3__6</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nkhoma</surname> <given-names>S. C.</given-names>
</name>
<name>
<surname>Nair</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Cheeseman</surname> <given-names>I. H.</given-names>
</name>
<name>
<surname>Rohr-Allegrini</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Singlam</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Nosten</surname> <given-names>F.</given-names>
</name>
<etal/>
</person-group>. (<year>2012</year>). <article-title>Close kinship within multiple-genotype malaria parasite infections</article-title>. <source>Proc. Biol. Sci.</source> <volume>279</volume>, <fpage>2589</fpage>&#x2013;<lpage>2598</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1098/rspb.2012.0113</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nkhoma</surname> <given-names>S. C.</given-names>
</name>
<name>
<surname>Trevino</surname> <given-names>S. G.</given-names>
</name>
<name>
<surname>Gorena</surname> <given-names>K. M.</given-names>
</name>
<name>
<surname>Nair</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Khoswe</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Jett</surname> <given-names>C.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Co-transmission of related malaria parasite lineages shapes within-host parasite diversity</article-title>. <source>Cell Host Microbe.</source> <volume>27</volume>, <fpage>93</fpage>&#x2013;<lpage>103.e4</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.chom.2019.12.001</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Noor</surname> <given-names>A. M.</given-names>
</name>
<name>
<surname>Kinyoki</surname> <given-names>D. K.</given-names>
</name>
<name>
<surname>Mundia</surname> <given-names>C. W.</given-names>
</name>
<name>
<surname>Kabaria</surname> <given-names>C. W.</given-names>
</name>
<name>
<surname>Mutua</surname> <given-names>J. W.</given-names>
</name>
<name>
<surname>Alegana</surname> <given-names>V. A.</given-names>
</name>
<etal/>
</person-group>. (<year>2014</year>). <article-title>The changing risk of plasmodium falciparum malaria infection in africa: 2000&#x2013;10: a spatial and temporal analysis of transmission intensity</article-title>. <source>Lancet</source> <volume>383</volume>, <fpage>1739</fpage>&#x2013;<lpage>1747</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S0140-6736(13)62566-0</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Obama</surname> <given-names>H. C. J. T.</given-names>
</name>
<name>
<surname>Schneider</surname> <given-names>K. A.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Estimating multiplicity of infection, haplotype frequencies, and linkage disequilibria from multi-allelic markers for molecular disease surveillance</article-title>. <source>bioRxiv</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.1101/2023.08.29.555251</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Okell</surname> <given-names>L. C.</given-names>
</name>
<name>
<surname>Griffin</surname> <given-names>J. T.</given-names>
</name>
<name>
<surname>Roper</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Mapping sulphadoxine-pyrimethamine-resistant plasmodium falciparum malaria in infected humans and in parasite populations in africa</article-title>. <source>Sci. Rep.</source> <volume>7</volume>, <fpage>7389</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-017-06708-9</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pacheco</surname> <given-names>M. A.</given-names>
</name>
<name>
<surname>Forero-Pe&#xf1;a</surname> <given-names>D. A.</given-names>
</name>
<name>
<surname>Schneider</surname> <given-names>K. A.</given-names>
</name>
<name>
<surname>Chavero</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Gamardo</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Figuera</surname> <given-names>L.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Malaria in Venezuela: changes in the complexity of infection reflects the increment in transmission intensity</article-title>. <source>Malaria J.</source> <volume>19</volume>, <fpage>176</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12936-020-03247-z</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pacheco</surname> <given-names>M. A.</given-names>
</name>
<name>
<surname>Lopez-Perez</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Vallejo</surname> <given-names>A. F.</given-names>
</name>
<name>
<surname>Herrera</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Ar&#xe9;valo-Herrera</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Escalante</surname> <given-names>A. A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Multiplicity of infection and disease severity in plasmodium vivax</article-title>. <source>PloS Negl. Trop. Dis.</source> <volume>10</volume>, <elocation-id>e0004355</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pntd.0004355</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Piegorsch</surname> <given-names>W. W.</given-names>
</name>
</person-group> (<year>1990</year>). <article-title>Maximum likelihood estimation for the negative binomial dispersion parameter</article-title>. <source>Biometrics</source>, <volume>46</volume>(<issue>3</issue>), <fpage>863</fpage>&#x2013;<lpage>867</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.2307/2532104</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Plucinski</surname> <given-names>M. M.</given-names>
</name>
<name>
<surname>Morton</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Bushman</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Dimbu</surname> <given-names>P. R.</given-names>
</name>
<name>
<surname>Udhayakumar</surname> <given-names>V.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Robust algorithm for systematic classification of malaria late treatment failures as recrudescence or reinfection using microsatellite genotyping</article-title>. <source>Antimicrob. Agents Chemother.</source> <volume>59</volume>, <fpage>6096</fpage>&#x2013;<lpage>6100</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1128/AAC.00072-15</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="book">
<person-group person-group-type="author">
<collab>R Core Team</collab>
</person-group> (<year>2023</year>). &#x201c;<article-title>R: A language and environment for statistical computing</article-title>,&#x201d; in <source>R foundation for statistical computing</source> (<publisher-loc>Vienna, Austria</publisher-loc>).</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Read</surname> <given-names>A. F.</given-names>
</name>
<name>
<surname>Taylor</surname> <given-names>L. H.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>The ecology of genetically diverse infections</article-title>. <source>Science</source> <volume>292</volume>, <fpage>1099</fpage>&#x2013;<lpage>1102</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1126/science.1059410</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Saha</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Paul</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Bias-corrected maximum likelihood estimator of the negative binomial dispersion parameter</article-title>. <source>Biometrics</source> <volume>61</volume>, <fpage>179</fpage>&#x2013;<lpage>185</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/j.0006-341X.2005.030833.x</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schneider</surname> <given-names>K. A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Large and finite sample properties of a maximum-likelihood estimator for multiplicity of infection</article-title>. <source>PloS One.</source> <volume>13</volume>, <elocation-id>e0194148</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0194148</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Schneider</surname> <given-names>K. A.</given-names>
</name>
</person-group> (<year>2021</year>). <source>Charles darwin meets ronald ross: A population-genetic framework for the evolutionary dynamics of malaria</source> Vol. <volume>6</volume> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>), <fpage>149</fpage>&#x2013;<lpage>191</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978&#x2013;3-030&#x2013;50826-56</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schneider</surname> <given-names>K. A.</given-names>
</name>
<name>
<surname>Escalante</surname> <given-names>A. A.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>A likelihood approach to estimate the number of co-infections</article-title>. <source>PloS One.</source> <volume>9</volume>, <elocation-id>e97899</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0097899</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schneider</surname> <given-names>K. A.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>An analytical model for genetic hitchhiking in the evolution of antimalarial drug resistance</article-title>. <source>Theor. Population Biol.</source> <volume>78</volume>, <fpage>93</fpage>&#x2013;<lpage>108</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.tpb.2010.06.005</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schneider</surname> <given-names>K. A.</given-names>
</name>
<name>
<surname>Salas</surname> <given-names>C. J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Evolutionary genetics of malaria</article-title>. <source>Front. Genet.</source> <volume>13</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fgene.2022.1030463</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schneider</surname> <given-names>K. A.</given-names>
</name>
<name>
<surname>Tsoungui Obama</surname> <given-names>H. C. J.</given-names>
</name>
<name>
<surname>Kamanga</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Kayanula</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Adil Mahmoud Yousif</surname> <given-names>N.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>The many definitions of multiplicity of infection</article-title>. <source>Front. Epidemiol.</source>, <volume>2</volume>, <elocation-id>961593</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fepid.2022.961593</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sinha</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Kar</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Deora</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Dash</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Tiwari</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Kori</surname> <given-names>L.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>India-embo lecture course: understanding malaria from molecular epidemiology, population genetics, and evolutionary perspectives</article-title>. <source>Trends Parasitol.</source> <volume>39</volume>, <fpage>307</fpage>&#x2013;<lpage>313</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.pt.2023.02.010</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tsoungui Obama</surname> <given-names>H. C. J.</given-names>
</name>
<name>
<surname>Schneider</surname> <given-names>K. A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A maximum-likelihood method to estimate haplotype frequencies and prevalence alongside multiplicity of infection from snp data</article-title>. <source>Front. Epidemiol.</source> <volume>2</volume>, <elocation-id>943625</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fepid.2022.943625</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wong</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Wenger</surname> <given-names>E. A.</given-names>
</name>
<name>
<surname>Hartl</surname> <given-names>D. L.</given-names>
</name>
<name>
<surname>Wirth</surname> <given-names>D. F.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Modeling the genetic relatedness of Plasmodium falciparum parasites following meiotic recombination and cotransmission</article-title>. <source>PloS Comput. Biol.</source> <volume>14</volume>, <elocation-id>e1005923</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pcbi.1005923</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="book">
<person-group person-group-type="author">
<collab>World Health Organization</collab>
</person-group> (<year>2021</year>). <source>Global tuberculosis report 2021</source> (<publisher-name>World Health Organization</publisher-name>).</citation>
</ref>
<ref id="B47">
<citation citation-type="book">
<person-group person-group-type="author">
<collab>World Health Organization</collab>
</person-group> (<year>2022</year>). <source>Global genomic surveillance strategy for pathogens with pandemic and epidemic potential, 2022&#x2013;2032</source> Vol. <volume>VI</volume> (<publisher-loc>Geneva</publisher-loc>: <publisher-name>World Health Organization</publisher-name>), <fpage>21</fpage>.</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname> <given-names>S. J.</given-names>
</name>
<name>
<surname>Hendry</surname> <given-names>J. A.</given-names>
</name>
<name>
<surname>Almagro-Garcia</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Pearson</surname> <given-names>R. D.</given-names>
</name>
<name>
<surname>Amato</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Miles</surname> <given-names>A.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). <article-title>The origins and relatedness structure of mixed infections vary with local prevalence of P. falciparum malaria</article-title>. <source>eLife.</source> <volume>8</volume>, <elocation-id>e40845</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.7554/eLife.40845</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>