<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">855052</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2022.855052</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>
<italic>De Novo</italic> Assembly of <italic>Plasmodium knowlesi</italic> Genomes From Clinical Samples Explains the Counterintuitive Intrachromosomal Organization of Variant <italic>SICAvar</italic> and <italic>kir</italic> Multiple Gene Family Members</article-title>
<alt-title alt-title-type="left-running-head">Oresegun et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<italic>Plasmodium knowlesi</italic> Genome Clinical Samples</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Oresegun</surname>
<given-names>Damilola R.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="fn" rid="FN1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1099874/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Thorpe</surname>
<given-names>Peter</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="fn" rid="FN1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1692342/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Benavente</surname>
<given-names>Ernest Diez</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1301109/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Campino</surname>
<given-names>Susana</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1253346/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Muh</surname>
<given-names>Fauzi</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1555272/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Moon</surname>
<given-names>Robert William</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/894191/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Clark</surname>
<given-names>Taane Gregory</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/642245/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Cox-Singh</surname>
<given-names>Janet</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1087368/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Division of Infection and Global Health</institution>, <institution>School of Medicine</institution>, <institution>University of St Andrews</institution>, <addr-line>Scotland</addr-line>, <country>United Kingdom</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Faculty of Infectious and Tropical Diseases</institution>, <institution>London School of Hygiene &#x26; Tropical Medicine</institution>, <addr-line>London</addr-line>, <country>United Kingdom</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Faculty of Epidemiology and Population Health</institution>, <institution>London School of Hygiene &#x26; Tropical Medicine</institution>, <addr-line>London</addr-line>, <country>United Kingdom</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/915243/overview">Andrew Paul Jackson</ext-link>, University of Liverpool, United Kingdom</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/471621/overview">Chenqi Wang</ext-link>, University of South Florida, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/176315/overview">Mary Rose Galinski</ext-link>, Emory University, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Janet Cox-Singh, <email>jcs26@st-andrews.ac.uk</email>
</corresp>
<fn fn-type="equal" id="FN1">
<label>
<sup>&#x2020;</sup>
</label>
<p>These authors have contributed equally to this work</p>
</fn>
<fn fn-type="other">
<p>This article was submitted to Human and Medical Genomics, a section of the journal Frontiers in Genetics</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>23</day>
<month>05</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>855052</elocation-id>
<history>
<date date-type="received">
<day>14</day>
<month>01</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>15</day>
<month>04</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Oresegun, Thorpe, Benavente, Campino, Muh, Moon, Clark and Cox-Singh.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Oresegun, Thorpe, Benavente, Campino, Muh, Moon, Clark and Cox-Singh</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>
<italic>Plasmodium knowlesi</italic>, a malaria parasite of Old World macaque monkeys, is used extensively to model <italic>Plasmodium</italic> biology. Recently, <italic>P. knowlesi</italic> was found in the human population of Southeast Asia, particularly Malaysia. <italic>P. knowlesi</italic> causes uncomplicated to severe and fatal malaria in the human host with features in common with the more prevalent and virulent malaria caused by <italic>Plasmodium falciparum</italic>. As such, <italic>P. knowlesi</italic> presents a unique opportunity to develop experimental translational model systems for malaria pathophysiology informed by clinical data from same-species human infections. Experimental lines of <italic>P. knowlesi</italic> represent well-characterized genetically stable parasites, and to maximize their utility as a backdrop for understanding malaria pathophysiology, genetically diverse contemporary clinical isolates, essentially wild-type, require comparable characterization. The Oxford Nanopore PCR-free long-read sequencing platform was used to sequence and <italic>de novo</italic> assemble <italic>P. knowlesi</italic> genomes from frozen clinical samples. The sequencing platform and assembly pipelines were designed to facilitate capturing data and describing, for the first time, <italic>P. knowlesi schizont-infected cell agglutination</italic> (<italic>SICA</italic>) <italic>var</italic> and <italic>Knowlesi-Interspersed Repeats</italic> (<italic>kir</italic>) multiple gene families in parasites acquired from nature. The <italic>SICAvar</italic> gene family members code for antigenically variant proteins analogous to the virulence-associated <italic>P. falciparum</italic> erythrocyte membrane protein (<italic>PfEMP1</italic>) multiple <italic>var</italic> gene family. Evidence presented here suggests that the <italic>SICAvar</italic> family members have arisen through a process of gene duplication, selection pressure, and variation. Highly evolving genes including <italic>PfEMP1</italic>family members tend to be restricted to relatively unstable sub-telomeric regions that drive change with core genes protected in genetically stable intrachromosomal locations. The comparable <italic>SICAvar</italic> and <italic>kir</italic> gene family members are counter-intuitively located across chromosomes. Here, we demonstrate that, in contrast to conserved core genes, <italic>SICAvar</italic> and <italic>kir</italic> genes occupy otherwise gene-sparse chromosomal locations that accommodate rapid evolution and change. The novel methods presented here offer the malaria research community not only new tools to generate comprehensive genome sequence data from small clinical samples but also new insight into the complexity of clinically important real-world parasites.</p>
</abstract>
<kwd-group>
<kwd>
<italic>Plasmodium knowlesi</italic>
</kwd>
<kwd>genomes</kwd>
<kwd>clinical samples</kwd>
<kwd>
<italic>SICAvar</italic>
</kwd>
<kwd>
<italic>kir</italic>
</kwd>
<kwd>Nanopore</kwd>
<kwd>
<italic>de novo</italic>
</kwd>
<kwd>malaria</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>
<italic>Plasmodium knowlesi</italic> is a malaria parasite first described in a natural host, the long-tailed macaque monkey (<italic>Macaca fascicularis</italic>), in the early part of the 20<sup>th</sup> century (<xref ref-type="bibr" rid="B47">Knowles and Gupta, 1932</xref>). Although an incidental find, <italic>P. knowlesi</italic> was soon exploited as a model parasite for malaria research as recently reviewed (<xref ref-type="bibr" rid="B11">Butcher and Mitchell, 2018</xref>; <xref ref-type="bibr" rid="B34">Galinski et al., 2018</xref>; <xref ref-type="bibr" rid="B71">Pasini et al., 2018</xref>). Experimental <italic>P. knowlesi</italic> was well characterized over time with several lines adapted from natural macaque hosts and one human infection originating in geographically distinct regions (<xref ref-type="bibr" rid="B15">Chin et al., 1965</xref>; <xref ref-type="bibr" rid="B16">Chin et al., 1968</xref>; <xref ref-type="bibr" rid="B11">Butcher and Mitchell, 2018</xref>; <xref ref-type="bibr" rid="B34">Galinski et al., 2018</xref>; <xref ref-type="bibr" rid="B71">Pasini et al., 2018</xref>). Taken together, experimental lines of <italic>P. knowlesi</italic> remain important members of the malaria research arsenal.</p>
<p>What sets <italic>P. knowlesi</italic> apart is that it occupies several important niche areas&#x2014;as an experimental model, a natural parasite of Southeast Asian macaque monkeys, and the causative agent of zoonotic malaria in the human host (<xref ref-type="bibr" rid="B78">Singh et al., 2004</xref>). In nature, transmission is established in the jungles of Southeast Asia, areas that support the sylvan mosquito vectors, the parasite, and the natural macaque hosts. People who enter transmission zones are susceptible to infected mosquito bites and infection. <italic>P. knowlesi</italic> has effectively crossed the vertebrate host species divide and is responsible for malaria in contemporary human hosts (<xref ref-type="bibr" rid="B90">World-Health-Organization, 2021</xref>).</p>
<p>Zoonotic malaria caused by <italic>P. knowlesi</italic> is currently the most common type of malaria in Malaysia, with most of the cases reported in Malaysian Borneo (<xref ref-type="bibr" rid="B14">Chin et al., 2020</xref>). Indeed, naturally acquired <italic>P. knowlesi</italic> malaria causes a spectrum of disease from uncomplicated to severe and fatal infections with tantalizing similarity to severe adult malaria caused by <italic>P. falciparum</italic> (<xref ref-type="bibr" rid="B20">Cox-Singh et al., 2008</xref>; <xref ref-type="bibr" rid="B22">Daneshvar et al., 2009</xref>; <xref ref-type="bibr" rid="B21">Cox-Singh et al., 2010</xref>; <xref ref-type="bibr" rid="B23">Daneshvar et al., 2018</xref>).</p>
<p>The clinical similarities observed in patients with severe <italic>P. knowlesi</italic> and <italic>P. falciparum</italic> infections suggest that <italic>P. knowlesi</italic> has the potential to serve as a translational animal model system for severe malaria pathophysiology that has hitherto eluded medical science (<xref ref-type="bibr" rid="B69">Ozwara et al., 2003</xref>; <xref ref-type="bibr" rid="B21">Cox-Singh et al., 2010</xref>; <xref ref-type="bibr" rid="B19">Cox-Singh and Culleton, 2015</xref>; <xref ref-type="bibr" rid="B65">Onditi et al., 2015</xref>; <xref ref-type="bibr" rid="B18">Cox-Singh, 2018</xref>).</p>
<p>To take this idea forward, it seemed prudent to compare genome sequences derived from contemporary clinical isolates of <italic>P. knowlesi</italic> with <italic>P. knowlesi</italic> genomes generated from <italic>P. knowlesi</italic>-infected red blood cells propagated in rhesus monkeys (<xref ref-type="bibr" rid="B70">Pain et al., 2008</xref>; <xref ref-type="bibr" rid="B51">Lapp et al., 2018</xref>) or <italic>in vitro</italic> cultures (<xref ref-type="bibr" rid="B9">Benavente et al., 2018</xref>). Our data from clinical isolates were primarily compared to the first reference genome (<xref ref-type="bibr" rid="B70">Pain et al., 2008</xref>) and a cultured parasite (PkA1-H.1) (<xref ref-type="bibr" rid="B9">Benavente et al., 2018</xref>) control reference sequence and assembly generated using the same procedures.</p>
<p>Previously, we developed methods to produce high-quality Illumina short-read <italic>P. knowlesi</italic> genome sequence data from frozen clinical blood samples (<xref ref-type="bibr" rid="B73">Pinheiro et al., 2015</xref>). The outputs of that work identified genome-wide diversity, including a genomic dimorphism in <italic>P. knowlesi</italic> isolated from patients, but the Illumina platform was not suitable to resolve complex multiple gene family members.</p>
<p>
<italic>Plasmodium</italic> species have a number of multiple gene families that code for infected host red blood cell surface proteins. The proteins are antigenic and highly variable to avoid host immune recognition and parasite destruction (<xref ref-type="bibr" rid="B87">Wahlgren et al., 2017</xref>; <xref ref-type="bibr" rid="B40">Harrison et al., 2020</xref>). Among these are the <italic>P. falciparum</italic> erythrocyte membrane protein (<italic>PfEMP1</italic>) gene family members with an estimated 67 copies in the <italic>P. falciparum</italic> 3D7 reference genome and variable copy numbers in clinical isolates (<italic>n</italic> &#x3d; 47&#x2013;90) (<xref ref-type="bibr" rid="B35">Gardner et al., 2002</xref>; <xref ref-type="bibr" rid="B67">Otto et al., 2018</xref>). <italic>PfEMP1</italic> genes are expressed in a mutually exclusive manner with only one predominantly expressed at any one time (<xref ref-type="bibr" rid="B43">Hviid and Jensen, 2015</xref>; <xref ref-type="bibr" rid="B1">Abdi et al., 2017</xref>; <xref ref-type="bibr" rid="B6">Andrade et al., 2020</xref>). Importantly, <italic>PfEMP1</italic> gene expression is implicated in <italic>P. falciparum</italic> virulence and progression to severe disease (<xref ref-type="bibr" rid="B53">Lavstsen et al., 2012</xref>; <xref ref-type="bibr" rid="B1">Abdi et al., 2017</xref>; <xref ref-type="bibr" rid="B76">Shabani et al., 2017</xref>; <xref ref-type="bibr" rid="B87">Wahlgren et al., 2017</xref>; <xref ref-type="bibr" rid="B59">Milner, 2018</xref>; <xref ref-type="bibr" rid="B81">Tessema et al., 2019</xref>; <xref ref-type="bibr" rid="B45">Jensen et al., 2020</xref>). While other multiple gene families are described in all <italic>Plasmodium</italic> species studied to date, <italic>PfEMP1</italic> gene-like families are rare, and among the parasites that cause human disease, they are found only in <italic>P. falciparum</italic> and <italic>P. knowlesi</italic> (<xref ref-type="bibr" rid="B35">Gardner et al., 2002</xref>; <xref ref-type="bibr" rid="B70">Pain et al., 2008</xref>). The comparable <italic>P. knowlesi schizont-infected cell agglutination variant antigen</italic> (<italic>SICAvar</italic>) gene family has been reported in detail in experimental parasites that had been passaged in rhesus monkeys (<xref ref-type="bibr" rid="B4">al-Khedery et al., 1999</xref>; <xref ref-type="bibr" rid="B70">Pain et al., 2008</xref>; <xref ref-type="bibr" rid="B52">Lapp et al., 2015</xref>; <xref ref-type="bibr" rid="B34">Galinski et al., 2018</xref>; <xref ref-type="bibr" rid="B51">Lapp et al., 2018</xref>) but to our knowledge not in wild-type parasites, including <italic>P. knowlesi</italic> isolated from patients. Given the <italic>PfEMP1</italic> gene association with severe disease in <italic>P. falciparum,</italic> we are particularly interested in describing <italic>P. knowlesi SICAvar</italic> multiple gene family member organization, location, and copy number in clinical isolates using amplification-free genome sequencing. With the methodology presented here and in the study by <xref ref-type="bibr" rid="B66">Oresegun et al. (2021</xref>), we can move forward in subsequent studies to achieve this goal.</p>
<p>Multiple gene family members are similar with long stretches of regions of low complexity that require long-read sequencing technologies to resolve (<xref ref-type="bibr" rid="B73">Pinheiro et al., 2015</xref>; <xref ref-type="bibr" rid="B41">Heather and Chain, 2016</xref>). Recently, the PacBio long-read sequencing platform was used to describe, for the first time, the core <italic>P. falciparum</italic> genome in clinical isolates and demark sub-telomeric regions to compare genome organization and diversity between clinical isolates from different geographical regions and the commonly used <italic>P. falciparum</italic> clone 3D7 (<xref ref-type="bibr" rid="B67">Otto et al., 2018</xref>).</p>
<p>The PacBio platform is outside of our reach because we have small-volume frozen whole blood samples that yield parasite DNA well below the quantity required for amplification-free PacBio sequencing (<xref ref-type="bibr" rid="B9">Benavente et al., 2018</xref>; <xref ref-type="bibr" rid="B51">Lapp et al., 2018</xref>; <xref ref-type="bibr" rid="B67">Otto et al., 2018</xref>). Here, we use the accessible, portable, and affordable Oxford Nanopore Technologies MinION long-read sequencing platform, suitable for small-quantity input DNA, to sequence and <italic>de novo</italic> assemble two new <italic>P. knowlesi</italic> reference genome sequences representing each genetically dimorphic form of <italic>P. knowlesi</italic> found in our patient cohort (<xref ref-type="bibr" rid="B73">Pinheiro et al., 2015</xref>; <xref ref-type="bibr" rid="B2">Ahmed et al., 2014</xref>).</p>
<p>The new reference genomes will, for the first time, provide insight into clinically relevant contemporary <italic>P. knowlesi</italic> parasites. These diverse parasites are essentially wild-type and the product of ongoing mosquito transmission and recombination in nature (<xref ref-type="bibr" rid="B7">Assefa et al., 2015</xref>; <xref ref-type="bibr" rid="B73">Pinheiro et al., 2015</xref>; <xref ref-type="bibr" rid="B25">Divis et al., 2018</xref>; <xref ref-type="bibr" rid="B3">Ahmed and Quan, 2019</xref>; <xref ref-type="bibr" rid="B32">Fong et al., 2019</xref>). The genomes will offer a valuable resource not only for our studies on members of the <italic>SICAvar</italic> gene family and virulence but also to the wider malaria research community working on comparative biology of malaria parasites, drug discovery, and vaccine development.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>Materials and Methods</title>
<sec id="s2-1">
<title>Sample Selection</title>
<p>
<italic>P. knowlesi</italic> DNA extracted from archived clinical samples collected with informed consent as part of a non-interventional study were used (<xref ref-type="bibr" rid="B2">Ahmed et al., 2014</xref>). The isolates were selected to represent each of the two genetically distinct clusters, KH273 (sks047) and KH195 (sks048), of <italic>P. knowlesi</italic>&#x2013;infected patients in the study cohort (<xref ref-type="bibr" rid="B2">Ahmed et al., 2014</xref>; <xref ref-type="bibr" rid="B73">Pinheiro et al., 2015</xref>). Control <italic>P. knowlesi</italic> DNA was extracted from the experimental line <italic>P. knowlesi</italic> A1-H.1 adapted to <italic>in vitro</italic> culture in human erythrocytes, the culture kindly donated by Robert Moon (<xref ref-type="bibr" rid="B61">Moon et al., 2013</xref>). In order to distinguish the genome data generated here for <italic>P. knowlesi</italic> A1-H.1 from those already existing, we use the unique abbreviation StAPkA1H1 (<xref ref-type="bibr" rid="B24">Diez Benavente et al., 2017</xref>; <xref ref-type="bibr" rid="B9">Benavente et al., 2018</xref>).</p>
</sec>
<sec id="s2-2">
<title>Plasmodium DNA Extraction</title>
<p>Human DNA was depleted from 200 to 400&#xa0;&#xb5;l thawed clinical samples using a previously described method (<xref ref-type="bibr" rid="B66">Oresegun et al., 2021</xref>). Briefly, surviving human leucocytes in thawed samples were removed using anti-human CD45 DynaBeads (ThermoFisher Scientific). The resulting parasite pellet was washed to remove soluble human DNA (hDNA), and parasite-enriched DNA (pDNA) was extracted using the QIAamp Blood Mini Kit (QIAGEN) with final elution into 150&#xa0;&#xb5;l AE buffer. DNA concentrations were quantified using the Qubit 2.0 fluorometer (Qubit&#x2122;, Invitrogen) and real-time qPCR on RotorGene (QIAGEN). Recovered DNA was concentrated, and short fragments were removed by mixing at a ratio of 1:1 by volume with AMPureXP magnetic beads (Beckman Coulter) following the manufacturer&#x2019;s instructions. Briefly, the AMPureXP bead mixture was placed in a magnetic field, and DNA bound to the beads was rinsed twice with 70% ethanol before air-drying to allow residual ethanol to evaporate. Parasite-enriched DNA was eluted in 10&#xa0;&#xb5;l nuclease-free H<sub>2</sub>O (Ambion). One microliter of recovered DNA concentrate was used for DNA quantification using a Qubit fluorimeter (ThermoFisher Scientific), and 7.5&#xa0;&#xb5;l was taken forward for sequencing library preparation.</p>
</sec>
<sec id="s2-3">
<title>Library Preparation and Sequencing</title>
<p>Parasite-enriched DNA was sequenced using the Oxford Nanopore Technologies (ONT) MinION long-read sequencing platform. Library preparations were selected to suit PCR-free sequencing for the small pDNA quantities available to study (&#x223c;400&#xa0;ng). Sequencing libraries were prepared following the manufacturer&#x2019;s instructions for the SQK-RBK004 ONT sequencing kit. Sequencing was performed using R9.4.1 flowcells or R10 flowcells (<xref ref-type="bibr" rid="B66">Oresegun et al., 2021</xref>). Previously sequenced Illumina reads for the patient isolates (sks047 and sks048) were retrieved from the European Nucleotide Archive, with accession codes ERR366425 and ERR274221, respectively (<xref ref-type="bibr" rid="B73">Pinheiro et al., 2015</xref>). Further short-read sequencing was carried out on PCR-enriched DNA using the Illumina MiSeq platform at the London School of Hygiene and Tropical Medicine and methods established by Diez <xref ref-type="bibr" rid="B10">Benavente et al. (2019)</xref>.</p>
</sec>
<sec>
<title>Reference Genomes</title>
<p>For chromosome scaffolding and quality assessment comparison, the <italic>P. knowlesi</italic> PKNH reference genome (<xref ref-type="bibr" rid="B70">Pain et al., 2008</xref>) (version 2) was downloaded from Sanger (<ext-link ext-link-type="uri" xlink:href="http://ftp/ftp.sanger.ac.uk/pub/genedb/releases/latest/Pknowlesi/">ftp://ftp.sanger.ac.uk/pub/genedb/releases/latest/Pknowlesi/&#x23;</ext-link>). In addition, further comparisons were carried out using the <italic>P. knowlesi</italic> PkA1H1 reference genome (<xref ref-type="bibr" rid="B9">Benavente et al., 2018</xref>) from NCBI [accession code: GCA_900162085].</p>
</sec>
<sec id="s2-4">
<title>
<italic>De Novo</italic> Genome Assembly</title>
<p>MinION FAST5 file outputs were locally base called using the high accuracy model of the guppy basecaller (v4.0.15; Ubuntu 19.10; GTX1060) with the following parameters: &#x201c;<italic>-r -v -q 0 --qscore-filtering -x auto</italic>.&#x201d; Demultiplexing was carried out using qcat software (v1.1.0) with the &#x201c;<italic>--detect-middle --trim -k --guppy</italic>&#x201d; parameters, and then adapter removal was carried out using porechop (v0.2.4) with default parameters and the most recent versions released from ONT technologies. Human DNA (hDNA) contamination was removed from the adapter-free reads by alignment against the human GRCh38.p13 reference genome (retrieved from NCBI accession code: GCF_000001405.39) (<xref ref-type="bibr" rid="B58">Lander et al., 2001</xref>) using minimap2 (v2.17) (<xref ref-type="bibr" rid="B55">Li, 2018</xref>) with &#x201c;<italic>-ax map-ont</italic>&#x201d; default parameters. Unmapped reads were separated from the binary sequence alignment (BAM) file using samtools (v1.10) (<xref ref-type="bibr" rid="B56">Li et al., 2009</xref>; <xref ref-type="bibr" rid="B54">Li, 2011</xref>) and converted back to FASTQ using bedtools (v2.29.2) (<xref ref-type="bibr" rid="B74">Quinlan and Hall, 2010</xref>) for <italic>de novo</italic> genome assembly using Flye (v2.8.1) (<xref ref-type="bibr" rid="B49">Kolmogorov et al., 2019</xref>) with an expected genome size of 25&#xa0;Mb and &#x201c;<italic>--nano-raw</italic>&#x201d; default parameters. Successful assemblies were assessed for contamination using BlobTools (v1.0.1) (<xref ref-type="bibr" rid="B50">Laetsch and Blaxter, 2017</xref>). Contigs not taxonomically assigned as Apicomplexan were discarded.</p>
</sec>
<sec id="s2-5">
<title>Assembly Polishing and Correction</title>
<p>Draft assemblies were polished using four iterations of racon (v1.4.13) (<xref ref-type="bibr" rid="B86">Vaser et al., 2017</xref>); in the default setting, raw long-read isolate sequence reads which did not align to the human GRCh38.p13 (henceforth parasite-reads) were retained. As part of the polishing step, alignments of parasite-reads against the draft assembly were performed using minimap2 (v2.17) (<xref ref-type="bibr" rid="B55">Li, 2018</xref>). A consensus sequence was subsequently generated from the racon output using medaka (v1.0.3; default settings) (<xref ref-type="bibr" rid="B68">Oxford Nanopore Technologies, 2019</xref>). Further polishing and correction were carried out using Illumina paired-end reads where available, using three iterations of pilon (v1.23) with default parameters &#x201c;<italic>-Xmx120G, --tracks, --fix all, circles</italic>&#x201d; (<xref ref-type="bibr" rid="B88">Walker et al., 2014</xref>).</p>
</sec>
<sec id="s2-6">
<title>Masking Repetitive Elements</title>
<p>The <italic>P. knowlesi</italic> PKNH reference mitochondrial (MIT) and apicoplast (API) sequences were extracted and individually aligned against draft <italic>P. knowlesi</italic> assemblies using MegaBLAST (v.2.9; default parameters) (<xref ref-type="bibr" rid="B62">Morgulis et al., 2008</xref>; <xref ref-type="bibr" rid="B70">Pain et al., 2008</xref>). Contigs which aligned to the reference PKNH MIT and API genomes were subsequently removed and circularized on Circlator (v1.5.5) (<xref ref-type="bibr" rid="B42">Hunt et al., 2015</xref>) with the command &#x201c;<italic>circlator all --data_type nanopore-raw --bwa_opts &#x2018;-x ont2d&#x2019; --merge_min_id 85 --merge_breaklen 1000</italic>.&#x201d; API/MIT-free draft nuclear assemblies (henceforth, draft assemblies) were taken forward through RepeatModeler (v1.0.10) (<xref ref-type="bibr" rid="B31">Flynn et al., 2020</xref>), and the outputs were utilized as input for Censor (<xref ref-type="bibr" rid="B48">Kohany et al., 2006</xref>) where the options &#x201c;<italic>Eukaryota</italic>&#x201d; and &#x201c;<italic>Report simple repeats</italic>&#x201d; were selected. Identified transposable elements and repeats in the censor outputs were classified based on the class of repeats to make a repeat library for each assembly. Repeat libraries of each draft assembly were combined and misplaced, redundant sequences were removed using CD-HIT (v4.8.1; &#x201c;<italic>-c 1.0 -n 10 -d 0 -g 1 -M 60000</italic>&#x201d; parameters) (<xref ref-type="bibr" rid="B57">Li and Godzik, 2006</xref>; <xref ref-type="bibr" rid="B33">Fu et al., 2012</xref>). This generated a singular &#x201c;master&#x201d; repeat library encompassing the non-redundant list of identified elements across the three draft assemblies.</p>
<p>With the master repeat library, RepeatMasker (v4.0.7) was run on each draft assembly producing a tab-separated value (TSV) output of the identified repeats in the assembly. Then, using &#x2018;One Code to Find Them All&#x2019; (OCFTA) (<xref ref-type="bibr" rid="B8">Bailly-Bechet et al., 2014</xref>), each TSV file was parsed to clarify further repeat positions found using RepeatMasker. Next, the LTRHarvest (<xref ref-type="bibr" rid="B28">Ellinghaus et al., 2008</xref>) module of GenomeTools (v1.6.1) (<xref ref-type="bibr" rid="B38">Gremme et al., 2013</xref>) was used to find secondary structures of long terminal repeats (LTRs) and other alternatives in the DRAFT assemblies. Here, the &#x201c;<italic>suffixerator</italic>&#x201d; function was implemented with &#x201c;<italic>-tis -suf -lcp -des -ssp -sds -dna</italic>&#x201d; parameters while the &#x201c;<italic>ltrharvest</italic>&#x201d; function was run with &#x201c;<italic>-mintsd 5 -maxtsd 100</italic>&#x201d;&#x2019; parameters. Concurrently, TransposonPSI was also used on the DRAFT assemblies with default parameters to find repeat elements based on their coding sequences.</p>
<p>Redundant repeat element sequences were removed from the outputs of RepeatMasker, OCTFA, LTRHarvest, and TransposonPSI using a custom script, to generate a genome feature file (GFF3) where each transposable and repetitive element of each DRAFT assembly is represented once. Then, within each draft assembly, repeat elements were masked using the coordinates present in the non-redundant GFF3 file and the &#x201c;<italic>maskfasta</italic>&#x201d; function of bedtools (v2.27; default settings and &#x201c;<italic>-soft</italic>&#x201d;).</p>
</sec>
<sec id="s2-7">
<title>Prediction and Annotation</title>
<p>The masked draft assemblies were checked for chimeric contigs using Ragtag (v1.0.1) (<xref ref-type="bibr" rid="B5">Alonge et al., 2019</xref>) where both the &#x201c;<italic>correct</italic>&#x201d; and &#x201c;<italic>scaffold</italic>&#x201d; functions were run with the &#x201c;<italic>--debug --aligner nucmer --nucmer-params &#x3d; &#x2018;-maxmatch -l 100 -c 500&#x2019;</italic>&#x201d; parameters (<xref ref-type="bibr" rid="B56">Li et al., 2009</xref>; <xref ref-type="bibr" rid="B54">Li, 2011</xref>).</p>
<p>With the chimeric contigs broken, masked draft assemblies were uploaded on the Companion webserver (<xref ref-type="bibr" rid="B80">Steinbiss et al., 2016</xref>) for gene prediction and annotation using the sequence prefix of &#x201c;PKA1H1_STAND&#x201d; for the cultured experimental line (StAPKA1H1) and &#x201c;PKCLINC&#x201d; for patient isolates (sks047 and sks048). Companion software was run with no transcript evidence, 500&#xa0;bp minimum match length, and 80% match similarity for contig placement, 0.8 AUGUSTUS (<xref ref-type="bibr" rid="B79">Stanke et al., 2006</xref>) score threshold, and taxid 5851. Additionally, pseudochromosomes were contiguated, reference proteins were aligned to the target sequence, pseudogene detection was carried out, and RATT was used for reference gene models.</p>
</sec>
<sec id="s2-8">
<title>Comparative Genomics, Quality Assessment and Analyses</title>
<p>As the pipeline progressed, assembly metrics were checked using assembly-stats (v1.0.1) and pomoxis (v0.3.4). Additionally, draft genomes were further assessed for completeness and accuracy using Benchmarking Universal Single-Copy Orthologues (<italic>BUSCO</italic>) v5.0 with &#x201c;<italic>-l plasmodium_odb10 -f -m geno --long</italic>&#x201d; parameters (<xref ref-type="bibr" rid="B77">Sim&#xe3;o et al., 2015</xref>). GFF3 files generated on Companion were parsed for genes of interest, including multigene families known to span the core genome and telomeric regions. Chromosomes of the annotated draft genomes were individually aligned against the corresponding <italic>P. knowlesi</italic> PKNH reference chromosome (<xref ref-type="bibr" rid="B70">Pain et al., 2008</xref>) with minimap2 parameters &#x201c;<italic>-ax asm5</italic>. &#x201d; The resulting alignment files were analyzed on Qualimap (v.2.2.2) (<xref ref-type="bibr" rid="B64">Okonechnikov et al., 2016</xref>) with parameters &#x201c;<italic>&#x2212;nw 800&#x2212;hm 7</italic>. &#x201d; Gene density, chromosome structure, and multigene family plots were generated using the karyoploteR visualization package (<xref ref-type="bibr" rid="B36">Gel and Serra, 2017</xref>). Dotplots to identify repetitions, breaks, and inversions were generated from minimap2 whole genome alignments using D-GENIES default settings (<xref ref-type="bibr" rid="B12">Cabanettes and Klopp, 2018</xref>).</p>
</sec>
<sec id="s2-9">
<title>Structural Variant Analyses</title>
<p>The StAPkA1H1 draft genome, assembled here, was used as the reference for structural variant calling and subsequent variant annotation to ensure parity across sequencing technologies. Read alignment-based structural variant calling (henceforth reads-based) was achieved using the Oxford Nanopore structural variation pipeline (ONTSVP) (<ext-link ext-link-type="uri" xlink:href="https://github.com/nanoporetech/pipeline-structural-variation">https://github.com/nanoporetech/pipeline-structural-variation</ext-link>), while the assembly-based approach was completed with Assemblytics (<xref ref-type="bibr" rid="B63">Nattestad and Schatz, 2016</xref>). Using a modified Snakefile, FASTQ isolates parasite-reads and the StAPkA1H1 draft genome; the ONTSVP first parses the input reads using catfishq (<ext-link ext-link-type="uri" xlink:href="https://github.com/philres/catfishq">https://github.com/philres/catfishq</ext-link>) and seqtk (<ext-link ext-link-type="uri" xlink:href="https://github.com/lh3/seqtk">https://github.com/lh3/seqtk</ext-link>) before carrying out alignment using lra with parameters &#x201c;<italic>-ONT -p s</italic>&#x201d; (<xref ref-type="bibr" rid="B75">Ren and Chaisson, 2020</xref>). The resulting alignment file was sorted and indexed using samtools, and read coverage was then calculated using mosdepth (&#x201c;&#x2212;<italic>x</italic>&#x2212;<italic>n</italic>&#x2212;<italic>b 1000000</italic>&#x201d;) (<xref ref-type="bibr" rid="B72">Pedersen and Quinlan, 2018</xref>). Structural variants (SVs) were called using cuteSV (<xref ref-type="bibr" rid="B46">Jiang et al., 2020</xref>) with parameters &#x201c;<italic>--min-size 30 --max-size 100,000 --retain_work_dir --report_readid --min_support 2</italic>.&#x201d; Variants were subsequently filtered for length (30&#xa0;bp), depth (8 reads), quality (Q30), and structural variant type (SVTYPE) such as insertions (INS) by default, before filtered variants were sorted and indexed. Failed SV types were manually filtered based on length (30&#xa0;bp) and quality (Q30) alone to determine the presence of high-quality, low-occurrence variants.</p>
<p>For the assembly-based structural variant calling for the clinical isolates sks047 and sks048 and StAPkA1H1, draft genomes were aligned against the PKNH reference genome (<xref ref-type="bibr" rid="B70">Pain et al., 2008</xref>) using nucmer with &#x201c;--maxmatch <italic>-l 100 -c 500</italic>&#x201d; parameters and outputs uploaded onto Assemblytics (<ext-link ext-link-type="uri" xlink:href="http://assemblytics.com">http://assemblytics.com</ext-link>) (<xref ref-type="bibr" rid="B63">Nattestad and Schatz, 2016</xref>) with default parameters and a minimum SV length of 30&#xa0;bp. BEDfile outputs of Assemblytics were converted to variant call format (VCF) files using SURVIVOR (v1.0.7) (<xref ref-type="bibr" rid="B44">Jeffares et al., 2017</xref>). VCF files for successful reads-based and assembly-based SV calling as well as the failed SV-type VCF files were further filtered to remove any variants less than 50&#xa0;bp in length and less than Q5 in quality using a bcftools one-liner (<ext-link ext-link-type="uri" xlink:href="https://github.com/samtools/BCFtools">https://github.com/samtools/BCFtools</ext-link>). A quality filter was not applicable for the assembly-based approach due to the lack of quality information in the original BEDfile output of Assemblytics. Variants exceeding these thresholds were annotated using vcfanno (v0.3.2) (<xref ref-type="bibr" rid="B63">Nattestad and Schatz, 2016</xref>) and subsequently sorted and indexed. Annotated variants, relevant BAM alignment files, and GFF files were visualized on IGV (<xref ref-type="bibr" rid="B85">Thorvaldsd&#xf3;ttir et al., 2013</xref>). Using IGV, a gene locus previously identified to be associated with dimorphism&#x2014;<italic>PknbpX</italic>a (<xref ref-type="bibr" rid="B73">Pinheiro et al., 2015</xref>)&#x2014;was analyzed to determine the presence of structural variants. Summary statistics were calculated using the &#x2018;stats&#x2019; function of SURVIVOR with parameters &#x201c;&#x2212;1&#x2212;1&#x2212;1.&#x201d; VCF files were compared using the &#x201c;isec&#x201d; function of bcftools with default settings, including analyses of the variants present within genes.</p>
</sec>
<sec id="s2-10">
<title>Duplication, Clustering, Genomic Organization and dN/dS Analyses</title>
<p>Scripts used can be found here: <ext-link ext-link-type="uri" xlink:href="https://github.com/peterthorpe5/plasmidium_genomes">https://github.com/peterthorpe5/plasmidium_genomes</ext-link>. Gene duplication analyses were performed using the similarity searches from DIAMOND-BlastP (1e-5) with the MCSanX toolkit (<xref ref-type="bibr" rid="B89">Wang et al., 2012</xref>). Orthologues clustering and dN/dS were performed as described in the study by <xref ref-type="bibr" rid="B82">Thorpe et al. (2018</xref>). Briefly, OrthoFinder (v2.2.7) (<xref ref-type="bibr" rid="B29">Emms and Kelly, 2019</xref>) was used to cluster all the amino acids sequences for the genomes used in this study. The resulting sequences from the clusters of interest were aligned using MUSCLE (v3.8.1551) (<xref ref-type="bibr" rid="B27">Edgar, 2004</xref>) and refined using MUSCLE. The resulting amino acid alignment was used as a template to back-translate the nucleotide coding sequence using Biopython for subsequent nucleotide alignment (<xref ref-type="bibr" rid="B17">Cock et al., 2009</xref>). The nucleotide alignment was filtered to remove any insertions and deletions and return an alignment with no gaps using trimAL (v1.4.1) (<xref ref-type="bibr" rid="B13">Capella-Gutierrez et al., 2009</xref>). The resulting alignment was subjected to dN/dS analysis using Codonphyml (v1.00 201407.24) (-m GY --fmodel F3X4 -t e -f empirical -w g -a e) (<xref ref-type="bibr" rid="B37">Gil et al., 2013</xref>). Genomic organization of classes of genes of interest was performed as described in the studies by <xref ref-type="bibr" rid="B30">Eves-van den Akker et al. (2016</xref>) and <xref ref-type="bibr" rid="B82">Thorpe et al. (2018</xref>, <xref ref-type="bibr" rid="B83">2020</xref>). For UpSet visualization the scripts can be found in the github link above.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<sec id="s3-1">
<title>Evaluating Draft <italic>de novo</italic> Genomes</title>
<p>The genome pipeline, beginning with Oxford Nanopore Technologies (ONT) MinION sequencing through to <italic>de novo</italic> assembly and genome annotation with downstream analyses, is shown (<xref ref-type="fig" rid="F1">Figure 1</xref>). The pipeline was used to produce <italic>P. knowlesi</italic> genomes using DNA extracted from two clinical isolates, sks047 and sks048, and, for comparison, DNA extracted from the well-characterized cultured line, <italic>P. knowlesi</italic> A1-H.1. For the purpose of clarity, the <italic>P. knowlesi</italic> A1-H.1 <italic>de novo</italic> draft genome assembled here is referred to as StAPkA1H1 (please see the Methods section). Read coverages of 225x, 71x, and 65x were obtained for StAPkA1H1, sks047, and sks048, respectively (<xref ref-type="table" rid="T1">Table 1</xref>). The draft assemblies resolved into 100 or fewer contigs before further reduction to &#x3c;72 contigs after scaffolding (<xref ref-type="table" rid="T1">Table 1</xref>). The quality of the draft assemblies was improved with Medaka&#x2019;s polishing resulting in <italic>BUSCO</italic> scores that increased from 68.6 to 89.7 (a 30.8% increase), 67.2 to 85.5 (a 27.2% increase), and 68.8 to 85.9 (a 24.8% increase) for StAPkA1H1, sks047, and sks048, respectively, with <italic>BUSCO</italic> completeness scores for the clinical isolates reaching 95% (<xref ref-type="table" rid="T1">Table 1</xref>). The observed increase in the number of contigs from 23.57 to 23.63&#xa0;Mb (0.22% increase) for sks047 and 24.49 to 24.56&#xa0;Mb (0.32% increase) for sks048 was likely due to the addition of relatively shorter reads (<xref ref-type="table" rid="T1">Table 1</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>
<italic>Plasmodium knowlesi de novo</italic> genome pipeline. The pipeline represents major forms of manipulation taken and tools utilized to generate, annotate, and analyze the two reference genomes derived from clinical isolates and the experimental line.</p>
</caption>
<graphic xlink:href="fgene-13-855052-g001.tif"/>
</fig>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Overview of assembly and quality metrics of the <italic>de novo</italic> assembled draft assemblies.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left">Isolate</th>
<th rowspan="2" align="center">Coverage</th>
<th colspan="5" align="center">
<italic>De novo</italic> assembly length (Mb)</th>
<th colspan="5" align="center">Contigs/scaffolds/chromosomes</th>
<th colspan="5" align="center">BUSCO completeness score (%)</th>
</tr>
<tr>
<th align="center">Raw</th>
<th align="center">Medaka</th>
<th align="center">Pilon</th>
<th align="center">RagTag</th>
<th align="center">Complete</th>
<th align="center">Raw</th>
<th align="center">Medaka</th>
<th align="center">Pilon</th>
<th align="center">RagTag</th>
<th align="center">Complete</th>
<th align="center">Raw</th>
<th align="center">Medaka</th>
<th align="center">Pilon</th>
<th align="center">RagTag</th>
<th align="center">Complete</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">PKNH<xref ref-type="table-fn" rid="Tfn1">
<sup>a</sup>
</xref>
</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="char" char=".">24.36</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">15</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="char" char=".">97.6</td>
</tr>
<tr>
<td align="left">PKA1H1<xref ref-type="table-fn" rid="Tfn2">
<sup>b</sup>
</xref>
</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="char" char=".">24.27</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">14</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="char" char=".">94.4</td>
</tr>
<tr>
<td align="left">StAPkA1H1</td>
<td align="center">225X</td>
<td align="char" char=".">24.15</td>
<td align="char" char=".">24.14</td>
<td align="center">N/A</td>
<td align="char" char=".">24.39</td>
<td align="char" char=".">24.39</td>
<td align="center">73</td>
<td align="center">111</td>
<td align="center">N/A</td>
<td align="center">71</td>
<td align="center">15</td>
<td align="char" char=".">68.6</td>
<td align="char" char=".">89.7</td>
<td align="center">&#x2014;</td>
<td align="char" char=".">89.7</td>
<td align="char" char=".">89.5</td>
</tr>
<tr>
<td align="left">sks047</td>
<td align="center">71X</td>
<td align="char" char=".">23.57</td>
<td align="char" char=".">23.63</td>
<td align="center">23.64</td>
<td align="char" char=".">24.17</td>
<td align="char" char=".">24.17</td>
<td align="center">100</td>
<td align="center">116</td>
<td align="center">116</td>
<td align="center">69</td>
<td align="center">15</td>
<td align="char" char=".">67.2</td>
<td align="char" char=".">85.5</td>
<td align="char" char=".">95.7</td>
<td align="char" char=".">95.9</td>
<td align="char" char=".">95.9</td>
</tr>
<tr>
<td align="left">sks048</td>
<td align="center">65X</td>
<td align="char" char=".">24.49</td>
<td align="char" char=".">24.56</td>
<td align="center">24.57</td>
<td align="char" char=".">24.81</td>
<td align="char" char=".">24.81</td>
<td align="center">74</td>
<td align="center">94</td>
<td align="center">94</td>
<td align="center">50</td>
<td align="center">15</td>
<td align="char" char=".">68.8</td>
<td align="char" char=".">85.9</td>
<td align="char" char=".">95.7</td>
<td align="char" char=".">95.7</td>
<td align="char" char=".">95.6</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Legend to <xref ref-type="table" rid="T1">Table 1</xref>: Quality improvements in the three <italic>de novo</italic> draft assemblies StPkA1H1, sks047, and sks048 were achieved by polishing with Medaka (<xref ref-type="bibr" rid="B68">Oxford Nanopore Technologies, 2019</xref>) and Pilon (<xref ref-type="bibr" rid="B88">Walker et al., 2014</xref>), checks for chimeric contig and scaffolding with RagTag (<xref ref-type="bibr" rid="B5">Alonge et al., 2019</xref>), and annotation of the draft assemblies with Companion (<xref ref-type="bibr" rid="B80">Steinbiss et al., 2016</xref>). The published P. knowlesi PKNH and PkA1H1 reference genomes generated from experimental lines were available in their complete forms. Information on raw reads and assembly was not available for comparison here.</p>
</fn>
<fn id="Tfn1">
<label>a</label>
<p>Pain et al.(<xref ref-type="bibr" rid="B70">2008</xref>).</p>
</fn>
<fn id="Tfn2">
<label>b</label>
<p>Diez Benavente et al.(<xref ref-type="bibr" rid="B24">2017</xref>).</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The combination of previously sequenced Illumina reads data with 34x and 166x short read coverage for sks047 and sks048, respectively, offered the opportunity for Pilon polishing the newly generated ONT sequence data for the clinical isolates. Pilon polishing resulted in improved <italic>BUSCO</italic> scores with sks047 seeing an 11.9% improvement (85.5&#x2013;95.7) and sks048 showing an 11.4% improvement (85.9&#x2013;95.7) (<xref ref-type="table" rid="T1">Table 1</xref>). Although Pilon did not change the number of contigs, both sks047 and sks048 saw a total length increase of 0.05% and increases in <italic>BUSCO</italic> scores. Additional Illumina sequencing was not available for StAPkA1H1, and Pilon polishing was not possible.</p>
<p>Scaffolding, chromosome structuring, and subsequent annotation initially proved difficult due to large sections of chromosomes 2 and 3 consistently being incorrectly placed in chromosomes 14 and 13, respectively. These large-scale inconsistencies were the result of contig chimers and were minimized or entirely corrected by de-chimerization using RagTag. Chromosomes corrected by RagTag retained regions of variability for the draft assemblies, although RagTag did not provide a complete solution in resolving all variable sequences (<xref ref-type="sec" rid="s11">Supplementary Figures S1, S2</xref>). In addition, it is possible that RagTag did not entirely retain highly variable regions such as telomeric regions that may have resulted in loss of coverage of genes positioned at extreme chromosomal boundaries (<xref ref-type="sec" rid="s11">Supplementary Figure S1</xref>).</p>
</sec>
<sec id="s3-2">
<title>Genome Annotation and Gene Content</title>
<p>Companion software resolved all three nuclear genomes, StAPkA1H1, sks047, and sks048, into 15 chromosomes&#x2013;14 Pk chromosomes and 1 &#x201c;bin&#x201d; or &#x201c;00&#x201d; chromosome (chr 00) holding sequence fragments which could not be confidently placed by the Companion pipeline (<xref ref-type="table" rid="T2">Table 2</xref>). Each draft genome was assigned a similar or greater number of coding genes than the <italic>P. knowlesi</italic> PKNH reference genome (5327 genes) when full protein-coding genes and pseudogenes annotated with predicted function (implying missing &#x201c;start&#x201d; and/or &#x201c;stop&#x201d; codons) were combined. The StAPkA1H1 draft assembly had 5358 genes (4385 coding &#x2b;973 pseudogenes), while the patient isolate draft genomes sks047 and sks048 had 5327 genes (4886 coding &#x2b;441 pseudogenes) and 5398 genes (4904 coding &#x2b;494 pseudogenes), respectively (<xref ref-type="table" rid="T2">Table 2</xref>). Non-coding genes were also found in all three draft genomes, including multiple small nuclear RNA (snRNA) (<xref ref-type="sec" rid="s11">Supplementary File S1</xref>). <italic>P. knowlesi schizont-infected cell agglutination variant antigen</italic> (<italic>SICAvar</italic>) and the <italic>Knowlesi-Interspersed Repeats</italic> (<italic>kir</italic>) multiple gene families were annotated in each draft genome (<xref ref-type="table" rid="T2">Table 2</xref>). There were consistently fewer <italic>kir</italic> gene family members in the draft genomes derived from the clinical isolates sks047 and sks048 with 26 and 25 <italic>kir</italic> genes, respectively, compared with 51 <italic>kir</italic> genes in the experimental cultured line StAPkA1H1 and 56 in the published PKNH reference genome (<xref ref-type="table" rid="T2">Table 2</xref>). It is unlikely that this is a result of assembly error given that StAPkA1H1 and the clinical isolates sks047 and sks048 were sequenced and <italic>de novo</italic> assembled in parallel using the same methodologies with the exception of Pilon polishing for StAPkA1H1.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Summary of the complete <italic>de novo</italic> draft genomes compared to the published <italic>P. knowlesi</italic> PKNH and PkA1H1 reference genomes.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left">Isolate</th>
<th rowspan="2" align="center">Complete assembly length (Mb)<xref ref-type="table-fn" rid="Tfn3">
<sup>a</sup>
</xref>
</th>
<th rowspan="2" align="center">Contigs</th>
<th rowspan="2" align="center">Chromosomes</th>
<th rowspan="2" align="center">N50 (Mb)</th>
<th rowspan="2" align="center">N count</th>
<th rowspan="2" align="center">Gaps</th>
<th rowspan="2" align="center">Genes<xref ref-type="table-fn" rid="Tfn4">
<sup>b</sup>
</xref>
</th>
<th rowspan="2" align="center">Total pseudo-genes</th>
<th rowspan="2" align="center">Shared orthologous clusters with reference</th>
<th rowspan="2" align="center">Unique orthologous clusters</th>
<th rowspan="2" align="center">Singleton clusters</th>
<th rowspan="2" align="center">KIRs</th>
<th colspan="3" align="center">
<italic>SICAvars</italic>
<xref ref-type="table-fn" rid="Tfn5">
<sup>c</sup>
</xref>
</th>
</tr>
<tr>
<th align="center">T1</th>
<th align="center">T2</th>
<th align="center">SDM&#x2019;s</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">PKNH (<xref ref-type="bibr" rid="B70">Pain et al., 2008</xref>)</td>
<td align="char" char=".">24.36</td>
<td align="center">&#x2014;</td>
<td align="center">15</td>
<td align="char" char=".">2.16</td>
<td align="center">11,381</td>
<td align="center">98</td>
<td align="center">5327</td>
<td align="center">12</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">56</td>
<td align="center">89</td>
<td align="center">20</td>
<td align="center">127</td>
</tr>
<tr>
<td align="left">PkA1-H.1 <xref ref-type="bibr" rid="B24">(Diez Benavente et al., 2017</xref>)</td>
<td align="char" char=".">24.27</td>
<td align="center">156</td>
<td align="center">14</td>
<td align="char" char=".">2.19</td>
<td align="center">148,255</td>
<td align="center">142</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
</tr>
<tr>
<td align="left">StAPkA1H1</td>
<td align="char" char=".">24.39</td>
<td align="center">71</td>
<td align="center">15</td>
<td align="char" char=".">2.13</td>
<td align="center">288,598</td>
<td align="center">127</td>
<td align="center">5358</td>
<td align="center">973</td>
<td align="center">4172</td>
<td align="center">3</td>
<td align="center">62</td>
<td align="center">51</td>
<td align="center">191</td>
<td align="center">15</td>
<td align="center">88</td>
</tr>
<tr>
<td align="left">sks047</td>
<td align="char" char=".">24.17</td>
<td align="center">69</td>
<td align="center">15</td>
<td align="char" char=".">2.09</td>
<td align="center">544,896</td>
<td align="center">109</td>
<td align="center">5327</td>
<td align="center">441</td>
<td align="center">4666</td>
<td align="center">9</td>
<td align="center">82</td>
<td align="center">26</td>
<td align="center">115</td>
<td align="center">9</td>
<td align="center">181</td>
</tr>
<tr>
<td align="left">sks048</td>
<td align="char" char=".">24.81</td>
<td align="center">50</td>
<td align="center">15</td>
<td align="char" char=".">2.21</td>
<td align="center">283,076</td>
<td align="center">84</td>
<td align="center">5398</td>
<td align="center">494</td>
<td align="center">4664</td>
<td align="center">11</td>
<td align="center">100</td>
<td align="center">25</td>
<td align="center">153</td>
<td align="center">7</td>
<td align="center">196</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Legend to <xref ref-type="table" rid="T2">Table 2</xref>: <italic>SICAvar</italic> domain fragments are found annotated across the genomes; combinations of these fragments can form complete <italic>SICAvar</italic> proteins, indicating the possibility of a larger number of <italic>SICAvar</italic> proteins present in native genomes. Gene data for reference PkA1H1 were unavailable.</p>
</fn>
<fn id="Tfn3">
<label>a</label>
<p>Total genome length excluding the mitochondrial and apicoplast genome sequences.</p>
</fn>
<fn id="Tfn4">
<label>b</label>
<p>Total number of coding genes and pseudogenes identified with a function.</p>
</fn>
<fn id="Tfn5">
<label>c</label>
<p>
<italic>SICAvar</italic> type 1 (T1); <italic>SICAvar</italic> type 2 (T2); <italic>SICAvar</italic> single domain fragments (SDMs). Single domain fragments code for <italic>SICAvar</italic> protein fragments.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>All three draft genomes had more <italic>SICAvar</italic> type 1 genes annotated than the reference PKNH genome. StAPkA1H1 had 191 <italic>SICAvar</italic> type 1 genes, sks047 had 115 <italic>SICAvar</italic> type 1 genes, and sks048 had 153 <italic>SICAvar</italic> type 1 genes. The reference PKNH genome is reported with 89 <italic>SICAvar</italic> type 1 genes (<xref ref-type="table" rid="T2">Table 2</xref>)<italic>. SICAvar</italic> gene fragments in each of the clinical isolate draft genomes, sks047 and sks048, outnumbered annotated <italic>SICAvar</italic> type 1 genes (<xref ref-type="table" rid="T2">Table 2</xref>). Conversely, the StAPkA1H1 draft genome had approximately half the number of <italic>SICAvar</italic> gene fragments compared with the clinical isolates and compared with StAPkA1H1 <italic>SICAvar</italic> type 1 genes (<xref ref-type="table" rid="T2">Table 2</xref>).</p>
<p>In regions of the draft genomes where gaps could not be resolved, contigs with evidence that they belonged together, either by long reads spanning them or by similarity to the reference, were scaffolded with N bases proportional to the gap size (<xref ref-type="table" rid="T2">Table 2</xref>). Higher N counts were observed in the three draft genomes generated here compared with the published reference genome (PKNH). In addition, sequences placed in the draft genome chr 00 may reflect the higher N counts in chromosomes 1&#x2013;14. The chr 00 of StAPkA1H1 clustered with the PKNH reference chr 00 (<xref ref-type="sec" rid="s11">Supplementary Figures S1,i</xref>) suggesting the StAPkA1H1 draft genome had a similar structure to the PKNH reference genome, including &#x201c;unplaced&#x201d; genes. In contrast, sks047 and sks048 chr 00 sequences were distributed across the reference genome, suggesting no single chromosome was more challenging to scaffold after de-chimerization (<xref ref-type="sec" rid="s11">Supplementary Figure S2ii,iii</xref>). The number of gaps in the three draft genomes was variable but within the range of the PKNH reference genome (<xref ref-type="table" rid="T2">Table 2</xref>).</p>
<p>Orthologous genes were determined using a similarity approach by OrthoMCL in Companion and showed that all three draft genomes shared &#x3e;4000 orthologs with the PKNH reference genome (<xref ref-type="table" rid="T2">Table 2</xref>). These orthologous genes can be considered as the core <italic>P. knowlesi</italic> gene set and are indicative of reliable and accurate assemblies (<xref ref-type="table" rid="T2">Table 2</xref>). In particular, draft genomes from the contemporary patient isolates sks047 and sks048 had &#x3e;4600 shared orthologues with the PKNH reference genome (<xref ref-type="table" rid="T2">Table 2</xref>).</p>
</sec>
<sec id="s3-3">
<title>Chromosome Structure</title>
<p>Dotplots of alignment of the three draft genomes show that they are syntenic with the PKNH reference regardless of gaps present in the genomes generated from patient isolates (Supplementary Figure S3). The unplaced sequences in chr00 account for at least 40% of gaps in the three draft genomes (<xref ref-type="table" rid="T2">Table 2</xref>). Indeed, each draft genome&#x2019;s chromosome structure conforms to that of the PKNH reference genome with uniform coverage across the chromosomes in regions with no gaps (<xref ref-type="sec" rid="s11">Supplementary Figure S4</xref>). This is also apparent in fragmented chromosomes, which retained the same chromosomal structure as PKNH (<xref ref-type="sec" rid="s11">Supplementary Figure S5</xref>). While coverage remained largely uniform, structural variations (&#x3e;10&#xa0;kb), for example, duplications and inversions, were present in the draft assemblies as seen in duplications present in multiple chromosomes in sks047 and sks048 (<xref ref-type="sec" rid="s11">Supplementary Figure S4b</xref>).</p>
<p>Additionally, inversions were present in almost every chromosome, often as inverted duplicate sequences, with the most striking instance observed in chromosome 5 of sks048 (<xref ref-type="sec" rid="s11">Supplementary Figure S4a,iii</xref>), where multiple duplicated inversions were observed. Frameshifts were present across chromosomes in all of the draft genomes (Supplementary Figure S4b). Given the robust clinical isolate draft genome assembly, the frameshifts observed deserve further investigation. Associated gaps do not appear to have impacted the distribution of genes within the draft genomes (<xref ref-type="fig" rid="F2">Figure 2</xref>). The mean annotated gene density shows the PKNH reference genome to have 22.05 genes per 100&#xa0;kbp, StAPkA1H1 to have 18.15, sks047 to have 20.25, and sks048 to have 19.80 (<xref ref-type="fig" rid="F2">Figure 2</xref>). Increased gene density may be achieved with manual pseudogene curation since mean gene density is inversely correlated with the number of pseudogenes, <italic>p</italic> &#x3d; 0.003186 (<xref ref-type="table" rid="T2">Table 2</xref>).</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Gene density plots for the <italic>P. knowlesi</italic> PKNH reference genome, StPkA1H1, sks047, and sks048 draft genomes. Gene density is calculated based on the number of identified genes within a sliding window of 100&#xa0;kb. Mean density shows the PKNH reference genome to have 22.05 genes per 100&#xa0;kb, StAPkA1H1 to have 18.15, sks047 to have 20.25, and sks048 to have 19.8. Plots were generated using karyoploteR (<xref ref-type="bibr" rid="B36">Gel and Serra, 2017</xref>).</p>
</caption>
<graphic xlink:href="fgene-13-855052-g002.tif"/>
</fig>
<p>With the exception of <italic>SICAvar</italic> and the <italic>Interspersed Repeat (IR)</italic> genes, analysis of the other multigene families reveals similar retention copy numbers in the three draft genomes and the PKNH reference (<xref ref-type="table" rid="T3">Table 3</xref>). Given the high similarity between the experimental lines StAPkA1H1 and PKNH in dotplots and other analyses, the total number of IR genes in the two different laboratory passaged lines, PKNH with 70 and StAPkA1H1 with 67, compared with clinical isolates, sks047 with 53 and sks047 with 52, may reflect gene retention through passive artificial passage. The clinical samples had fewer annotated <italic>kir</italic> genes than the experimental lines and in contrast have interspersed genes annotated as <italic>P. vivax vir</italic> that are absent in experimental lines (<xref ref-type="table" rid="T3">Table 3</xref>). Clinical isolates are effectively wild-type <italic>P. knowlesi</italic>, and the lower <italic>kir</italic> gene copy number and the presence of <italic>vir</italic>-like genes possibly reflect continual recombination and selection pressure during mosquito transmission in nature.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Number of annotated protein copies of the multigene families identified.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Genes</th>
<th align="center">Abbreviation</th>
<th align="center">PKNH</th>
<th align="center">StAPkA1H1</th>
<th align="center">sks047</th>
<th align="center">sks048</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Circumsporozoite protein</td>
<td align="left">
<italic>CSP/CS-TRAP</italic>
</td>
<td align="center">2</td>
<td align="center">2</td>
<td align="center">2</td>
<td align="center">2</td>
</tr>
<tr>
<td align="left">Cytoadherence linked asexual protein/gene</td>
<td align="left">
<italic>CLAG</italic>
</td>
<td align="center">2</td>
<td align="center">2</td>
<td align="center">2</td>
<td align="center">2</td>
</tr>
<tr>
<td align="left">Duffy binding/Duffy-antigen protein [erythrocyte binding protein (alpha/beta/gamma)]</td>
<td align="left">
<italic>DBP/DaBP</italic> [<italic>ERYBP(a/b/g)</italic>]</td>
<td align="center">3</td>
<td align="center">3</td>
<td align="center">3</td>
<td align="center">3</td>
</tr>
<tr>
<td align="left">Early transcribed membrane protein</td>
<td align="left">
<italic>ETRAMP</italic>
</td>
<td align="center">9</td>
<td align="center">9</td>
<td align="center">9</td>
<td align="center">9</td>
</tr>
<tr>
<td align="left">Knob-associated histidine-rich protein</td>
<td align="left">
<italic>KAHRP</italic>
</td>
<td align="center">1</td>
<td align="center">1</td>
<td align="center">1</td>
<td align="center">1</td>
</tr>
<tr>
<td align="left">
<italic>Knowlesi</italic> interspersed repeats</td>
<td align="left">
<italic>KIR</italic>
</td>
<td align="center">56</td>
<td align="center">51</td>
<td align="center">26</td>
<td align="center">25</td>
</tr>
<tr>
<td align="left">
<italic>Knowlesi</italic> interspersed repeats-like proteins</td>
<td align="left">
<italic>KIRLP</italic>
</td>
<td align="center">9</td>
<td align="center">9</td>
<td align="center">6</td>
<td align="center">5</td>
</tr>
<tr>
<td align="left">Vivax interspersed repeats</td>
<td align="left">
<italic>VIR</italic>
</td>
<td align="center">0</td>
<td align="center">0</td>
<td align="center">17</td>
<td align="center">16</td>
</tr>
<tr>
<td align="left">
<italic>Plasmodium</italic> interspersed repeats</td>
<td align="left">
<italic>PIR</italic>
</td>
<td align="center">5</td>
<td align="center">7</td>
<td align="center">4</td>
<td align="center">6</td>
</tr>
<tr>
<td align="left">Merozoite surface protein</td>
<td align="left">
<italic>MSP</italic>
</td>
<td align="center">13</td>
<td align="center">10</td>
<td align="center">10</td>
<td align="center">10</td>
</tr>
<tr>
<td align="left">Multidrug resistance (-associated protein)</td>
<td align="left">
<italic>MDRP/MDRaP</italic>
</td>
<td align="center">4</td>
<td align="center">3</td>
<td align="center">3</td>
<td align="center">3</td>
</tr>
<tr>
<td align="left">Reticulocyte binding protein</td>
<td align="left">
<italic>Pknbp/rbp</italic>
</td>
<td align="center">2</td>
<td align="center">2</td>
<td align="center">2</td>
<td align="center">2</td>
</tr>
<tr>
<td align="left">Sporozoite invasion-associated protein</td>
<td align="left">
<italic>SPIAP</italic>
</td>
<td align="center">2</td>
<td align="center">2</td>
<td align="center">2</td>
<td align="center">2</td>
</tr>
<tr>
<td align="left">Tryptophan-rich antigen</td>
<td align="left">
<italic>TrpRA</italic>
</td>
<td align="center">29</td>
<td align="center">29</td>
<td align="center">30</td>
<td align="center">29</td>
</tr>
<tr>
<td align="left">ATP-binding cassette (ABC) transporter</td>
<td align="left">
<italic>ABCtrp</italic>
</td>
<td align="center">15</td>
<td align="center">15</td>
<td align="center">15</td>
<td align="center">15</td>
</tr>
<tr>
<td align="left">Apicomplexan apetala2 transcription factor</td>
<td align="left">
<italic>ApiAP2</italic>
</td>
<td align="center">29</td>
<td align="center">28</td>
<td align="center">28</td>
<td align="center">28</td>
</tr>
<tr>
<td align="left">Schizont-infected agglutination variant proteins</td>
<td align="left">
<italic>SICAvar</italic>
</td>
<td align="center">109</td>
<td align="center">206</td>
<td align="center">124</td>
<td align="center">160</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Chromosome positional analyses of the <italic>kir</italic> genes show varied distribution across chromosomes with only three <italic>kir</italic> genes represented in chr 00 in the clinical isolate draft genomes, perhaps supporting constrained <italic>kir</italic> gene copy number in nature (<xref ref-type="sec" rid="s11">Supplementary Figure S6</xref>). <italic>SICAvar</italic> genes appear to be distributed across the genome, on all chromosomes, including the chromosomal extremities with more members annotated than previously reported by <xref ref-type="bibr" rid="B70">Pain et al. (2008)</xref>, particularly on chromosomes 10, 11, and 12 (<xref ref-type="sec" rid="s11">Supplementary Figure S7</xref>).</p>
</sec>
<sec id="s3-4">
<title>Structural Variation</title>
<p>Following filtering for length, quality, and depth, reads-based structural variants (SVs) were called using the ONT SV pipeline and assembly-based SVs were called using Assemblytics (<xref ref-type="bibr" rid="B63">Nattestad and Schatz, 2016</xref>). The reads-based approach returned 1,316 and 1,398 SVs for sks047 and sks048, respectively (<xref ref-type="table" rid="T4">Table 4</xref>). The assembly-based approach returned 856 and 839 SVs for sks047 and sks048, respectively (<xref ref-type="table" rid="T4">Table 4</xref>). The reads-based approach is expected to return more variants due to a higher error rate in the raw reads used than in the collapsed assembly-based methodology.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Summary of reads-based and assembly-based structural variants.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left">Isolate</th>
<th colspan="2" align="center">Total SVs</th>
<th colspan="2" align="center">Insertions</th>
<th colspan="2" align="center">Deletions</th>
</tr>
<tr>
<th align="center">Reads</th>
<th align="center">Assembly</th>
<th align="center">Reads</th>
<th align="center">Assembly</th>
<th align="center">Reads</th>
<th align="center">Assembly</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">sks047</td>
<td align="center">1,316</td>
<td align="center">856</td>
<td align="center">564</td>
<td align="center">396</td>
<td align="center">752</td>
<td align="center">460</td>
</tr>
<tr>
<td align="left">sks048</td>
<td align="center">1,398</td>
<td align="center">839</td>
<td align="center">667</td>
<td align="center">480</td>
<td align="center">731</td>
<td align="center">359</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Legend to <xref ref-type="table" rid="T4">Table 4</xref>: Reads-based SV calling involved filtering draft genomes for quality, length, and depth before aligning sks047 and sks048 input reads against the StAPkA1H1 genome using the Oxford Nanopore structural variant pipeline. Assembly-based structural variants were called using Assemblytics (<xref ref-type="bibr" rid="B63">Nattestad and Schatz, 2016</xref>) by aligning the complete draft genomes of sks047 and sks048 against the StAPkA1H1 genome.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>SVs that exceeded the quality, length, and read depth threshold are distributed across the genome on all chromosomes within coding and non-coding regions. Within the 101 shared SVs, 68 were within annotated genes, including within the <italic>SICAvar</italic> and <italic>kir</italic> multigene family members (<xref ref-type="sec" rid="s11">Supplementary Table S1</xref>). There were different variation signatures between the experimental cultured line StAPkA1H1 compared with the two clinical isolates, sks047 and sks048 (<xref ref-type="fig" rid="F3">Figure 3</xref>). StAPkA1H1 had more tandem variants than the clinical isolates, sks047 and sks048. In comparison, the clinical isolates show more variation in their repeat sequences with similar insertion and deletion (red and blue) and repeat expansion and contraction (green and purple) signatures than StAPkA1H1 (<xref ref-type="fig" rid="F3">Figure 3</xref>).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Assembly-based structural variation, size 50&#x2013;10,000&#xa0;bp, of StAPkA1H1, sks047, and sks048 draft genomes against the PKNH reference genome (<xref ref-type="bibr" rid="B70">Pain et al., 2008</xref>). Nucmer alignment was generated using parameters &#x201c;&#x2014;maxmatch&#x2212;l 100&#x2212;c 500&#x201d; with default and Assemblytics parameters (<xref ref-type="bibr" rid="B63">Nattestad and Schatz, 2016</xref>). Expansions (green and orange) refer to insertions that occur within repeat or tandem variants, while contractions (purple and brown) refer to deletions in these regions. More variation is present in the tandem variants (brown and orange) of StAPkA1H1 than those of the draft clinical isolate genomes, sks047 and sks048. In comparison the clinical isolates show more variation in their repeat sequences with similar insertion and deletions (red and blue) and repeat expansion and contraction (green and purple) signatures.</p>
</caption>
<graphic xlink:href="fgene-13-855052-g003.tif"/>
</fig>
</sec>
<sec id="s3-5">
<title>Gene Duplication</title>
<p>Gene duplication was quantified and classified using MCScanX (<xref ref-type="bibr" rid="B89">Wang et al., 2012</xref>). All genes within the draft genomes for the StAPkA1H1 cultured line and sks047 and sks048 clinical isolates were classified as either singleton (no identified duplication, proximal (two identified duplicated genes with &#x3c;20 genes between them), dispersed (&#x3e;20 genes between the 2 candidate genes), tandem (duplication events next to each other), and segmental/whole genome duplication (WGD) (&#x3e;4 co-linear genes with &#x3c;25 genes between them). To gain an insight into differences in gene duplication, duplication types were classified for the BUSCO eucaryotic core control gene population and for the PkSICAvar type 1, PkSICAvar type 2, and the kir multiple gene families in the three draft genomes, StAPkA1H1, sks047, and sks048 (<xref ref-type="fig" rid="F4">Figure 4</xref>). The duplication profile of the control population BUSCO genes was well matched between each of the draft genomes and also to the BUSCO duplication profile for the PKNH reference genome (Mann&#x2013;Whitney U test StAPkA1H1, <italic>p</italic> &#x3d; 0.92; sks047, <italic>p</italic> &#x3d; 0.67; sks048, <italic>p</italic> &#x3d; 0.66; PKNH <italic>p</italic> &#x3d; 0.40). Therefore, there were no observed excess duplication types for BUSCO genes (<xref ref-type="fig" rid="F4">Figure 4</xref>). However, duplication profiles for the genes annotated <italic>SICAvar</italic> type 1, <italic>SICAvar</italic> type 2, and kir in the draft genomes, StAPkA1H1, sks047, and sks048, were markedly different from the BUSCO gene profiles with no evidence for singleton genes (<xref ref-type="fig" rid="F4">Figure 4</xref>). When compared to 100 randomly obtained genes as a population, this result profile was statistically significant (Mann&#x2013;Whitney U test, <italic>p</italic> &#x3c; 1.0e&#x2212;9).</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Gene duplication classes for the draft genome assemblies for StAPkA1H1 (experimental cultured line) and the clinical isolates sks047 and sks048. Gene duplication was quantified and classified using MCScanX (<xref ref-type="bibr" rid="B89">Wang et al., 2012</xref>) for all genes in each genome and identified as black bars, singleton (no identified duplication); orange bars, dispersed (&#x3e;20 genes between the 2 candidate genes); blue bars, proximal (two identified duplicated genes with &#x3c;20 genes between them); pink bars, tandem (duplication events next to each other); and green bars, segmental/whole genome duplication (WGD) (&#x3e;4 co-linear genes with &#x3c;25 genes between them). The gene pools for each genome were divided into <italic>BUSCO</italic> (core genome genes) for comparison with the genes making up the <italic>SICAvar</italic> type 1 or <italic>SICAvar</italic> type 2 or <italic>kir</italic> multiple gene families. The draft genomes, StAPkA1H1, sks047, and sks048, had roughly similar profiles for <italic>BUSCO</italic> genes. Singletons (dark blue bars) were absent from the multiple gene families for all of the draft genomes.</p>
</caption>
<graphic xlink:href="fgene-13-855052-g004.tif"/>
</fig>
</sec>
<sec>
<title>Positive Selection: Nonsynonymous (dN)/Synonymous (dS) Substitutions</title>
<p>In order to determine if the <italic>SICAvar</italic> type 1, <italic>SICAvar</italic> type 2, and <italic>kir</italic> genes are under selection pressure, the associated predicted proteins from each genome, StAPkA1H1, PKNH, sks047, and sks048, were translated into amino acid sequences and grouped into putative orthologous gene clusters containing <italic>SICAvar</italic> type 1 or <italic>SICAvar</italic> type 2 or <italic>kir</italic> or <italic>BUSCO</italic> (control group) using OrthoFinder. The amino acid sequences were aligned, and the alignments used to &#x201c;backtranslate&#x201d; into nucleotide coding sequences. The mean dN/dS values for <italic>SICAvar</italic> type 1, <italic>SICAvar</italic> type 2, <italic>kir,</italic> and <italic>BUSCO</italic> gene clusters were 2.40, 2.74, 2.35, and 0.35, respectively, and the differences were statistically significant (Wilcoxon rank sum test <italic>p</italic>-value adjustment method Bonferroni: <italic>SICAvar</italic> type 1, &#x3d; 4.1e-08; <italic>SICAvar</italic> type 2 &#x3d; 0.0063 and <italic>kir</italic>, <italic>p</italic> &#x3d; 6.7e-13, <xref ref-type="table" rid="T5">Table 5</xref> and <xref ref-type="sec" rid="s11">Supplementary Figure S8</xref>).</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Non-synonymous versus synonymous (dN/dS) analysis of <italic>SICAvar</italic> type 1, <italic>SICAvar</italic> type 2, <italic>kir</italic>, and <italic>BUSCO</italic> gene clusters represented collectively in the StAPkA1H1, sks047, and sks048 draft genomes and the PKNH reference genome.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Cluster group</th>
<th align="center">Cluster count (n)</th>
<th align="center">Mean dN/dS per cluster</th>
<th align="center">Standard deviation</th>
<th align="center">Median</th>
<th align="center">Inter quartile range</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<italic>BUSCO</italic>
</td>
<td align="center">153</td>
<td align="char" char=".">0.353</td>
<td align="char" char=".">0.723</td>
<td align="char" char=".">0.101</td>
<td align="char" char=".">0.27</td>
</tr>
<tr>
<td align="left">
<italic>SICAvar</italic> type 1</td>
<td align="center">15</td>
<td align="char" char=".">2.4</td>
<td align="char" char=".">1.31</td>
<td align="char" char=".">2.37</td>
<td align="char" char=".">1.86</td>
</tr>
<tr>
<td align="left">
<italic>SICAvar</italic> type 2</td>
<td align="center">5</td>
<td align="char" char=".">2.74</td>
<td align="char" char=".">2.54</td>
<td align="char" char=".">1.83</td>
<td align="char" char=".">4.02</td>
</tr>
<tr>
<td align="left">
<italic>kir</italic>
</td>
<td align="center">26</td>
<td align="char" char=".">2.35</td>
<td align="char" char=".">1.19</td>
<td align="char" char=".">1.99</td>
<td align="char" char=".">1.5</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Legend to <xref ref-type="table" rid="T5">Table 5</xref>: <italic>SICAvar</italic> type 1, <italic>SICAvar</italic> type 2, <italic>kir</italic> genes, and <italic>BUSCO</italic> (control groups) genes were translated into amino acid sequences and clustered into orthologous groups using OrthoFinder (<xref ref-type="bibr" rid="B82">Thorpe et al., 2018</xref>). The amino acid sequences were aligned and the alignments &#x201c;backtranslated&#x201d; into nucleotide coding sequences for subsequent dN/dS analysis using Codophyml (<xref ref-type="bibr" rid="B37">Gil et al., 2013</xref>). In order to avoid false-positive dN/dS results, the nucleic acid alignment was filtered to dis-allow gaps, insertions, and deletions, and the final filtered nucleotide alignments with three or more sequences per cluster, the minimum requirement for Codophyml (<xref ref-type="bibr" rid="B37">Gil et al., 2013</xref>), were subjected to dN/dS analysis. <italic>SICAvar</italic> type 1, <italic>SICAvar</italic> type 2, and <italic>kir</italic> genes had a statistically significantly greater dN/dS value when than <italic>BUSCO</italic> gene clusters (Wilcoxon rank sum test <italic>p</italic> value adjustment method Bonferroni: <italic>SICAvar</italic> type 1, <italic>p</italic> &#x3d; 4.1e-08; <italic>SICAvar</italic> type 2, <italic>p</italic> &#x3d; 0.0063; and <italic>kir</italic>, <italic>p</italic> &#x3d; 6.7e-13).</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3-6">
<title>Genomic Organization of <italic>SICAvar</italic> Type 1, <italic>SICAvar</italic> Type 2 and <italic>kir</italic> Gene Family Members</title>
<p>
<italic>P. knowlesi SICAvar</italic> type 1, <italic>SICAvar</italic> type 2, and the <italic>kir</italic> gene family members appear to be variable, rapidly evolving genes, yet they are distributed across chromosomes, potentially destabilizing the core genome. To investigate this further, the distance from one gene to its neighbor was quantified in both a 3 prime (3&#x2019;) and 5 prime (5&#x2019;) direction, excluding genes at the start or end of a scaffold. The values were subjected to further analysis using the <italic>BUSCO</italic> core genes for comparison (<xref ref-type="fig" rid="F5">Figure 5A</xref>). With the exception of <italic>SICAvar</italic> type 2 in the 3&#x2019; direction, all of the <italic>SICAvar</italic> type 1, <italic>SICAvar</italic> type 2, and <italic>kir</italic> genes had a statistically significantly greater distance to their neighboring genes in both the 3&#x2019; and 5&#x2019; directions than <italic>BUSCO</italic> genes. In the 3&#x2019; direction, Kruskal&#x2013;Wallis chi-squared &#x3d; 272.15, df &#x3d; 4, <italic>p</italic>-value &#x3c; 2.2e-16. The Wilcoxon signed-rank test with Bonferroni <italic>p</italic>-value adjustment was <italic>SICAvar</italic> type 1 p &#x3d; 2e-16, <italic>SICAvar</italic> type 2 <italic>p</italic> &#x3d; 0.457, and <italic>kir p</italic> &#x3d; 1.1e-10. In the 5&#x2019; direction, all distances for <italic>SICAvar</italic> and <italic>kir</italic> genes were significantly different to the <italic>BUSCO</italic> control population. Kruskal&#x2013;Wallis chi-squared &#x3d; 269.33, df &#x3d; 4, <italic>p</italic>-value &#x3c; 2.2e-16 with Wilcoxon signed-rank test, Bonferroni <italic>p</italic>-value adjustment in comparison to <italic>BUSCO</italic>: <italic>SICAvar</italic> type 1 <italic>p</italic> &#x3d; 2e-16, <italic>SICAvar</italic> type 2 <italic>p</italic> &#x3d; 0.00123, and <italic>kir p</italic> &#x3d; 3.6e-09. The distribution of potentially destabilizing highly evolving <italic>SICAvar</italic> and <italic>kir</italic> genes across chromosomes in gene-sparse regions of the <italic>P. knowlesi</italic> genome would offer protection to core genes.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Genomic organization of multiple gene family members. <xref ref-type="fig" rid="F5">Figure 5A</xref> Heatmap gene density plots showing 5&#x2019; against 3&#x2019; intergenic distances (log10) for the draft genomes. (i) StAPkA1H1 experimental cultured line, (ii) clinical isolate sk047, and (iii) clinical isolate sk048. Gene density for intergenic distances is represented by color scale ranging from black (low) to white (high, maximum of 60 genes per bin). Genes classed as <italic>BUSCO</italic> (green dots), <italic>SICAvar</italic> type 1 (orange squares), <italic>SICAvar</italic> type 2 (pink diamond), and <italic>kir</italic> (blue triangles) are shown. SICAvar type 1 and <italic>kir</italic> genes had a significantly greater distance to their neighboring genes than the <italic>BUSCO</italic> genes (Wilcoxon signed-rank test, Bonferroni <italic>p</italic>-value adjustment, <italic>p &#x3d;</italic> &#x3c; 1e-09), suggesting that these gene family members are in gene-sparse regions. Genes situated at the start or end of scaffolds were excluded from the analysis. <xref ref-type="fig" rid="F5">Figure 5B</xref> OrthoFinder gene cluster outputs were visualized using &#x201c;UpSets&#x201d; to determine the membership of genes between clusters in the PKNH reference genome (<xref ref-type="bibr" rid="B70">Pain et al., 2008</xref>) and the draft genomes; StAPkA1H1 experimental cultured line, clinical isolate sk047, and clinical isolate sk048. All gene clusters (i), <italic>SICAvar</italic> type 1 gene clusters (ii), <italic>SICAvar</italic> type 2 gene clusters (iii), <italic>kir</italic> gene clusters (iv), and <italic>BUSCO</italic> gene clusters (v) are shown. The majority of all gene clusters were present in all isolates with the exception of <italic>SICAvar</italic> type 1 gene clusters with 10&#x2013;15 <italic>SICAvar</italic> type 1 clusters being unique per isolate. For <italic>kir</italic> genes, the majority of clusters were shared between all isolates with the exception of a single unique <italic>kir</italic> gene cluster in each of sk047 and sk048. The majority of <italic>SICAvar</italic> type 2 genes were orthologues between all isolates with some not identified in sk047 and sk048.</p>
</caption>
<graphic xlink:href="fgene-13-855052-g005.tif"/>
</fig>
<p>OrthoFinder gene cluster outputs were further visualized using &#x201c;UpSets&#x201d; to determine the membership of genes within each cluster. The majority of all gene clusters were present in all isolates with the exception of <italic>SICAvar</italic> type 1 clusters with between 10&#x2013;15 unique <italic>SICAvar</italic> type 1 clusters per isolate (<xref ref-type="fig" rid="F5">Figure 5B</xref>). For <italic>kir</italic> genes, the majority of clusters were shared between all isolates with the exception of a single unique <italic>kir</italic> gene cluster in each of sk047 and sk048. The majority of <italic>SICAvar</italic> type 2 genes were orthologues between all isolates with some not identified in sk047 and sk048 (<xref ref-type="fig" rid="F5">Figure 5B</xref>).</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<p>Here, we demonstrate the utility of accessible, portable, and affordable PCR-free long-read ONT MinION sequencing to <italic>de novo</italic> assemble <italic>P. knowlesi</italic> genomes from small clinical samples, essentially wild-type parasites. The new genome sequences are robust and add context to our understanding of <italic>P. knowlesi</italic> genome structure, organization, and variability.</p>
<p>Three <italic>Plasmodium knowlesi</italic> draft genomes were assembled from two <italic>P. knowlesi</italic> clinical isolates (sks047 and sks048), and the other was a control genome from the <italic>P. knowlesi</italic> A1-H.1 (StAPkA1H1) experimental cultured line (<xref ref-type="bibr" rid="B61">Moon et al., 2013</xref>). Comparison of the <italic>de novo</italic> StAPkA1H1 genome assembled here with the <italic>P. knowlesi</italic> A.1-H1 genome generated using Illumina and PacBio platforms (<xref ref-type="bibr" rid="B9">Benavente et al., 2018</xref>) and the <italic>P. knowlesi</italic> reference genome PKNH (<xref ref-type="bibr" rid="B70">Pain et al., 2008</xref>) demonstrated that our sequencing platform and subsequent assembly pipeline produced robust and reliable <italic>de novo P. knowlesi</italic> genome sequences.</p>
<p>The two clinical isolates (sks047 and sks048) and the control (StAPkA1H1) resolved into 14 chromosomes as expected for <italic>Plasmodium</italic> spp. and one &#x201c;bin&#x201d; chr00. The PKNH reference genome also resolves into 14 chromosomes and one chr00 where 1.73% of the total sequence comprising 62 genes was assigned (<xref ref-type="bibr" rid="B70">Pain et al., 2008</xref>). Chr00 of StAPkA1H1, sks047, and sks048 contain 1.59%, 2.09%, and 1.94% total sequence length with 18, 35, and 25 genes, respectively. Failure of sequences to pass quality thresholds would be expected to be randomly distributed genome-wide as observed in sks047 and sks048 chr00 sequences. The observed clustering of StAPkA1H1 chr00 with PKNH chr00 is difficult to explain unless <italic>de novo</italic> chromosome structuring was being overridden, forcing StAPkA1H1 contigs into a chr00 to fit the pattern set by the PKNH reference genome (<xref ref-type="bibr" rid="B7">Assefa et al., 2015</xref>; <xref ref-type="bibr" rid="B80">Steinbiss et al., 2016</xref>).</p>
<p>During chromosome structuring, we found that the minimap2 alignment function of RagTag was unable to resolve chimeric contigs for sks047, sks048, and StAPkA1H1, perhaps as a function of the algorithm heuristics in minimap2 or localized flaws in our pipeline. Consequently, sections of sks047 chromosomes 02 and 03, which were incorrectly placed in chromosomes 14 and 13 due to chimeric contigs, were successfully corrected using the nucmer aligner function of RagTag.</p>
<p>In general, RagTag struggled to resolve regions of low complexity and high variability, such as telomeric regions, although we report predicted genes within these telomeric regions, including some members of the <italic>SICAvar</italic> gene family and those described by <xref ref-type="bibr" rid="B51">Lapp et al. (2018)</xref>. More strikingly, the <italic>Duffy-binding protein</italic> and <italic>TrpRA</italic> genes are almost exclusively located at the extreme ends of the <italic>de novo</italic> assembled genomes presented here. Indeed, <xref ref-type="bibr" rid="B67">Otto et al., 2018</xref> reported that Companion, as used here, can construct <italic>Plasmodium</italic> chromosomes in their entirety (<xref ref-type="bibr" rid="B67">Otto et al., 2018</xref>).</p>
<p>The published PKNH <italic>P. knowlesi</italic> genome (<xref ref-type="bibr" rid="B70">Pain et al., 2008</xref>) and <italic>de novo</italic> assembled StPkA1H1 genome have a similar compliment of Interspersed Repeat (IR) genes while the clinical samples are similar to each other but quite different to the experimental parasite lines. The clinical isolates have approximately half the number of <italic>kir</italic> genes compared with PKNH and StPkA1H1. Unexpectedly, the clinical isolates have IR genes annotated as <italic>P. vivax</italic> (<italic>vir</italic>) that are absent in PKNH and StPkA1H1. The most parsimonious explanation for this difference is that the data from <italic>P. vivax vir</italic> genes are derived from clinical samples, and there are no well-established experimental lines for <italic>P. vivax</italic>. Published <italic>virs</italic> may more closely reflect IR diversity accrued in contemporary parasites from the Old World monkey parasite clade that includes <italic>P. knowlesi</italic> and <italic>P. vivax</italic> (<xref ref-type="bibr" rid="B78">Singh et al., 2004</xref>). Nonetheless, it is an interesting observation that deserves further investigation.</p>
<p>
<italic>BUSCO</italic> genes, with a similar duplication composition in the sks047, sks048, and StAPkA1H1 draft genomes, were used to compare duplication profiles for <italic>SICAvar</italic> type 1, <italic>SICAvar</italic> type 2, and <italic>kir</italic> gene family members that code for antigenically variable parasite proteins expressed on the surface of infected host red blood cells. The <italic>SICAvar</italic> type 1<italic>, SICAvar</italic> type 2, and <italic>kir</italic> gene population in all three draft genomes had significantly different duplication profiles when compared with 100 randomly selected genes (Mann&#x2013;Whitney U test: <italic>p</italic> &#x3c;&#x3c; 0.001). This suggests that the parasite genome tolerates high levels of duplication at these loci. Non-synonymous substitution over synonymous substitution (dN/dS) values greater than 1.0 are associated with positive selection pressure. <italic>BUSCO</italic> core eukaryotic genes are not thought to be under undue selection pressure and were used as a control gene set in dN/dS analysis to investigate selection pressure on clusters containing <italic>SICAvar</italic> type 1, <italic>SICAvar</italic> type 2, and <italic>kir</italic> genes. The mean dN/dS was 2.4 for <italic>SICAvar</italic> type 1 gene clusters, 2.74 for <italic>SICAvar</italic> type 2 clusters, and 2.35 for <italic>kir</italic> gene clusters while dN/dS scores for <italic>BUSCO</italic> gene clusters was 0.35, suggesting that the <italic>SICAvar</italic> type 1<italic>, SICAvar</italic> type 2, and <italic>kir</italic> gene populations are under strong positive selection pressure. Given that the protein products of these multiple gene family members are expressed at the forefront of parasite&#x2013;host interactions, positive selection in addition to the gene duplication profiles observed would be expected to accommodate antigenic variability and increase the chance of parasite survival in a hostile host environment.</p>
<p>On the backdrop of signatures of change and variability observed and the potential for genome destabilization at these rapidly evolving loci, the distribution of the <italic>SICAvar</italic> and <italic>kir</italic> genes within chromosomes seemed counterintuitive. Indeed, the ability of <italic>P. falciparum</italic> to tolerate the highly evolving <italic>PfEMP</italic> 1 gene family members is explained by their positioning in the extreme sub-telomeric regions of chromosomes that support higher rates of recombination in comparison to relatively more conserved centromeric regions (<xref ref-type="bibr" rid="B67">Otto et al., 2018</xref>). In the case of <italic>P. knowlesi</italic>, we found the rapidly evolving <italic>SICAvar</italic> and <italic>kir</italic> genes positioned in otherwise gene-sparse regions of chromosomes. With the exception of <italic>SICAvar</italic> type 2 genes in the 3&#x2019; direction, <italic>SICAvar</italic> and <italic>kir</italic> genes had significantly greater distances to neighboring genes in the 3&#x2019; and 5&#x2019; directions than <italic>BUSCO</italic> genes. Gene-sparse regions tolerate transposon and repetitive rich regions necessary to generate antigenic variability at these important loci while reducing the probability of impacting essential core gene function. Similar protective positioning of highly evolving genes is found in plant pathogens, for example, nematodes (<xref ref-type="bibr" rid="B30">Eves-van den Akker et al., 2016</xref>), aphids (<xref ref-type="bibr" rid="B82">Thorpe et al., 2018</xref>), phytophthora (<xref ref-type="bibr" rid="B39">Haas et al., 2009</xref>; <xref ref-type="bibr" rid="B84">Thorpe et al., 2021</xref>), and fungi (<xref ref-type="bibr" rid="B26">Dong et al., 2015</xref>). The capacity of some genomic regions to generate more variation than others is poorly understood, but in the field of plant pathogens, it is termed &#x201c;the two speed genome&#x201d; (<xref ref-type="bibr" rid="B26">Dong et al., 2015</xref>). The &#x201c;two speed genome&#x201d; concept may well describe accumulation of multiple gene family members in <italic>Plasmodium</italic> species, particularly the <italic>var</italic> genes, and consequently provide a biological model with which to explain antigenic variation in <italic>P. knowlesi</italic>.</p>
<p>To further demonstrate <italic>SICAvar</italic> type 1 genetic divergence, UpSet visualization of each of the draft genomes assembled here had between 10 and 15 unique <italic>SICAvar</italic> type 1 gene clusters, more than any other orthologous gene cluster. Indeed, only two <italic>SICAvar</italic> type 1 geneclusters were shared among the draft genomes. In contrast, the <italic>kir</italic> genes were less divergent with only one unique gene-cluster in sk048 and in sk047 with most <italic>kir</italic> gene clusters common between clinical isolates and experimental lines.</p>
<p>The ability to generate variation and maintain fitness is fundamental to pathogen&#x2013;host interactions. The pathogen needs a lifespan long enough to replicate, disseminate, and maintain germlines. The ability to generate diversity on genes that code for &#x201c;exposed&#x201d; proteins while protecting core gene function increases the chance of pathogen survival. The strong signatures of positive selection pressure and gene duplication on the <italic>P. knowlesi SICAvar</italic> type 1, <italic>SICAvar</italic> type 2, and <italic>kir</italic> genes irrefutably demonstrate their importance in the fitness and evolution of this particular pathogen. The methods developed here will be used to generate <italic>P. knowlesi</italic> genomes from patient isolates with matched metadata for parasite genome-wide disease association analyses. Experimental <italic>Plasmodium knowlesi</italic> is particularly receptive to genome editing, facilitating allele-specific phenotyping (<xref ref-type="bibr" rid="B60">Mohring et al., 2020</xref>). Parasites edited with clinically relevant disease-associated alleles can be taken forward and characterized <italic>in vitro</italic> and <italic>in vivo</italic> for cause and effect. In essence, <italic>P. knowlesi</italic> as the agent of zoonotic malaria and as an experimental parasite has the potential to closely model severe malaria pathophysiology.</p>
<p>On a broader landscape, an opportunity is presented to the global research community to generate genome-wide data from clinical infections to add &#x201c;real world&#x201d; context to malaria research.</p>
</sec>
</body>
<back>
<sec id="s5">
<title>Data Availability Statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found below: Sequencing data can be found at NCBI SRA BioProject, accession no: PRJNA799698; Scripts used to generate the data in this project are available in github: <ext-link ext-link-type="uri" xlink:href="https://github.com/damioresegun/Pknowlesi_denovo_genome_assembly">https://github.com/damioresegun/Pknowlesi_denovo_genome_assembly</ext-link> and <ext-link ext-link-type="uri" xlink:href="https://github.com/peterthorpe5/plasmidium_genomes">https://github.com/peterthorpe5/plasmidium_genomes</ext-link>.</p>
</sec>
<sec id="s6">
<title>Ethics Statement</title>
<p>The studies involving human participants were reviewed and approved by the University of St. Andrews Teaching and Research Ethics Committee. The patients/participants provided their written informed consent to participate in this study.</p>
</sec>
<sec id="s7">
<title>Author Contributions</title>
<p>DRO: data curation, formal analyses, investigation, methodology, and visualization. PT: formal analyses, software, investigation, and supervision. EDB: supervision. SC: resources. FM: resources. RWM: resources and draft editing. TGC: supervision, writing, and editing. JC-S: conceptualization, funding acquisition, methodology, project administration, resources, supervision, and writing the manuscript.</p>
</sec>
<sec id="s8">
<title>Funding</title>
<p>DRO is supported by the Wellcome Trust ISSF award 204821/Z/16/Z. Bioinformatics and computational biology analyses were supported by the University of St. Andrews Bioinformatics Unit (AMD3BIOINF), funded by Wellcome Trust ISSF awards 105621/Z/14/Z and 204821/Z/16/Z. The sample BioBank was compiled with informed consent (Medial Research Council, <ext-link ext-link-type="uri" xlink:href="http://www.mrc.ac.uk/">www.mrc.ac.uk</ext-link>, grant G0801971). Genome sequencing was supported by Tenovus Scotland (T16/03). TGC is funded by the Medical Research Council United Kingdom (grant nos. MR/M01360X/1, MR/N010469/1, MR/R025576/1, and MR/R020973/1) and BBSRC (grant no. BB/R013063/1). SC is funded by Medical Research Council United Kingdom grants (Refs. MR/M01360X/1, MR/R025576/1, and MR/R020973/1). RWM was supported by the UK Medical Research Council Career Award MR/M021157/1.</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors, and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ack>
<p>We would like to thank Joseph Ward for help with software and resources, Fiona Cook for providing resources for optimizing methodologies, and Cyrus J. Daneshvar for critically reading the manuscript.</p>
</ack>
<sec id="s11">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2022.855052/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2022.855052/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.zip" id="SM1" mimetype="application/zip" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abdi</surname>
<given-names>A. I.</given-names>
</name>
<name>
<surname>Hodgson</surname>
<given-names>S. H.</given-names>
</name>
<name>
<surname>Muthui</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Kivisi</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Kamuyu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Kimani</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Plasmodium Falciparum Malaria Parasite Var Gene Expression Is Modified by Host Antibodies: Longitudinal Evidence from Controlled Infections of Kenyan Adults with Varying Natural Exposure</article-title>. <source>BMC Infect. Dis.</source> <volume>17</volume> (<issue>1</issue>), <fpage>585</fpage>. <pub-id pub-id-type="doi">10.1186/s12879-017-2686-0</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ahmed</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Pinheiro</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Divis</surname>
<given-names>P. C.</given-names>
</name>
<name>
<surname>Siner</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zainudin</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>I. T.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Disease Progression in Plasmodium Knowlesi Malaria Is Linked to Variation in Invasion Gene Family Members</article-title>. <source>PLoS Negl. Trop. Dis.</source> <volume>8</volume> (<issue>8</issue>), <fpage>e3086</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pntd.0003086</pub-id> </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ahmed</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Quan</surname>
<given-names>F. S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Plasmodium Knowlesi Clinical Isolates from Malaysia Show Extensive Diversity and Strong Differential Selection Pressure at the Merozoite Surface Protein 7D (MSP7D)</article-title>. <source>Malar. J.</source> <volume>18</volume> (<issue>1</issue>), <fpage>150</fpage>. <pub-id pub-id-type="doi">10.1186/s12936-019-2782-2</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Al-Khedery</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Barnwell</surname>
<given-names>J. W.</given-names>
</name>
<name>
<surname>Galinski</surname>
<given-names>M. R.</given-names>
</name>
</person-group> (<year>1999</year>). <article-title>Antigenic Variation in Malaria: a 3&#x27; Genomic Alteration Associated with the Expression of a P. Knowlesi Variant Antigen</article-title>. <source>Mol. Cell</source> <volume>3</volume> (<issue>2</issue>), <fpage>131</fpage>&#x2013;<lpage>141</lpage>. <pub-id pub-id-type="doi">10.1016/s1097-2765(00)80304-4</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alonge</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Soyk</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ramakrishnan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Goodwin</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sedlazeck</surname>
<given-names>F. J.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>RaGOO: Fast and Accurate Reference-Guided Scaffolding of Draft Genomes</article-title>. <source>Genome Biol.</source> <volume>20</volume> (<issue>1</issue>), <fpage>224</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-019-1829-6</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Andrade</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Fleckenstein</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Thomson-Luque</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Doumbo</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lima</surname>
<given-names>N. F.</given-names>
</name>
<name>
<surname>Anderson</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Increased Circulation Time of Plasmodium Falciparum Underlies Persistent Asymptomatic Infection in the Dry Season</article-title>. <source>Nat. Med.</source> <volume>26</volume> (<issue>12</issue>), <fpage>1929</fpage>&#x2013;<lpage>1940</lpage>. <pub-id pub-id-type="doi">10.1038/s41591-020-1084-0</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Assefa</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lim</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Preston</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>Duffy</surname>
<given-names>C. W.</given-names>
</name>
<name>
<surname>Nair</surname>
<given-names>M. B.</given-names>
</name>
<name>
<surname>Adroub</surname>
<given-names>S. A.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Population Genomic Structure and Adaptation in the Zoonotic Malaria Parasite Plasmodium Knowlesi</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A.</source> <volume>112</volume> (<issue>42</issue>), <fpage>13027</fpage>&#x2013;<lpage>13032</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1509534112</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bailly-Bechet</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Haudry</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lerat</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>&#x201c;One Code to Find Them All&#x201d;: a Perl Tool to Conveniently Parse RepeatMasker Output Files</article-title>. <source>Mob. DNA</source> <volume>5</volume> (<issue>1</issue>), <fpage>13</fpage>. <pub-id pub-id-type="doi">10.1186/1759-8753-5-13</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Benavente</surname>
<given-names>E. D.</given-names>
</name>
<name>
<surname>de Sessions</surname>
<given-names>P. F.</given-names>
</name>
<name>
<surname>Moon</surname>
<given-names>R. W.</given-names>
</name>
<name>
<surname>Grainger</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Holder</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Blackman</surname>
<given-names>M. J.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>A Reference Genome and Methylome for the Plasmodium Knowlesi A1-H.1 Line</article-title>. <source>Int. J. Parasitol.</source> <volume>48</volume> (<issue>3-4</issue>), <fpage>191</fpage>&#x2013;<lpage>196</lpage>. <pub-id pub-id-type="doi">10.1016/j.ijpara.2017.09.008</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Benavente</surname>
<given-names>E. D.</given-names>
</name>
<name>
<surname>Gomes</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>De Silva</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>Grigg</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Walker</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Barber</surname>
<given-names>B. E.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Whole Genome Sequencing of Amplified Plasmodium Knowlesi DNA from Unprocessed Blood Reveals Genetic Exchange Events between Malaysian Peninsular and Borneo Subpopulations</article-title>. <source>Sci. Rep.</source> <volume>9</volume> (<issue>1</issue>), <fpage>9873</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-019-46398-z</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Butcher</surname>
<given-names>G. A.</given-names>
</name>
<name>
<surname>Mitchell</surname>
<given-names>G. H.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>The Role of Plasmodium Knowlesi in the History of Malaria Research</article-title>. <source>Parasitology</source> <volume>145</volume> (<issue>1</issue>), <fpage>6</fpage>&#x2013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.1017/S0031182016001888</pub-id> </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cabanettes</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Klopp</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>D-GENIES: Dot Plot Large Genomes in an Interactive, Efficient and Simple Way</article-title>. <source>PeerJ</source> <volume>6</volume>, <fpage>e4958</fpage>. <pub-id pub-id-type="doi">10.7717/peerj.4958</pub-id> </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Capella-Gutierrez</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Silla-Martinez</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Gabaldon</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>trimAl: a Tool for Automated Alignment Trimming in Large-Scale Phylogenetic Analyses</article-title>. <source>Bioinformatics</source> <volume>25</volume> (<issue>15</issue>), <fpage>1972</fpage>&#x2013;<lpage>1973</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp348</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chin</surname>
<given-names>A. Z.</given-names>
</name>
<name>
<surname>Maluda</surname>
<given-names>M. C. M.</given-names>
</name>
<name>
<surname>Jelip</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jeffree</surname>
<given-names>M. S. B.</given-names>
</name>
<name>
<surname>Culleton</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Ahmed</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Malaria Elimination in Malaysia and the Rising Threat of Plasmodium Knowlesi</article-title>. <source>J. Physiol. Anthropol.</source> <volume>39</volume> (<issue>1</issue>), <fpage>36</fpage>. <pub-id pub-id-type="doi">10.1186/s40101-020-00247-5</pub-id> </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chin</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Contacos</surname>
<given-names>P. G.</given-names>
</name>
<name>
<surname>Coatney</surname>
<given-names>G. R.</given-names>
</name>
<name>
<surname>Kimball</surname>
<given-names>H. R.</given-names>
</name>
</person-group> (<year>1965</year>). <article-title>A Naturally Acquited Quotidian-type Malaria in Man Transferable to Monkeys</article-title>. <source>Science</source> <volume>149</volume> (<issue>3686</issue>), <fpage>865</fpage>. <pub-id pub-id-type="doi">10.1126/science.149.3686.86510.1126/science.149.3686.865.a</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chin</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Contacos</surname>
<given-names>P. G.</given-names>
</name>
<name>
<surname>Collins</surname>
<given-names>W. E.</given-names>
</name>
<name>
<surname>Jeter</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Alpert</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>1968</year>). <article-title>Experimental Mosquito-Transmission of Plasmodium Knowlesi to Man and Monkey</article-title>. <source>Am. J. Trop. Med. Hyg.</source> <volume>17</volume> (<issue>3</issue>), <fpage>355</fpage>&#x2013;<lpage>358</lpage>. <pub-id pub-id-type="doi">10.4269/ajtmh.1968.17.355</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cock</surname>
<given-names>P. J.</given-names>
</name>
<name>
<surname>Antao</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>J. T.</given-names>
</name>
<name>
<surname>Chapman</surname>
<given-names>B. A.</given-names>
</name>
<name>
<surname>Cox</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Dalke</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>Biopython: Freely Available Python Tools for Computational Molecular Biology and Bioinformatics</article-title>. <source>Bioinformatics</source> <volume>25</volume> (<issue>11</issue>), <fpage>1422</fpage>&#x2013;<lpage>1423</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp163</pub-id> </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cox-Singh</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Plasmodium Knowlesi: Experimental Model, Zoonotic Pathogen and Golden Opportunity?</article-title> <source>Parasitology</source> <volume>145</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>5</lpage>. <pub-id pub-id-type="doi">10.1017/S0031182017001858</pub-id> </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cox-Singh</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Culleton</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Plasmodium Knowlesi: from Severe Zoonosis to Animal Model</article-title>. <source>Trends Parasitol.</source> <volume>31</volume> (<issue>6</issue>), <fpage>232</fpage>&#x2013;<lpage>238</lpage>. <pub-id pub-id-type="doi">10.1016/j.pt.2015.03.003</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cox-Singh</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Davis</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>K. S.</given-names>
</name>
<name>
<surname>Shamsul</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Matusop</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ratnam</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2008</year>). <article-title>Plasmodium Knowlesi Malaria in Humans Is Widely Distributed and Potentially Life Threatening</article-title>. <source>Clin. Infect. Dis.</source> <volume>46</volume> (<issue>2</issue>), <fpage>165</fpage>&#x2013;<lpage>171</lpage>. <pub-id pub-id-type="doi">10.1086/524888</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cox-Singh</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hiu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lucas</surname>
<given-names>S. B.</given-names>
</name>
<name>
<surname>Divis</surname>
<given-names>P. C.</given-names>
</name>
<name>
<surname>Zulkarnaen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chandran</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>Severe Malaria - a Case of Fatal Plasmodium Knowlesi Infection with Post-mortem Findings: a Case Report</article-title>. <source>Malar. J.</source> <volume>9</volume>, <fpage>10</fpage>. <pub-id pub-id-type="doi">10.1186/1475-2875-9-10</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Daneshvar</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Davis</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Cox-Singh</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Rafa&#x27;ee</surname>
<given-names>M. Z.</given-names>
</name>
<name>
<surname>Zakaria</surname>
<given-names>S. K.</given-names>
</name>
<name>
<surname>Divis</surname>
<given-names>P. C.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>Clinical and Laboratory Features of Human Plasmodium Knowlesi Infection</article-title>. <source>Clin. Infect. Dis.</source> <volume>49</volume> (<issue>6</issue>), <fpage>852</fpage>&#x2013;<lpage>860</lpage>. <pub-id pub-id-type="doi">10.1086/605439</pub-id> </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Daneshvar</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>William</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Davis</surname>
<given-names>T. M. E.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Clinical Features and Management of Plasmodium Knowlesi Infections in Humans</article-title>. <source>Parasitology</source> <volume>145</volume> (<issue>1</issue>), <fpage>18</fpage>&#x2013;<lpage>31</lpage>. <pub-id pub-id-type="doi">10.1017/S0031182016002638</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Diez Benavente</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Florez de Sessions</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Moon</surname>
<given-names>R. W.</given-names>
</name>
<name>
<surname>Holder</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Blackman</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Roper</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Analysis of Nuclear and Organellar Genomes of Plasmodium Knowlesi in Humans Reveals Ancient Population Structure and Recent Recombination Among Host-specific Subpopulations</article-title>. <source>PLOS Genet.</source> <volume>13</volume> (<issue>9</issue>), <fpage>e1007008</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pgen.1007008</pub-id> </citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Divis</surname>
<given-names>P. C. S.</given-names>
</name>
<name>
<surname>Duffy</surname>
<given-names>C. W.</given-names>
</name>
<name>
<surname>Kadir</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Conway</surname>
<given-names>D. J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Genome-wide Mosaicism in Divergence between Zoonotic Malaria Parasite Subpopulations with Separate Sympatric Transmission Cycles</article-title>. <source>Mol. Ecol.</source> <volume>27</volume> (<issue>4</issue>), <fpage>860</fpage>&#x2013;<lpage>870</lpage>. <pub-id pub-id-type="doi">10.1111/mec.14477</pub-id> </citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dong</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Raffaele</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kamoun</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>The Two-Speed Genomes of Filamentous Pathogens: Waltz with Plants</article-title>. <source>Curr. Opin. Genet. Dev.</source> <volume>35</volume>, <fpage>57</fpage>&#x2013;<lpage>65</lpage>. <pub-id pub-id-type="doi">10.1016/j.gde.2015.09.001</pub-id> </citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Edgar</surname>
<given-names>R. C.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>MUSCLE: Multiple Sequence Alignment with High Accuracy and High Throughput</article-title>. <source>Nucleic Acids Res.</source> <volume>32</volume> (<issue>5</issue>), <fpage>1792</fpage>&#x2013;<lpage>1797</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkh340</pub-id> </citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ellinghaus</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Kurtz</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Willhoeft</surname>
<given-names>U.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>LTRharvest, an Efficient and Flexible Software for De Novo Detection of LTR Retrotransposons</article-title>. <source>BMC Bioinforma.</source> <volume>9</volume>, <fpage>18</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-9-18</pub-id> </citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Emms</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Kelly</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>OrthoFinder: Phylogenetic Orthology Inference for Comparative Genomics</article-title>. <source>Genome Biol.</source> <volume>20</volume> (<issue>1</issue>), <fpage>238</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-019-1832-y</pub-id> </citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Eves-van den Akker</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Laetsch</surname>
<given-names>D. R.</given-names>
</name>
<name>
<surname>Thorpe</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Lilley</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Danchin</surname>
<given-names>E. G.</given-names>
</name>
<name>
<surname>Da Rocha</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>The Genome of the Yellow Potato Cyst Nematode, Globodera Rostochiensis, Reveals Insights into the Basis of Parasitism and Virulence</article-title>. <source>Genome Biol.</source> <volume>17</volume> (<issue>1</issue>), <fpage>124</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-016-0985-1</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Flynn</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Hubley</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Goubert</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Rosen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Clark</surname>
<given-names>A. G.</given-names>
</name>
<name>
<surname>Feschotte</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>RepeatModeler2 for Automated Genomic Discovery of Transposable Element Families</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A.</source> <volume>117</volume> (<issue>17</issue>), <fpage>9451</fpage>&#x2013;<lpage>9457</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1921046117</pub-id> </citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fong</surname>
<given-names>M. Y.</given-names>
</name>
<name>
<surname>Lau</surname>
<given-names>Y. L.</given-names>
</name>
<name>
<surname>Jelip</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ooi</surname>
<given-names>C. H.</given-names>
</name>
<name>
<surname>Cheong</surname>
<given-names>F. W.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Genetic Characterisation of the Erythrocyte-Binding Protein (PkbetaII) of Plasmodium Knowlesi Isolates from Malaysia</article-title>. <source>J. Genet.</source> <volume>98</volume>. <pub-id pub-id-type="doi">10.1007/s12041-019-1109-y</pub-id> </citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Niu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>CD-HIT: Accelerated for Clustering the Next-Generation Sequencing Data</article-title>. <source>Bioinforma. Oxf. Engl.</source> <volume>28</volume> (<issue>23</issue>), <fpage>3150</fpage>&#x2013;<lpage>3152</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bts565</pub-id> </citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Galinski</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Lapp</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Peterson</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Ay</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Joyner</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Kg</surname>
<given-names>L. E. R.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Plasmodium Knowlesi: a Superb <italic>In Vivo</italic> Nonhuman Primate Model of Antigenic Variation in Malaria</article-title>. <source>Parasitology</source> <volume>145</volume> (<issue>1</issue>), <fpage>85</fpage>&#x2013;<lpage>100</lpage>. <pub-id pub-id-type="doi">10.1017/S0031182017001135</pub-id> </citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gardner</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Hall</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Fung</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>White</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Berriman</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hyman</surname>
<given-names>R. W.</given-names>
</name>
<etal/>
</person-group> (<year>2002</year>). <article-title>Genome Sequence of the Human Malaria Parasite Plasmodium Falciparum</article-title>. <source>Nature</source> <volume>419</volume> (<issue>6906</issue>), <fpage>498</fpage>&#x2013;<lpage>511</lpage>. <pub-id pub-id-type="doi">10.1038/nature01097</pub-id> </citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gel</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Serra</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>karyoploteR: an R/Bioconductor Package to Plot Customizable Genomes Displaying Arbitrary Data</article-title>. <source>Bioinformatics</source> <volume>33</volume> (<issue>19</issue>), <fpage>3088</fpage>&#x2013;<lpage>3090</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btx346</pub-id> </citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gil</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zanetti</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Zoller</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Anisimova</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>CodonPhyML: Fast Maximum Likelihood Phylogeny Estimation under Codon Substitution Models</article-title>. <source>Mol. Biol. Evol.</source> <volume>30</volume> (<issue>6</issue>), <fpage>1270</fpage>&#x2013;<lpage>1280</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/mst034</pub-id> </citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gremme</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Steinbiss</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kurtz</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>GenomeTools: a Comprehensive Software Library for Efficient Processing of Structured Genome Annotations</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinform</source> <volume>10</volume> (<issue>3</issue>), <fpage>645</fpage>&#x2013;<lpage>656</lpage>. <pub-id pub-id-type="doi">10.1109/tcbb.2013.68</pub-id> </citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Haas</surname>
<given-names>B. J.</given-names>
</name>
<name>
<surname>Kamoun</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zody</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>R. H.</given-names>
</name>
<name>
<surname>Handsaker</surname>
<given-names>R. E.</given-names>
</name>
<name>
<surname>Cano</surname>
<given-names>L. M.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>Genome Sequence and Analysis of the Irish Potato Famine Pathogen Phytophthora Infestans</article-title>. <source>Nature</source> <volume>461</volume> (<issue>7262</issue>), <fpage>393</fpage>&#x2013;<lpage>398</lpage>. <pub-id pub-id-type="doi">10.1038/nature08358</pub-id> </citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Harrison</surname>
<given-names>T. E.</given-names>
</name>
<name>
<surname>Reid</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Cunningham</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Langhorne</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Higgins</surname>
<given-names>M. K.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Structure of the Plasmodium-Interspersed Repeat Proteins of the Malaria Parasite</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A.</source> <volume>117</volume> (<issue>50</issue>), <fpage>32098</fpage>&#x2013;<lpage>32104</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.2016775117</pub-id> </citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Heather</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Chain</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>The Sequence of Sequencers: The History of Sequencing DNA</article-title>. <source>Genomics</source> <volume>107</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1016/j.ygeno.2015.11.003</pub-id> </citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hunt</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Silva</surname>
<given-names>N. D.</given-names>
</name>
<name>
<surname>Otto</surname>
<given-names>T. D.</given-names>
</name>
<name>
<surname>Parkhill</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Keane</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Harris</surname>
<given-names>S. R.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Circlator: Automated Circularization of Genome Assemblies Using Long Sequencing Reads</article-title>. <source>Genome Biol.</source> <volume>16</volume>, <fpage>294</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-015-0849-0</pub-id> </citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hviid</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Jensen</surname>
<given-names>A. T.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>PfEMP1-A Parasite Protein Family of Key Importance in Plasmodium Falciparum Malaria Immunity and Pathogenesis</article-title>. <source>Adv. Parasitol.</source> <volume>88</volume>, <fpage>51</fpage>&#x2013;<lpage>84</lpage>. <pub-id pub-id-type="doi">10.1016/bs.apar.2015.02.004</pub-id> </citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jeffares</surname>
<given-names>D. C.</given-names>
</name>
<name>
<surname>Jolly</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Hoti</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Speed</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Shaw</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Rallis</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Transient Structural Variations Have Strong Effects on Quantitative Traits and Reproductive Isolation in Fission Yeast</article-title>. <source>Nat. Commun.</source> <volume>8</volume> (<issue>1</issue>), <fpage>14061</fpage>. <pub-id pub-id-type="doi">10.1038/ncomms14061</pub-id> </citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jensen</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Adams</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hviid</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Cerebral Plasmodium Falciparum Malaria: The Role of PfEMP1 in its Pathogenesis and Immunity, and PfEMP1-Based Vaccines to Prevent it</article-title>. <source>Immunol. Rev.</source> <volume>293</volume> (<issue>1</issue>), <fpage>230</fpage>&#x2013;<lpage>252</lpage>. <pub-id pub-id-type="doi">10.1111/imr.12807</pub-id> </citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Long-read-based Human Genomic Structural Variation Detection with cuteSV</article-title>. <source>Genome Biol.</source> <volume>21</volume> (<issue>1</issue>), <fpage>189</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-020-02107-y</pub-id> </citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Knowles</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>B. M. D.</given-names>
</name>
</person-group> (<year>1932</year>). <article-title>A Study of Monkey-Malaria, and its Experimental Transmission to Man</article-title>. <source>Ind. Med. Gaz.</source> <volume>67</volume> (<issue>6</issue>), <fpage>301</fpage>&#x2013;<lpage>320</lpage>. </citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kohany</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Gentles</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Hankus</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Jurka</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Annotation, Submission and Screening of Repetitive Elements in Repbase: RepbaseSubmitter and Censor</article-title>. <source>BMC Bioinforma.</source> <volume>7</volume> (<issue>1</issue>), <fpage>474</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-7-474</pub-id> </citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kolmogorov</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Pevzner</surname>
<given-names>P. A.</given-names>
</name>
</person-group>. (<year>2019</year>). <article-title>Assembly of Long, Error-Prone Reads Using Repeat Graphs</article-title>. <source>Nat. Biotechnol.</source> <volume>37</volume> (<issue>5</issue>), <fpage>540</fpage>&#x2013;<lpage>546</lpage>. <pub-id pub-id-type="doi">10.1038/s41587-019-0072-8</pub-id> </citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Laetsch</surname>
<given-names>D. R.</given-names>
</name>
<name>
<surname>Blaxter</surname>
<given-names>M. L.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>BlobTools: Interrogation of Genome Assemblies</article-title>. <source>F1000Research</source> <volume>6</volume>, <fpage>1287</fpage>. <pub-id pub-id-type="doi">10.12688/f1000research.12232.1</pub-id> </citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lander</surname>
<given-names>E. S.</given-names>
</name>
<name>
<surname>Linton</surname>
<given-names>L. M.</given-names>
</name>
<name>
<surname>Birren</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Nusbaum</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zody</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Baldwin</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> <collab>International Human Genome Sequencing Consortium</collab> (<year>2001</year>). <article-title>Initial Sequencing and Analysis of the Human Genome</article-title>. <source>Nature</source> <volume>409</volume> (<issue>6822</issue>), <fpage>860</fpage>&#x2013;<lpage>921</lpage>. <pub-id pub-id-type="doi">10.1038/35057062</pub-id> </citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lapp</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Geraldo</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Chien</surname>
<given-names>J. T.</given-names>
</name>
<name>
<surname>Ay</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Pakala</surname>
<given-names>S. B.</given-names>
</name>
<name>
<surname>Batugedara</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>PacBio Assembly of a Plasmodium Knowlesi Genome Sequence with Hi-C Correction and Manual Annotation of the SICAvar Gene Family</article-title>. <source>Parasitology</source> <volume>145</volume> (<issue>1</issue>), <fpage>71</fpage>&#x2013;<lpage>84</lpage>. <pub-id pub-id-type="doi">10.1017/S0031182017001329</pub-id> </citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lapp</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Mok</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Preiser</surname>
<given-names>P. R.</given-names>
</name>
<name>
<surname>Bozdech</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Plasmodium Knowlesi Gene Expression Differs in <italic>Ex Vivo</italic> Compared to <italic>In Vitro</italic> Blood-Stage Cultures</article-title>. <source>Malar. J.</source> <volume>14</volume>, <fpage>110</fpage>. <pub-id pub-id-type="doi">10.1186/s12936-015-0612-8</pub-id> </citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lavstsen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Turner</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Saguti</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Magistrado</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Rask</surname>
<given-names>T. S.</given-names>
</name>
<name>
<surname>Jespersen</surname>
<given-names>J. S.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Plasmodium Falciparum Erythrocyte Membrane Protein 1 Domain Cassettes 8 and 13 Are Associated with Severe Malaria in Children</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A.</source> <volume>109</volume> (<issue>26</issue>), <fpage>E1791</fpage>&#x2013;<lpage>E1800</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1120455109</pub-id> </citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>A Statistical Framework for SNP Calling, Mutation Discovery, Association Mapping and Population Genetical Parameter Estimation from Sequencing Data</article-title>. <source>Bioinforma. Oxf. Engl.</source> <volume>27</volume> (<issue>21</issue>), <fpage>2987</fpage>&#x2013;<lpage>2993</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btr509</pub-id> </citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Minimap2: Pairwise Alignment for Nucleotide Sequences</article-title>. <source>Bioinformatics</source> <volume>34</volume> (<issue>18</issue>), <fpage>3094</fpage>&#x2013;<lpage>3100</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty191</pub-id> </citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Handsaker</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wysoker</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Fennell</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Ruan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Homer</surname>
<given-names>N.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>The Sequence Alignment/Map Format and SAMtools</article-title>. <source>Bioinformatics</source> <volume>25</volume> (<issue>16</issue>), <fpage>2078</fpage>&#x2013;<lpage>2079</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp352</pub-id> </citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Godzik</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Cd-hit: a Fast Program for Clustering and Comparing Large Sets of Protein or Nucleotide Sequences</article-title>. <source>Bioinforma. Oxf. Engl.</source> <volume>22</volume> (<issue>13</issue>), <fpage>1658</fpage>&#x2013;<lpage>1659</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btl158</pub-id> </citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Milner</surname>
<given-names>D. A.</given-names>
<suffix>Jr</suffix>
</name>
</person-group> (<year>2018</year>). <article-title>Malaria Pathogenesis</article-title>. <source>Cold Spring Harb. Perspect. Med.</source> <volume>8</volume> (<issue>1</issue>), <fpage>a025569</fpage>. <pub-id pub-id-type="doi">10.1101/cshperspect.a025569</pub-id> </citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mohring</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Hart</surname>
<given-names>M. N.</given-names>
</name>
<name>
<surname>Patel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Baker</surname>
<given-names>D. A.</given-names>
</name>
<name>
<surname>Moon</surname>
<given-names>R. W.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>CRISPR-Cas9 Genome Editing of Plasmodium Knowlesi</article-title>. <source>Bio Protoc.</source> <volume>10</volume> (<issue>4</issue>), <fpage>e3522</fpage>. <pub-id pub-id-type="doi">10.21769/BioProtoc.3522</pub-id> </citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moon</surname>
<given-names>R. W.</given-names>
</name>
<name>
<surname>Hall</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Rangkuti</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Ho</surname>
<given-names>Y. S.</given-names>
</name>
<name>
<surname>Almond</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Mitchell</surname>
<given-names>G. H.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Adaptation of the Genetically Tractable Malaria Pathogen Plasmodium Knowlesi to Continuous Culture in Human Erythrocytes</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A.</source> <volume>110</volume> (<issue>2</issue>), <fpage>531</fpage>&#x2013;<lpage>536</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1216457110</pub-id> </citation>
</ref>
<ref id="B62">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Morgulis</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Coulouris</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Raytselis</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Madden</surname>
<given-names>T. L.</given-names>
</name>
<name>
<surname>Agarwala</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Sch&#xe4;ffer</surname>
<given-names>A. A.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Database Indexing for Production MegaBLAST Searches</article-title>. <source>Bioinformatics</source> <volume>24</volume> (<issue>16</issue>), <fpage>1757</fpage>&#x2013;<lpage>1764</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btn322</pub-id> </citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nattestad</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Schatz</surname>
<given-names>M. C.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Assemblytics: a Web Analytics Tool for the Detection of Variants from an Assembly</article-title>. <source>Bioinformatics</source> <volume>32</volume> (<issue>19</issue>), <fpage>3021</fpage>&#x2013;<lpage>3023</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btw369</pub-id> </citation>
</ref>
<ref id="B64">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Okonechnikov</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Conesa</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Garc&#xed;a-Alcalde</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Qualimap 2: Advanced Multi-Sample Quality Control for High-Throughput Sequencing Data</article-title>. <source>Bioinformatics</source> <volume>32</volume> (<issue>2</issue>), <fpage>292</fpage>&#x2013;<lpage>294</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btv566</pub-id> </citation>
</ref>
<ref id="B65">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Onditi</surname>
<given-names>F. I.</given-names>
</name>
<name>
<surname>Nyamongo</surname>
<given-names>O. W.</given-names>
</name>
<name>
<surname>Omwandho</surname>
<given-names>C. O.</given-names>
</name>
<name>
<surname>Maina</surname>
<given-names>N. W.</given-names>
</name>
<name>
<surname>Maloba</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Farah</surname>
<given-names>I. O.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Parasite Accumulation in Placenta of Non-immune Baboons during Plasmodium Knowlesi Infection</article-title>. <source>Malar. J.</source> <volume>14</volume>, <fpage>118</fpage>. <pub-id pub-id-type="doi">10.1186/s12936-015-0631-5</pub-id> </citation>
</ref>
<ref id="B66">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Oresegun</surname>
<given-names>D. R.</given-names>
</name>
<name>
<surname>Daneshvar</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Cox-Singh</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Plasmodium Knowlesi &#x2013; Clinical Isolate Genome Sequencing to Inform Translational Same-Species Model System for Severe Malaria</article-title>. <source>Front. Cell. Infect. Microbiol.</source> <volume>11</volume> (<issue>90</issue>), <fpage>607686</fpage>. <pub-id pub-id-type="doi">10.3389/fcimb.2021.607686</pub-id> </citation>
</ref>
<ref id="B67">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Otto</surname>
<given-names>T. D.</given-names>
</name>
<name>
<surname>Bohme</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Sanders</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Reid</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bruske</surname>
<given-names>E. I.</given-names>
</name>
<name>
<surname>Duffy</surname>
<given-names>C. W.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Long Read Assemblies of Geographically Dispersed Plasmodium Falciparum Isolates Reveal Highly Structured Subtelomeres</article-title>. <source>Wellcome Open Res.</source> <volume>3</volume>, <fpage>52</fpage>. <pub-id pub-id-type="doi">10.12688/wellcomeopenres.14571.1</pub-id> </citation>
</ref>
<ref id="B68">
<citation citation-type="book">
<collab>Oxford Nanopore Technologies</collab> (<year>2019</year>). <source>Medaka: Consensus Sequence Tool for Nanopore Sequences (version v0.6.5). Linux, Python. 2017</source>. <publisher-loc>Oxford</publisher-loc>: <publisher-name>Oxford Nanopore Technologies</publisher-name>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://github.com/nanoporetech/medaka">https://github.com/nanoporetech/medaka</ext-link>
</comment>. </citation>
</ref>
<ref id="B69">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ozwara</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Langermans</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Maamun</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Farah</surname>
<given-names>I. O.</given-names>
</name>
<name>
<surname>Yole</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Mwenda</surname>
<given-names>J. M.</given-names>
</name>
<etal/>
</person-group> (<year>2003</year>). <article-title>Experimental Infection of the Olive Baboon (Paplio Anubis) with Plasmodium Knowlesi: Severe Disease Accompanied by Cerebral Involvement</article-title>. <source>Am. J. Trop. Med. Hyg.</source> <volume>69</volume> (<issue>2</issue>), <fpage>188</fpage>&#x2013;<lpage>194</lpage>. <pub-id pub-id-type="doi">10.4269/ajtmh.2003.69.188</pub-id> </citation>
</ref>
<ref id="B70">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pain</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bohme</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Berry</surname>
<given-names>A. E.</given-names>
</name>
<name>
<surname>Mungall</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Finn</surname>
<given-names>R. D.</given-names>
</name>
<name>
<surname>Jackson</surname>
<given-names>A. P.</given-names>
</name>
<etal/>
</person-group> (<year>2008</year>). <article-title>The Genome of the Simian and Human Malaria Parasite Plasmodium Knowlesi</article-title>. <source>Nature</source> <volume>455</volume> (<issue>7214</issue>), <fpage>799</fpage>&#x2013;<lpage>803</lpage>. <pub-id pub-id-type="doi">10.1038/nature07306</pub-id> </citation>
</ref>
<ref id="B71">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pasini</surname>
<given-names>E. M.</given-names>
</name>
<name>
<surname>Zeeman</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Voorberg-Van Der Wel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kocken</surname>
<given-names>C. H. M.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Plasmodium Knowlesi: a Relevant, Versatile Experimental Malaria Model</article-title>. <source>Parasitology</source> <volume>145</volume> (<issue>1</issue>), <fpage>56</fpage>&#x2013;<lpage>70</lpage>. <pub-id pub-id-type="doi">10.1017/S0031182016002286</pub-id> </citation>
</ref>
<ref id="B72">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pedersen</surname>
<given-names>B. S.</given-names>
</name>
<name>
<surname>Quinlan</surname>
<given-names>A. R.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Mosdepth: Quick Coverage Calculation for Genomes and Exomes</article-title>. <source>Bioinformatics</source> <volume>34</volume> (<issue>5</issue>), <fpage>867</fpage>&#x2013;<lpage>868</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btx699</pub-id> </citation>
</ref>
<ref id="B73">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pinheiro</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Ahmed</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Millar</surname>
<given-names>S. B.</given-names>
</name>
<name>
<surname>Sanderson</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Otto</surname>
<given-names>T. D.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>W. C.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Plasmodium Knowlesi Genome Sequences from Clinical Isolates Reveal Extensive Genomic Dimorphism</article-title>. <source>PLoS One</source> <volume>10</volume> (<issue>4</issue>), <fpage>e0121303</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0121303</pub-id> </citation>
</ref>
<ref id="B74">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Quinlan</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Hall</surname>
<given-names>I. M.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>BEDTools: a Flexible Suite of Utilities for Comparing Genomic Features</article-title>. <source>Bioinformatics</source> <volume>26</volume> (<issue>6</issue>), <fpage>841</fpage>&#x2013;<lpage>842</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btq033</pub-id> </citation>
</ref>
<ref id="B75">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ren</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chaisson</surname>
<given-names>M. J. P.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Lra: the Long Read Aligner for Sequences and Contigs</article-title>. <source>PLoS Comput. Biol.</source> <volume>17</volume> (<issue>6</issue>), <fpage>e1009078</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1009078</pub-id> </citation>
</ref>
<ref id="B76">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shabani</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Hanisch</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Opoka</surname>
<given-names>R. O.</given-names>
</name>
<name>
<surname>Lavstsen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>John</surname>
<given-names>C. C.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Plasmodium Falciparum EPCR-Binding PfEMP1 Expression Increases with Malaria Disease Severity and Is Elevated in Retinopathy Negative Cerebral Malaria</article-title>. <source>BMC Med.</source> <volume>15</volume> (<issue>1</issue>), <fpage>183</fpage>. <pub-id pub-id-type="doi">10.1186/s12916-017-0945-y</pub-id> </citation>
</ref>
<ref id="B77">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sim&#xe3;o</surname>
<given-names>F. A.</given-names>
</name>
<name>
<surname>Waterhouse</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Ioannidis</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Kriventseva</surname>
<given-names>E. V.</given-names>
</name>
<name>
<surname>Zdobnov</surname>
<given-names>E. M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>BUSCO: Assessing Genome Assembly and Annotation Completeness with Single-Copy Orthologs</article-title>. <source>Bioinformatics</source> <volume>31</volume> (<issue>19</issue>), <fpage>3210</fpage>&#x2013;<lpage>3212</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btv351</pub-id> </citation>
</ref>
<ref id="B78">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Singh</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Kim Sung</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Matusop</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Radhakrishnan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shamsul</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Cox-Singh</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2004</year>). <article-title>A Large Focus of Naturally Acquired Plasmodium Knowlesi Infections in Human Beings</article-title>. <source>Lancet</source> <volume>363</volume> (<issue>9414</issue>), <fpage>1017</fpage>&#x2013;<lpage>1024</lpage>. <pub-id pub-id-type="doi">10.1016/S0140-6736(04)15836-4</pub-id> </citation>
</ref>
<ref id="B79">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stanke</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Keller</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Gunduz</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Hayes</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Waack</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Morgenstern</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>AUGUSTUS: Ab Initio Prediction of Alternative Transcripts</article-title>. <source>Nucleic Acids Res.</source> <volume>34</volume> (<issue>Web Server issue</issue>), <fpage>W435</fpage>&#x2013;<lpage>W439</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkl200</pub-id> </citation>
</ref>
<ref id="B80">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Steinbiss</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Silva-Franco</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Brunk</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Foth</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Hertz-Fowler</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Berriman</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Companion: a Web Server for Annotation and Analysis of Parasite Genomes</article-title>. <source>Nucleic Acids Res.</source> <volume>44</volume> (<issue>Web Server issue</issue>), <fpage>W29</fpage>&#x2013;<lpage>W34</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkw292</pub-id> </citation>
</ref>
<ref id="B81">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tessema</surname>
<given-names>S. K.</given-names>
</name>
<name>
<surname>Nakajima</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Jasinskas</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Monk</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Lekieffre</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>E.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Protective Immunity against Severe Malaria in Children Is Associated with a Limited Repertoire of Antibodies to Conserved PfEMP1 Variants</article-title>. <source>Cell Host Microbe</source> <volume>26</volume> (<issue>5</issue>), <fpage>579</fpage>&#x2013;<lpage>590 e575</lpage>. <pub-id pub-id-type="doi">10.1016/j.chom.2019.10.012</pub-id> </citation>
</ref>
<ref id="B82">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thorpe</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Escudero-Martinez</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Cock</surname>
<given-names>P. J. A.</given-names>
</name>
<name>
<surname>Eves-van den Akker</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bos</surname>
<given-names>J. I. B.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Shared Transcriptional Control and Disparate Gain and Loss of Aphid Parasitism Genes</article-title>. <source>Genome Biol. Evol.</source> <volume>10</volume> (<issue>10</issue>), <fpage>2716</fpage>&#x2013;<lpage>2733</lpage>. <pub-id pub-id-type="doi">10.1093/gbe/evy183</pub-id> </citation>
</ref>
<ref id="B83">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thorpe</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Escudero-Martinez</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Eves-van den Akker</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bos</surname>
<given-names>J. I. B.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Transcriptional Changes in the Aphid Species Myzus Cerasi under Different Host and Environmental Conditions</article-title>. <source>Insect Mol. Biol.</source> <volume>29</volume> (<issue>3</issue>), <fpage>271</fpage>&#x2013;<lpage>282</lpage>. <pub-id pub-id-type="doi">10.1111/imb.12631</pub-id> </citation>
</ref>
<ref id="B84">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thorpe</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Vetukuri</surname>
<given-names>R. R.</given-names>
</name>
<name>
<surname>Hedley</surname>
<given-names>P. E.</given-names>
</name>
<name>
<surname>Morris</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Whisson</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Welsh</surname>
<given-names>L. R. J.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Draft Genome Assemblies for Tree Pathogens Phytophthora Pseudosyringae and Phytophthora Boehmeriae</article-title>. <source>G3 (Bethesda)</source> <volume>11</volume> (<issue>11</issue>), <fpage>jkab282</fpage>. <pub-id pub-id-type="doi">10.1093/g3journal/jkab282</pub-id> </citation>
</ref>
<ref id="B85">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thorvaldsd&#xf3;ttir</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Robinson</surname>
<given-names>J. T.</given-names>
</name>
<name>
<surname>Mesirov</surname>
<given-names>J. P.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Integrative Genomics Viewer (IGV): High-Performance Genomics Data Visualization and Exploration</article-title>. <source>Briefings Bioinforma.</source> <volume>14</volume> (<issue>2</issue>), <fpage>178</fpage>&#x2013;<lpage>192</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbs017</pub-id> </citation>
</ref>
<ref id="B86">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vaser</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Sovi&#x107;</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Nagarajan</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>&#x160;iki&#x107;</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Fast and Accurate De Novo Genome Assembly from Long Uncorrected Reads</article-title>. <source>Genome Res.</source> <volume>27</volume> (<issue>5</issue>), <fpage>737</fpage>&#x2013;<lpage>746</lpage>. <pub-id pub-id-type="doi">10.1101/gr.214270.116</pub-id> </citation>
</ref>
<ref id="B87">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wahlgren</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Goel</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Akhouri</surname>
<given-names>R. R.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Variant Surface Antigens of Plasmodium Falciparum and Their Roles in Severe Malaria</article-title>. <source>Nat. Rev. Microbiol.</source> <volume>15</volume> (<issue>8</issue>), <fpage>479</fpage>&#x2013;<lpage>491</lpage>. <pub-id pub-id-type="doi">10.1038/nrmicro.2017.47</pub-id> </citation>
</ref>
<ref id="B88">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Walker</surname>
<given-names>B. J.</given-names>
</name>
<name>
<surname>Abeel</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Shea</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Priest</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Abouelliel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sakthikumar</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Pilon: An Integrated Tool for Comprehensive Microbial Variant Detection and Genome Assembly Improvement</article-title>. <source>PLoS ONE</source> <volume>9</volume> (<issue>11</issue>), <fpage>e112963</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0112963</pub-id> </citation>
</ref>
<ref id="B89">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Debarry</surname>
<given-names>J. D.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>MCScanX: a Toolkit for Detection and Evolutionary Analysis of Gene Synteny and Collinearity</article-title>. <source>Nucleic Acids Res.</source> <volume>40</volume> (<issue>7</issue>), <fpage>e49</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkr1293</pub-id> </citation>
</ref>
<ref id="B90">
<citation citation-type="book">
<collab>World-Health-Organization</collab> (<year>2021</year>). <source>World Malaria Report 2021</source>. <publisher-loc>Geneva</publisher-loc>. </citation>
</ref>
</ref-list>
</back>
</article>