<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article article-type="methods-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">890651</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2022.890651</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>kngMap: Sensitive and Fast Mapping Algorithm for Noisy Long Reads Based on the <italic>K</italic>-Mer Neighborhood Graph</article-title>
<alt-title alt-title-type="left-running-head">Wei et al.</alt-title>
<alt-title alt-title-type="right-running-head">kngMap</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Wei</surname>
<given-names>Ze-Gang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/665461/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Fan</surname>
<given-names>Xing-Guo</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Hao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Xiao-Dan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Fei</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1711946/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Qian</surname>
<given-names>Yu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/72891/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Zhang</surname>
<given-names>Shao-Wu</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1205008/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Institute of Physics and Optoelectronics Technology</institution>, <institution>Baoji University of Arts and Sciences</institution>, <addr-line>Baoji</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Key Laboratory of Information Fusion Technology of Ministry of Education</institution>, <institution>School of Automation</institution>, <institution>Northwestern Polytechnical University</institution>, <addr-line>Xi&#x2019;an</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/778584/overview">Pu-Feng Du</ext-link>, Tianjin University, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/129222/overview">Lin Wan</ext-link>, Academy of Mathematics and Systems Science (CAS), China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/560593/overview">Juan Wang</ext-link>, Inner Mongolia University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/772968/overview">Han Zhang</ext-link>, Nankai University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Yu Qian, <email>qianyu0272@163.com</email>; Shao-Wu Zhang, <email>zhangsw@nwpu.edu.cn</email>
</corresp>
<fn fn-type="other">
<p>This article was submitted to Computational Genomics, a section of the journal Frontiers in Genetics</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>05</day>
<month>05</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>890651</elocation-id>
<history>
<date date-type="received">
<day>06</day>
<month>03</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>07</day>
<month>04</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Wei, Fan, Zhang, Zhang, Liu, Qian and Zhang.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Wei, Fan, Zhang, Zhang, Liu, Qian and Zhang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>With the rapid development of single molecular sequencing (SMS) technologies such as PacBio single-molecule real-time and Oxford Nanopore sequencing, the output read length is continuously increasing, which has dramatical potentials on cutting-edge genomic applications. Mapping these reads to a reference genome is often the most fundamental and computing-intensive step for downstream analysis. However, these long reads contain higher sequencing errors and could more frequently span the breakpoints of structural variants (SVs) than those of shorter reads, leading to many unaligned reads or reads that are partially aligned for most state-of-the-art mappers. As a result, these methods usually focus on producing local mapping results for the query read rather than obtaining the whole end-to-end alignment. We introduce kngMap, a novel <italic>k</italic>-mer neighborhood graph-based mapper that is specifically designed to align long noisy SMS reads to a reference sequence. By benchmarking exhaustive experiments on both simulated and real-life SMS datasets to assess the performance of kngMap with ten other popular SMS mapping tools (e.g., BLASR, BWA-MEM, and minimap2), we demonstrated that kngMap has higher sensitivity that can align more reads and bases to the reference genome; meanwhile, kngMap can produce consecutive alignments for the whole read and span different categories of SVs in the reads. kngMap is implemented in C&#x2b;&#x2b; and supports multi-threading; the source code of kngMap can be downloaded for free at: <ext-link ext-link-type="uri" xlink:href="https://github.com/zhang134/kngMap">https://github.com/zhang134/kngMap</ext-link> for academic usage.</p>
</abstract>
<kwd-group>
<kwd>sequence alignment</kwd>
<kwd>sequence mapping</kwd>
<kwd>single molecular sequencing</kwd>
<kwd>third-generation sequencing</kwd>
<kwd>long noisy reads</kwd>
</kwd-group>
<contract-num rid="cn001">21JK0486</contract-num>
<contract-num rid="cn002">2021JQ-811 2022JZ-03</contract-num>
<contract-num rid="cn003">61873202 61473232 91430111</contract-num>
<contract-sponsor id="cn001">Education Department of Shaanxi Province</contract-sponsor>
<contract-sponsor id="cn002">Natural Science Basic Research Program of Shaanxi Province</contract-sponsor>
<contract-sponsor id="cn003">National Natural Science Foundation of China</contract-sponsor>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Single molecular sequencing (SMS) developed by Pacific Biosciences and Oxford Nanopore Technologies has been increasingly applied in DNA sequencing studies since its emergence (<xref ref-type="bibr" rid="B34">Ono et al., 2012</xref>; <xref ref-type="bibr" rid="B38">Rhoads and Au, 2015</xref>). Compared to next-generation sequencing (NGS), SMS can produce considerably longer reads with less sequencing bias and lower cost (<xref ref-type="bibr" rid="B45">Yang et al., 2017</xref>; <xref ref-type="bibr" rid="B29">Marchet et al., 2019</xref>). Recently, the N50 and maximum lengths of the reads generated by the MinION nanopore sequencer can achieve more than 100&#xa0;kbp and 1&#xa0;Mbp, respectively (<xref ref-type="bibr" rid="B8">Chen et al., 2021</xref>). Such long read lengths offer new solutions to bioinformatic questions that are hard to be resolved by NGS, such as the <italic>de novo</italic> genome assembly, genome resequencing, structural variant (SV) discovery, and transcriptome analysis (<xref ref-type="bibr" rid="B42">St&#xf6;cker et al., 2016</xref>; <xref ref-type="bibr" rid="B44">Wei and Zhang, 2018</xref>; <xref ref-type="bibr" rid="B24">Liu et al., 2019</xref>; <xref ref-type="bibr" rid="B4">Cao et al., 2021</xref>).</p>
<p>Read mapping is to find the possible genomic origins of reads by aligning them against a reference genome. It has become one of the most basic and computing-intensive step in downstream pipelines for SMS dataset analysis (<xref ref-type="bibr" rid="B46">Zhang et al., 2018</xref>). However, since the SMS reads have a higher error rate (&#x223c;15%) than the NGS reads, most of the read mapping methods developed for NGS short read data, such as Bowtie (<xref ref-type="bibr" rid="B15">Langmead et al., 2009</xref>), SOAP3 (<xref ref-type="bibr" rid="B27">Liu et al., 2012</xref>), CloudBurst (<xref ref-type="bibr" rid="B31">Michael, 2009</xref>), and FEM (<xref ref-type="bibr" rid="B46">Zhang et al., 2018</xref>), are not suitable to handle with SMS reads. Thus, a growing number of long noisy read mapping approaches specially designed for SMS long reads have been proposed during the past decade.</p>
<p>Broadly speaking, most existing mapping methods or tools for SMS reads adopt the typical seed-chain-align strategy (<xref ref-type="bibr" rid="B26">Liu et al., 2016a</xref>). In brief, they first find the matched <italic>k</italic>-mers (also called seeds) of query reads to the reference genome; then, the candidate regions (also called alignment skeleton) for aligning in the query read and the reference genome are chosen based on the seeds. Finally, the local subsequences surrounding the seeds are base-to-base aligned to compose the final read alignment with the seeds. Based on the index technique of seed searching, SMS mappers can be categorized into three groups: Burrows&#x2013;Wheeler Transform full-text minute-space (BWT-FM) index-based (<xref ref-type="bibr" rid="B14">Langmead and Salzberg, 2012</xref>), hash table index-based, and short read aligner-based methods.</p>
<p>BWT-FM index-based methods find the matched seeds by constructing the BWT-FM index of the reference genome; such methods include BLASR (<xref ref-type="bibr" rid="B5">Chaisson and Tesler, 2012</xref>), BWA-MEM (<xref ref-type="bibr" rid="B17">Li, 2013</xref>), lordFAST (<xref ref-type="bibr" rid="B10">Haghshenas et al., 2018</xref>), and smsMap (<xref ref-type="bibr" rid="B43">Wei et al., 2020</xref>). BLASR (<xref ref-type="bibr" rid="B5">Chaisson and Tesler, 2012</xref>) initially builds the BWT-FM index of the genome to find short exact matches; then, the candidate aligned region is generated by using sparse dynamic programming (SDP), and the final detailed alignment within the area defined by SDP is performed by dynamic programming. BWA-MEM (<xref ref-type="bibr" rid="B17">Li, 2013</xref>) first searches the supermaximal exact matches according to the BWT-FM index of the reference and detects a group of seeds that are colinear and close to each other by a chaining algorithm, which then extends the seed with a banded affine-gap-penalty dynamic programming. lordFAST (<xref ref-type="bibr" rid="B10">Haghshenas et al., 2018</xref>) selects the candidate alignment regions in the genome using the longest matches identified by the BWT-FM index; then, the base-to-base alignment between consecutive seeds is obtained by performing dynamic programming. Recently, the smsMap (<xref ref-type="bibr" rid="B43">Wei et al., 2020</xref>) mapper proposed by us also utilizes the BWT-FM index to quickly find the matches in the reference genome and then locates the best starting positions in genome and query read by defining a credibility function. Finally, the detailed alignment result is obtained with the column reduction banded alignment method.</p>
<p>Hash-based mapping methods search the matched seeds through a hash table index of the genome. YAHA (<xref ref-type="bibr" rid="B9">Faust and Hall, 2012</xref>) utilizes a hash table index to store the <italic>k</italic>-mer locations in the reference and then extends seeds to generate fragments that contain contiguous matching bases between the query sequence and the reference genome. Next, it combines the fragments to select potential regions for alignment; lastly, it completes the full alignment by applying a modified version of Smith&#x2013;Waterman algorithm to the unmatched regions. rHAT (<xref ref-type="bibr" rid="B25">Liu et al., 2015</xref>) starts to build the regional hash table (RHT) of the genome; then, the matches of the <italic>k</italic>-mers within the query read are retrieved through RHT and a direct acyclic graph is built to compose the skeleton of the alignment. At the end, unaligned pairs of segments in the skeleton are aligned with the banded Smith&#x2013;Waterman algorithm. GraphMap (<xref ref-type="bibr" rid="B12">Ivan et al., 2016</xref>) implements the <italic>q</italic>-gram seeding strategy that allows for fast lookup of inexact matches by constructing a hash index of the reference sequence, which then generates alignment anchors through a fast graph-based ordering of seeds and extends anchors to achieve final alignments. minimap2 (<xref ref-type="bibr" rid="B19">Li, 2018</xref>) indexes minimizers in a hash table that allows for fast lookup of exact matches and then identifies colinear anchor sets; final base-level alignments are obtained by applying dynamic programming to regions between adjacent anchors. conLSH (<xref ref-type="bibr" rid="B6">Chakraborty and Bandyopadhyay, 2020</xref>) computes context-based locality-sensitive hashing values of genomes to facilitate seed search and then generates a series of sites for candidate alignment after extension; finally, it produces the best possible alignment results by applying the sparse dynamic programming-based approach. Later, S-conLSH (<xref ref-type="bibr" rid="B7">Chakraborty et al., 2021</xref>), an improved version of conLSH, was developed by introducing the spaced context-based locality-sensitive hashing for mapping long noisy SMS reads.</p>
<p>Short read aligner-based methods retrieve matches by using extant mappers of short reads. For example, LAMSA (<xref ref-type="bibr" rid="B23">Liu et al., 2016b</xref>) first looks for long approximate matches through short read aligner GEM (<xref ref-type="bibr" rid="B30">Marco-Sola et al., 2012</xref>), finds a set of possible alignment skeletons, and fills the gaps within the skeletons to generate valid alignments for the whole read. NGMLR (<xref ref-type="bibr" rid="B40">Sedlazeck et al., 2018</xref>) identifies similar segments in read and genome by short aligner of NextGenMap (<xref ref-type="bibr" rid="B41">Sedlazeck et al., 2013</xref>); then, it extracts the sub-sequences in read and reference to compute a pairwise sequence alignment using a convex gap cost model. Lastly, it reports the set of linear alignments with the highest joint score as the mapping results.</p>
<p>With the rapid development of SMS sequencing technologies, read length is continuously increasing; these longer reads could more frequently span the breakpoints of SVs than those of shorter reads (<xref ref-type="bibr" rid="B13">Kolmogorov et al., 2019</xref>; <xref ref-type="bibr" rid="B37">Ren et al., 2021</xref>). This may greatly influence read alignment since most state-of-the-art mappers do not consider the SV events, or few methods were designed for handling relatively small variants. Meanwhile, when the matched seeds are diversely distributed in some read parts caused by high sequencing errors, most extant aligners are incapable of obtaining the alignment of the read, leading to many unaligned reads or reads that are partially aligned. As a result, these methods usually focus on producing local mapping results for the query read rather than obtaining the whole end-to-end alignment, leading to a low mapping sensitivity.</p>
<p>To address the aforementioned issues, in this study, we propose a <italic>k</italic>-mer neighborhood graph mapper (named kngMap), a novel long-read mapping algorithm which is specifically designed to improve mapping sensitivity and deal with SV events. Overall, kngMap works in four main stages. It initially constructs a searching index for the reference genome to quickly find matched <italic>k</italic>-mers for query reads. Such matches are then used to construct a <italic>k</italic>-mer <italic>d</italic>-neighborhood graph where matched <italic>k</italic>-mers are viewed as vertices and each pair of matched <italic>k</italic>-mers is connected by a direct edge. Third, a high quality of alignment skeleton is identified by designing a chaining approach. At the end, each unaligned gap in the alignment skeleton is classified into several categories of SV events and filled with a specific alignment method according to its category. The whole read alignment is accomplished by integrating the skeletons and the alignments of the gaps. We benchmarked kngMap on simulated dataset reads with different types of SVs and real-life datasets generated from PacBio SMRT and Oxford Nanopore platforms. The experimental results demonstrated that kngMap has superior ability in terms of base-level sensitivity and end-to-end alignment, which can produce consecutive alignments for the whole read; meanwhile, it can deal with different categories of SVs in the reads.</p>
</sec>
<sec id="s2">
<title>2 Methods</title>
<p>The main motivation of kngMap is to efficiently improve mapping sensitivity and simultaneously have a superior ability of dealing with SVs for long reads generated by SMS sequencing technologies. The underlying design principle is to effectively find one high-quality alignment skeleton in the reference genome for each read, even though the read contains SVs and more sequencing errors, before the costly procedure of base-to-base alignment to the reference genome. An overview of the kngMap algorithm is depicted in <xref ref-type="fig" rid="F1">Figure 1</xref>. kngMap mainly includes four stages: 1) building a searching index of the reference genome in advance, which is used to quickly find matched <italic>k</italic>-mers for a query read <xref ref-type="fig" rid="F1">Figure 1A</xref>; 2) constructing a <italic>k</italic>-mer <italic>d</italic>-neighborhood graph where matched <italic>k</italic>-mers are viewed as vertices and each pair of matched <italic>k</italic>-mers is connected by a direct edge based on the positions in genome and the query read <xref ref-type="fig" rid="F1">Figure 1B</xref>; 3) building and refining the high quality of the alignment skeleton by designing a chaining approach <xref ref-type="fig" rid="F1">Figure 1C</xref>; and 4) classifying the unaligned gaps between pairs of consecutive matched <italic>k</italic>-mers in the alignment skeleton into several categories of SV events and handling each of them with a specific alignment method according to its category <xref ref-type="fig" rid="F1">Figure 1D</xref>. A more detailed illustration of each step is provided later.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Schematic representation of stages in kngMap. <bold>(A)</bold> Building a searching index of the reference genome. <bold>(B)</bold> Constructing a <italic>k</italic>-mer <italic>d</italic>-neighborhood graph where matched <italic>k</italic>-mers (also called anchors) are viewed as vertices and each pair of matched <italic>k</italic>-mers is connected by a direct edge based on the positions in genome and the query read. <bold>(C)</bold> Building and refining the high quality of alignment skeletons by designing a chaining approach. <bold>(D)</bold> Classifying the unaligned gaps between pairs of consecutive anchor matches in the selected chain into several categories of SV events and handling each of them with a specific alignment method according to its category. <bold>(E)</bold> Generating the detailed base-to-base alignment by performing dynamic programming between consecutive anchor matches in the selected chain.</p>
</caption>
<graphic xlink:href="fgene-13-890651-g001.tif"/>
</fig>
<sec id="s2-1">
<title>2.1 Constructing Reference Genome Index</title>
<p>In order to quickly find the occurrence positions of a <italic>k</italic>-mer (subsequence with length of <italic>k</italic>) in the reference genome, a lookup index for the reference genome is usually needed to be built. Currently, two superior index techniques, BWT-FM index (<xref ref-type="bibr" rid="B22">Lippert, 2005</xref>) and hashing table (<xref ref-type="bibr" rid="B32">Ning et al., 2001</xref>), are successfully used by most extant mapping methods (<xref ref-type="bibr" rid="B21">Lindner and Friedel, 2012</xref>). BWT-FM index allows long reference genome to be searched efficiently with low memory usage (<xref ref-type="bibr" rid="B11">Hayashi and Taura, 2013</xref>). Hash table takes linear time to find the positions in the genome when a certain <italic>k</italic>-mer is given (<xref ref-type="bibr" rid="B3">Berlin et al., 2015</xref>). Compared with BWT-FM index, the indexing speed based on the hash table is faster (<xref ref-type="bibr" rid="B18">Li, 2016</xref>; <xref ref-type="bibr" rid="B36">Prezza et al., 2016</xref>), but it will consume more computational memory. Both index techniques have their own advantages (<xref ref-type="bibr" rid="B1">Alser et al., 2021</xref>). Therefore, kngMap utilizes these two strategies [the BWT-FM technique implemented in combined-index (<xref ref-type="bibr" rid="B10">Haghshenas et al., 2018</xref>) and the hash index implemented in minimizers (<xref ref-type="bibr" rid="B18">Li, 2016</xref>)] to construct the index for the reference genome (the default index module is hash).</p>
</sec>
<sec id="s2-2">
<title>2.2 Generating <italic>k</italic>-Mer <italic>d</italic>-Neighborhood Graph</title>
<p>Once the index of the reference genome is constructed, all the matches of the <italic>k</italic>-mers and their matched positions for a query read can be retrieved through the genome index; these matches are subsequently used to form a <italic>k</italic>-mer <italic>d</italic>-neighborhood graph. Specifically, given a set of long reads, kngMap constructs the <italic>d</italic>-neighborhood anchor graph for each read at a time in two steps as follows:</p>
<sec id="s2-2-1">
<title>Step 1: Extracting the Matched k-Mers</title>
<p>KngMap extracts all <italic>k</italic>-mers for one query read and finds the matched positions through the reference genome index built in the aforementioned stage. Each of the matches [also called anchors in some references (<xref ref-type="bibr" rid="B5">Chaisson and Tesler, 2012</xref>; <xref ref-type="bibr" rid="B10">Haghshenas et al., 2018</xref>)] of the <italic>k</italic>-mers can be denoted as a tuple: <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where <italic>x</italic> and <italic>y</italic> are the matched positions on the read and the reference genome, respectively, and <italic>l</italic> is the length of the match (here <italic>l</italic> &#x3d; <italic>k</italic>). In other words, <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> indicates that the substring interval [<italic>x</italic> &#x2212; <italic>k</italic> &#x2b; 1, <italic>x</italic>] on the reference matches the interval [<italic>y</italic> &#x2212; <italic>k</italic> &#x2b; 1, <italic>y</italic>] on the query exactly.</p>
</sec>
<sec id="s2-2-2">
<title>Step 2: Generating the Anchor d-Neighborhood Graph</title>
<p>A <italic>d</italic>-neighborhood anchor graph is formed according to the list of anchors ordered by their positions on the reference genome and then on the read. In this graph, each anchor is viewed as nodes; two nodes (<italic>V</italic>
<sub>
<italic>i</italic>
</sub> &#x2192; <italic>V</italic>
<sub>
<italic>j</italic>
</sub>) are connected with a direct edge if the pair of vertices meets the following condition:<disp-formula id="e1">
<mml:math id="m3">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>d</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>x</mml:mi>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>n</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>g</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>d</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>y</mml:mi>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>n</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>q</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>y</mml:mi>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <inline-formula id="inf3">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the position of node <italic>i</italic> in the reference genome, <inline-formula id="inf4">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the position of node <italic>j</italic> in the query read, and <italic>d</italic> is a constraint parameter which is applied to model the maximum distance between two anchors in the alignment for a read. The edge direct (from <italic>V</italic>
<sub>
<italic>i</italic>
</sub> to <italic>V</italic>
<sub>
<italic>j</italic>
</sub>) denotes that <italic>V</italic>
<sub>
<italic>i</italic>
</sub> is the ancestor (also called precursor or predecessor) node of <italic>V</italic>
<sub>
<italic>j</italic>
</sub>. The setting of parameter <italic>d</italic> is motivated by the fact that, given a certain error rate, the distance between two anchors on the read can be modeled by a geometric distribution (<xref ref-type="bibr" rid="B5">Chaisson and Tesler, 2012</xref>). In kngMap, this property is utilized to describe the maximum allowed distance on the read between two vertices connected by an edge in considering the error rate of the read. Theoretically, two continuous <italic>k</italic>-mers in an error-free read are also matched in the continuous positions in the genome (<xref ref-type="sec" rid="s11">Supplementary Figure S1</xref>). However, in fact, with the sequence errors derived from SMS sequencing, they can be matched with two discontinuous positions in the genome (<xref ref-type="sec" rid="s11">Supplementary Figure S2</xref>), and the distance between these two discontinuous positions is usually lower than <italic>d</italic>. Under this circumstance, <italic>d</italic> can prevent a lot of unnecessary connections; furthermore, it can facilitate building the skeleton in the next step due to the fact that some vertices are lack of precursors. Here, we set <inline-formula id="inf5">
<mml:math id="m6">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, which is a variable value that depends on the length of the query read, where <inline-formula id="inf6">
<mml:math id="m7">
<mml:mi>&#x3bb;</mml:mi>
</mml:math>
</inline-formula> (default value is 1.2) is an empirical value according to the length ratio (defined as the length of the aligned region in reference divided by the length of the aligned region in read) distribution of the aligned results shown in <xref ref-type="sec" rid="s11">Supplementary Figure S3</xref>. With the setting of <italic>d</italic>, it has a high probability that the distance between two neighboring true positive nodes can be successfully connected with an edge.</p>
</sec>
</sec>
<sec id="s2-3">
<title>2.3 Generating the Skeleton of Alignment</title>
<p>Building the skeleton of alignment is a core step for a mapping algorithm since it will dramatically facilitate the base-to-base alignment. The alignment skeleton (<xref ref-type="bibr" rid="B28">Liu et al., 2021</xref>) is a set of nonoverlapping anchors that have the highest possibility to form the candidate aligned region in the read and reference genome. In consideration of the potential breakpoints within reads, we proposed a specific chaining strategy to find the skeleton of alignment in the following two steps:</p>
<sec id="s2-3-1">
<title>Step 1: Generate the Initial Alignment Skeleton</title>
<p>KngMap finds the optimal path connecting from <italic>V</italic>
<sub>
<italic>start</italic>
</sub> to <italic>V</italic>
<sub>
<italic>end</italic>
</sub>, which maximizes the total number of matched bases as the skeleton of alignment. This is implemented by scoring the vertices in the <italic>d</italic>-neighborhood graph with the following recurrence equation:<disp-formula id="e2">
<mml:math id="m8">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:munder>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:munder>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2260;</mml:mo>
<mml:mo>&#x2205;</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <inline-formula id="inf7">
<mml:math id="m9">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the score assigned to the vertice <italic>V</italic>
<sub>
<italic>i</italic>
</sub> in the graph, <inline-formula id="inf8">
<mml:math id="m10">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the precursor set of <italic>V</italic>
<sub>
<italic>i</italic>
</sub>, and <inline-formula id="inf9">
<mml:math id="m11">
<mml:mi>&#x3b1;</mml:mi>
</mml:math>
</inline-formula> (default value is 1) is the reward score between the two vertices linked with an edge. With <xref ref-type="disp-formula" rid="e2">Eq. 2</xref>, each vertex can find a precursor maximizing its score, and the node (<italic>V</italic>
<sub>
<italic>end</italic>
</sub>) with the highest score among all nodes is regarded as the ending node. As a result, the path from <italic>V</italic>
<sub>
<italic>end</italic>
</sub> to a starting node (<italic>V</italic>
<sub>
<italic>start</italic>
</sub> backtracking from <italic>V</italic>
<sub>
<italic>end</italic>
</sub> to a node without precursor) formulates the initial alignment skeleton.</p>
<p>From <xref ref-type="disp-formula" rid="e2">Eq. 2</xref>, we can see that the parameter of reward score (<inline-formula id="inf10">
<mml:math id="m12">
<mml:mi>&#x3b1;</mml:mi>
</mml:math>
</inline-formula>) is set with a constant value; this is critical to deal with some regions containing SV events (e.g., longer insertion and deletion gaps) or high sequencing errors, especially when the lengths of these regions are long. By setting to a constant value, two anchors can successfully span the region with SV or high sequencing errors, as shown in <xref ref-type="sec" rid="s11">Supplementary Figure S4</xref>; otherwise, two chains will be generated if a low score is assigned to an anchor when the distance between this anchor and its ancestor is too large (<xref ref-type="bibr" rid="B19">Li, 2018</xref>). Therefore, with a constant scoring parameter <inline-formula id="inf11">
<mml:math id="m13">
<mml:mi>&#x3b1;</mml:mi>
</mml:math>
</inline-formula>, it can make sure that the edge connecting the two matches flanking the SV or regions with high errors can be detected. Thus, the scoring scheme in <xref ref-type="disp-formula" rid="e2">Eq. 2</xref> helps recover longer insertions and deletions.</p>
</sec>
<sec id="s2-3-2">
<title>Step 2: Refining the Alignment Skeleton</title>
<p>Next, kngMap refines the skeleton by pruning the initial alignment skeleton. This pruning process aims to avoid some false positions caused by sequencing errors or repeat regions. As shown in <xref ref-type="fig" rid="F1">Figure 1C</xref>, the skeleton is generated by the previous step. Obviously, the two ending anchors in <xref ref-type="fig" rid="F1">Figure 1C</xref> should not be considered for downstream detail alignment. Thus, an increased refining procedure is carefully designed to deal with such situations. In the skeleton, kngMap only chooses the chains (a subset of the anchors in the initial skeleton) with the highest increased score using the following equation:<disp-formula id="e3">
<mml:math id="m14">
<mml:mrow>
<mml:mi>arg</mml:mi>
<mml:mtext>&#x2002;</mml:mtext>
<mml:munder>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2002;</mml:mtext>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2002;</mml:mtext>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3c;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>where <italic>l</italic> is the constraint length parameter (default value <italic>l</italic> &#x3d; <italic>len</italic>(<italic>r</italic>)), which guarantees that the increased score is calculated in a fixed window length in the path. This setting is motivated by observing that the anchors of a read alignment are fallen in the region with a length of <italic>l</italic>, which is found by the statistics based on the alignment results (<xref ref-type="sec" rid="s11">Supplementary Figure S5</xref>). After the pruning, the two nodes that have the maximum increased scores are selected as the final starting and ending nodes (denoted as <inline-formula id="inf12">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf13">
<mml:math id="m16">
<mml:mrow>
<mml:msub>
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>), and the path between <inline-formula id="inf14">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf15">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is regarded as the final refined skeleton of alignment.</p>
</sec>
</sec>
<sec id="s2-4">
<title>2.4 Filling the Gaps Between Anchors</title>
<p>With the refined skeleton of alignment, the numbers of pairs of unaligned segments can be directly partitioned by the anchors, which is shown in <xref ref-type="sec" rid="s11">Supplementary Figure S6</xref>. In order to obtain the whole base-to-base alignment of the read, kngMap distinguishingly performs a detailed alignment between each pair of unaligned segments. As illustrated in <xref ref-type="sec" rid="s11">Supplementary Figure S6</xref>, the skeleton (from <italic>M</italic>
<sub>
<italic>1</italic>
</sub> to <italic>M</italic>
<sub>
<italic>7</italic>
</sub>) partitions the read and the reference into eight paired segments. Each pair of unaligned segments, that is, (<italic>S</italic>
<sub>
<italic>Ri</italic>
</sub>, <italic>S</italic>
<sub>
<italic>Gi</italic>
</sub>), <italic>i</italic> &#x3d; 1, 2, 3, 4, 5, 6, 7, will be categorized into one of the following conditions, which carefully take the SV into consideration according to the length of segments.</p>
<p>1) Match case: <inline-formula id="inf16">
<mml:math id="m19">
<mml:mrow>
<mml:mrow>
<mml:mo>&#x7c;</mml:mo>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x7c;</mml:mo>
</mml:mrow>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>2) Deletion case: <inline-formula id="inf17">
<mml:math id="m20">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3c;</mml:mo>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>3) Insertion case: <inline-formula id="inf18">
<mml:math id="m21">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3e;</mml:mo>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>Here, <italic>S</italic>
<sub>
<italic>Ri</italic>
</sub> and <italic>S</italic>
<sub>
<italic>Gi</italic>
</sub> are the paired partitioned subsequence in the read and genome, respectively, <inline-formula id="inf19">
<mml:math id="m22">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the length of <italic>S</italic>
<sub>
<italic>Ri</italic>
</sub>, <inline-formula id="inf20">
<mml:math id="m23">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the length of <italic>S</italic>
<sub>
<italic>Gi</italic>
</sub>, and <inline-formula id="inf21">
<mml:math id="m24">
<mml:mi>&#x3bc;</mml:mi>
</mml:math>
</inline-formula> is a user-defined parameter categorizing the cases. The parameter of <inline-formula id="inf22">
<mml:math id="m25">
<mml:mi>&#x3bc;</mml:mi>
</mml:math>
</inline-formula> is used to model specific categories of SVs and the default value of <inline-formula id="inf23">
<mml:math id="m26">
<mml:mi>&#x3bc;</mml:mi>
</mml:math>
</inline-formula> is set with 2, which is an empirical value based on the alignment statistics (<xref ref-type="sec" rid="s11">Supplementary Figure S7</xref>). A schematic illustration of the three categories of unaligned segments is shown in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Schematic illustration of the three categories of unaligned segments in the read and genome. <bold>(A)</bold> Match, <bold>(B)</bold> deletion, and <bold>(C)</bold> insertion. In the figure, the light blue, yellow, and green bars present the read, the reference genome, and the anchored <italic>k</italic>-mers, respectively. The dashed lines connecting the anchored <italic>k</italic>-mers indicate the matched positions of the <italic>k</italic>-mers on the reference genome and the read. The classification conditions are based on the lengths of pair of unaligned segments. The second and third rows give an example on how to generate the alignment for each case.</p>
</caption>
<graphic xlink:href="fgene-13-890651-g002.tif"/>
</fig>
<p>KngMap then fills the paired segments within the skeletons to solve the breakpoints of SVs and generate valid alignments for the whole read. Each of the gaps is filled by one of the following strategies according to its category. For the match case, as shown in <xref ref-type="fig" rid="F2">Figure 2A</xref>, kngMap directly performs an end-to-end alignment based on the NW (Needleman&#x2013;Wunsch) algorithm between the read segment (<italic>S</italic>
<sub>
<italic>Rj</italic>
</sub>) and the reference genome segment (<italic>S</italic>
<sub>
<italic>Gj</italic>
</sub>) to fill the gap; an example of dealing with this match case is also provided in <xref ref-type="fig" rid="F2">Figure 2A</xref>. For a deletion case (<xref ref-type="fig" rid="F2">Figure 2B</xref>), there is a potential deletion in <italic>S</italic>
<sub>
<italic>Rj</italic>
</sub>; moreover, we do not know if the SV position exists. In order to effectively deal with such cases, a semi-global alignment is performed, which is beneficial for handling multiple breakpoints within <italic>S</italic>
<sub>
<italic>Gj</italic>
</sub>. For the example in <xref ref-type="fig" rid="F2">Figure 2B</xref>, there are two deletions within the <italic>S</italic>
<sub>
<italic>Rj</italic>
</sub>; the semi-global alignment can effectively recognize them since it can find the most similar regions in <italic>S</italic>
<sub>
<italic>Gj</italic>
</sub> for <italic>S</italic>
<sub>
<italic>Rj</italic>
</sub>. In the implementation of the semi-global alignment, the genome segment (<italic>S</italic>
<sub>
<italic>Gj</italic>
</sub>) was chosen to be the target sequence and read segment (<italic>S</italic>
<sub>
<italic>Rj</italic>
</sub>) to be the query sequence as the alignment gaps at the query start and end are not penalized. Similarly, for an insertion case (<xref ref-type="fig" rid="F2">Figure 2C</xref>), there is a potential insertion in <italic>S</italic>
<sub>
<italic>Rj</italic>
</sub>, and we do not know if its positions exist. For this case, the semi-global alignment is also applied to recognize the insertion. The read segment (<italic>S</italic>
<sub>
<italic>Rj</italic>
</sub>) was chosen to be the target sequence and genome segment (<italic>S</italic>
<sub>
<italic>Gj</italic>
</sub>) to be the query sequence in the implementation of the semi-global alignment.</p>
</sec>
<sec id="s2-5">
<title>2.5 Extending the Boundaries of the Skeletons</title>
<p>All the operations mentioned previously fill the inner gaps of the skeleton, which are anchored by two matched <italic>k</italic>-mers. For the outer boundaries of the refined skeleton, that is, the two pairs of unsigned segments at the starts and ends of the skeletons as shown in <xref ref-type="sec" rid="s11">Supplementary Figure S8</xref>, kngMap assumes that these boundaries of the genomic region is SV-free and directly extends the boundaries by performing a modified global alignment to obtain the base-to-base alignment. To be more specific, for the alignment between the suffix of the read and the reference following the last anchor, kngMap first extracts the <italic>S</italic>
<sub>
<italic>Rend</italic>
</sub> as the query sequence and the <italic>S</italic>
<sub>
<italic>Gend</italic>
</sub> (with two times length than <italic>S</italic>
<sub>
<italic>Rend</italic>
</sub>) as the target sequence. Then, the modified global alignment (similar to the global NW alignment method but with a small twist gap at the query end that is not penalized) is performed for <italic>S</italic>
<sub>
<italic>Rend</italic>
</sub> and <italic>S</italic>
<sub>
<italic>Gend</italic>
</sub>; the modified global alignment can find out how well <italic>S</italic>
<sub>
<italic>Rend</italic>
</sub> fits at the beginning of <italic>S</italic>
<sub>
<italic>Gend</italic>
</sub>. A toy example of it is illustrated in <xref ref-type="sec" rid="s11">Supplementary Figure S8</xref>. Similarly, the alignment between the prefix of the read and the reference prior to the first anchor can be computed in an identical fashion. Finally, the whole read alignment is accomplished by integrating the skeletons and the alignments of the gaps <xref ref-type="fig" rid="F1">Figure 1E</xref>. Overall, the pseudo-code for the kngMap method procedure is described in <xref ref-type="statement" rid="Algorithm_1">
<bold>Algorithm 1</bold>
</xref>:</p>
<p>
<statement content-type="algorithm" id="Algorithm_1">
<label>Algorithm 1</label>
<p>kngMap Algorithm.</p>
<p>
<inline-graphic xlink:href="fgene-13-890651-fx1.tif"/>
</p>
</statement>
</p>
</sec>
</sec>
<sec id="s3">
<title>3 Results</title>
<p>The performance of kngMap was evaluated against 10 state-of-the-art long read aligners: BLASR (v) (<xref ref-type="bibr" rid="B5">Chaisson and Tesler, 2012</xref>), minimap2 (v2.12-r829) (<xref ref-type="bibr" rid="B19">Li, 2018</xref>), BWA-MEM (v0.7.17-r1194) (<xref ref-type="bibr" rid="B17">Li, 2013</xref>), GraphMap (v0.5.2) (<xref ref-type="bibr" rid="B12">Ivan et al., 2016</xref>), rHAT (v0.1.2) (<xref ref-type="bibr" rid="B25">Liu et al., 2015</xref>), NGMLR (v0.2.7) (<xref ref-type="bibr" rid="B40">Sedlazeck et al., 2018</xref>), lordFAST (v0.0.9) (<xref ref-type="bibr" rid="B10">Haghshenas et al., 2018</xref>), smsMap (<xref ref-type="bibr" rid="B43">Wei et al., 2020</xref>), conLSH (v0.0.1) (<xref ref-type="bibr" rid="B6">Chakraborty and Bandyopadhyay, 2020</xref>), and S-conLSH (v0.0.1) (<xref ref-type="bibr" rid="B7">Chakraborty et al., 2021</xref>). All methods were benchmarked on simulated and real-life SMS sequencing datasets. All the experiments were performed on a Linux machine server running Ubuntu 14.04 system equipped with two twelve-core (two threads per core) Intel Xeon Gold 5118 CPU @ 2.30GHz and 256 gigabytes (GB) RAM. The run command lines and parameters of each mapping tools are available in <xref ref-type="sec" rid="s11">Supplementary Table S1</xref>.</p>
<p>It is to be noted that with the rapid development of SMS sequencing technology, more than 99% of the raw sequencing data produced by the SMS sequencing platform has a read length of 1,000&#xa0;bp or longer (<xref ref-type="bibr" rid="B10">Haghshenas et al., 2018</xref>). Herein, reads with lengths shorter than 1,000&#xa0;bp were filtered so that the remaining reads that have 1,000&#xa0;bp or longer are aligned in the following experiments.</p>
<sec id="s3-1">
<title>3.1 Experiment on Simulated Datasets</title>
<sec id="s3-1-1">
<title>3.1.1 Simulation Without Structural Variations</title>
<p>We first adopted the simulated sequence datasets without structural variations (SVs) to evaluate the performance of kngMap algorithm against the aforementioned mapping methods. The simulated datasets were generated by PBSIM2 (<xref ref-type="bibr" rid="B33">Ono et al., 2020</xref>), a new simulator that can capture the characteristics of errors in reads for both PacBio and Nanopore sequencer. The <italic>H. sapiens</italic> (CHM1) genome sequences were downloaded from NCBI (assembly accession: GCA_000306695.2) and fed into PBSIM2 software using default error parameters (8% deletions, 7% insertions, and 1% substitutions). As a result, 100,000 simulated sequences with an average read length of 10,157&#xa0;bp were generated by PBSIM2. The detailed running commands of PBSIM2 and some statistics of simulated datasets can be found in <xref ref-type="sec" rid="s11">Supplementary Tables S2 and S3</xref>.</p>
<p>For the simulated dataset, we know exactly the true mapped region and bases in the reference genome for each read, so the correctly mapped reads (CMRs) and correctly mapped bases (CMBs) were applied to investigate the overall quality of the alignments. A read <italic>r</italic> is called CMR if this read is aligned to the derived genome with the correct strand, and the overlap between the aligned subsequence on the reference and the true mapping subsequence has at least <italic>p</italic> bases (<italic>p</italic> &#x3d; 0.9 &#xd7; <italic>len</italic>(<italic>r</italic>), where <italic>len</italic>(<italic>r</italic>) is the read length of <italic>r</italic>). A base is called CMB if it is located within <italic>T</italic> bp (here <italic>T</italic> &#x3d; 5) of the corresponding truth position on the genome. We further compared the aligned coverage (proportion of aligned bases for one read) to assess the alignment for different methods. A method with higher aligned coverage means that it can align more bases to the reference genome for a read. In addition, base-level sensitivity and precision (<xref ref-type="bibr" rid="B30">Marco-Sola et al., 2012</xref>) are used to evaluate the performance of different mappers for the simulated sequence dataset. Sensitivity is defined as the number of correct matched bases divided by the total number of bases, and precision is defined as the number of correct matched bases divided by the number of mapped bases.</p>
<p>Based on the aforementioned metric definition, <xref ref-type="table" rid="T1">Table 1</xref> shows the evaluation result of kngMap, rHAT, smsMap, lordFAST, BLASR, BWA-MEM, GraphMap, minimap2, NGMLR, conLSH, and S-conLSH on the simulated datasets without structural variations. As can be seen from <xref ref-type="table" rid="T1">Table 1</xref>, kngMap not only correctly mapped more reads than any other mapper but also correctly aligned 99.82% of the total number of bases, which improves the sensitivity by 0.1%&#x2013;2.4% over its competitors, except the conLSH and S-conLSH methods. For the base sensitivity and precision in <xref ref-type="table" rid="T1">Table 1</xref>, we can observe that kngMap still performed the best, indicating that most bases are correctly mapped in the alignments generated by kngMap. It is worth noting that the alignment results of conLSH and S-conLSH are obviously worse than those of other methods, so we do not show the results of these two methods in the following comparisons.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Mapping results of different mapping tools on the simulated human dataset. This dataset contains 10,000 reads and 1.015 billion bases. The best results are labeled with a bold typeface.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Method</th>
<th align="center">CMR</th>
<th align="center">CMB</th>
<th align="center">Aligned coverage (%)</th>
<th align="center">Sensitivity (%)</th>
<th align="center">Precision (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">kngMap</td>
<td align="center">
<bold>99,711</bold>
</td>
<td align="center">
<bold>1,013,921,165</bold>
</td>
<td align="char" char=".">
<bold>100</bold>
</td>
<td align="char" char=".">
<bold>99.82</bold>
</td>
<td align="char" char=".">99.82</td>
</tr>
<tr>
<td align="left">rHAT</td>
<td align="center">99,516</td>
<td align="center">1,008,128,007</td>
<td align="char" char=".">99.95</td>
<td align="char" char=".">99.25</td>
<td align="char" char=".">99.59</td>
</tr>
<tr>
<td align="left">smsMap</td>
<td align="center">98,747</td>
<td align="center">993,435,787</td>
<td align="char" char=".">99.98</td>
<td align="char" char=".">97.80</td>
<td align="char" char=".">97.80</td>
</tr>
<tr>
<td align="left">lordFAST</td>
<td align="center">99,709</td>
<td align="center">1,012,991,874</td>
<td align="char" char=".">99.99</td>
<td align="char" char=".">99.73</td>
<td align="char" char=".">99.85</td>
</tr>
<tr>
<td align="left">BLASR</td>
<td align="center">99,648</td>
<td align="center">1,010,806,725</td>
<td align="char" char=".">99.72</td>
<td align="char" char=".">99.51</td>
<td align="char" char=".">99.77</td>
</tr>
<tr>
<td align="left">BWA-MEM</td>
<td align="center">99,603</td>
<td align="center">1,010,085,677</td>
<td align="char" char=".">99.88</td>
<td align="char" char=".">99.44</td>
<td align="char" char=".">99.63</td>
</tr>
<tr>
<td align="left">GraphMap</td>
<td align="center">98,594</td>
<td align="center">1,001,398,553</td>
<td align="char" char=".">99.99</td>
<td align="char" char=".">98.85</td>
<td align="char" char=".">98.72</td>
</tr>
<tr>
<td align="left">minimap2</td>
<td align="center">99,698</td>
<td align="center">1,013,427,775</td>
<td align="char" char=".">99.89</td>
<td align="char" char=".">99.77</td>
<td align="char" char=".">99.90</td>
</tr>
<tr>
<td align="left">NGMLR</td>
<td align="center">97,207</td>
<td align="center">990,054,543</td>
<td align="char" char=".">99.83</td>
<td align="char" char=".">97.47</td>
<td align="char" char=".">98.67</td>
</tr>
<tr>
<td align="left">conLSH</td>
<td align="center">105</td>
<td align="center">1,207,640</td>
<td align="char" char=".">99.74</td>
<td align="char" char=".">0.11</td>
<td align="char" char=".">0.12</td>
</tr>
<tr>
<td align="left">S-conLSH</td>
<td align="center">2,349</td>
<td align="center">115,168</td>
<td align="char" char=".">99.98</td>
<td align="char" char=".">0.01</td>
<td align="char" char=".">0.01</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Importantly, kngMap achieved 100% aligned coverage, demonstrating that all mapped reads by kngMap are completely aligned, while the alignment by other aligners cannot cover the whole sequence, which means that these methods always output local mapping results and cannot obtain the end-to-end alignment for the whole read. After checking the details of the alignments, soft clipping at the starting or ending of sequences is usually produced by other methods. These results in aligned coverage explain that kngMap has a higher base-level sensitivity. This 100 percent coverage achieved by kngMap is beneficial for SMRT data analysis since higher aligned coverage is a key requirement for mapping tools and downstream mapping-based applications (<xref ref-type="bibr" rid="B39">Schmieder and Edwards, 2012</xref>; <xref ref-type="bibr" rid="B16">Laver et al., 2015</xref>).</p>
<p>In addition, we inspected the reads of the simulated human dataset which were incorrectly aligned by kngMap and found that 255 (88.29%) of the reads have a higher similarity than the original similarity simulated by the PBSIM2. To give an example, <xref ref-type="sec" rid="s11">Supplementary Material</xref> (named as <italic>S9_46798.txt</italic>) reports the alignment result for one simulated read with a length of 6,692&#xa0;bp; the mapped alignment identity obtained by kngMap is 90.74%, while the simulated identity generated by PBSIM2 is 87.50%, which is lower than the similarity produced by kngMap. Therefore, this further indicates that kngMap can find the most similar mapped region in the reference sequence for each query read.</p>
</sec>
<sec id="s3-1-2">
<title>3.1.2 Simulation in the Presence of Structural Variations</title>
<p>For assessing the ability of handling SVs for different methods, a simulated sequence dataset with different types of SVs was generated. Specifically, 16 SVs (i.e., five insertions, eight deletions, and three inversions) with different sizes were first added into the reference chr1 (chromosome 1 of CHM1) genome by the RSVSim (<xref ref-type="bibr" rid="B2">Bartenhagen and Dugas, 2013</xref>) simulator, an R package tool for the simulation of SVs with various sizes and SV types in any genome available as the FASTA file. Then, the chr1 genome of CHM1 with SVs was fed into PBSIM2 to produce the final simulation read set. Among the simulated reads, a total of 129 reads cover the SV breakpoints. The detailed SVs and its breakpoints are listed in <xref ref-type="sec" rid="s11">Supplementary Table S4</xref>.</p>
<p>Here, we provide the number of mapped reads that span SVs (&#x23;SVs) to evaluate the performance of different mapping tools. For one mapped read, if the start and end alignment coordinates in the genome cover the actual simulated breakpoints, we consider this read spanning SVs. Specifically, each of the breakpoints described by the ground truth in the genome can be denoted as a tuple: <inline-formula id="inf24">
<mml:math id="m27">
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>T</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:msubsup>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:msubsup>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msubsup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mo>&#x22ef;</mml:mo>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2002;</mml:mtext>
<mml:msubsup>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf25">
<mml:math id="m28">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:msubsup>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf26">
<mml:math id="m29">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:msubsup>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> are the starting and ending positions of the breakpoint on the reference genome, respectively, and <inline-formula id="inf27">
<mml:math id="m30">
<mml:mrow>
<mml:msubsup>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the total number of ground truth breakpoints. Similarly, each of the alignment coordinates mapped by a mapping tool for a simulated read with SV can be denoted as a tuple: <inline-formula id="inf28">
<mml:math id="m31">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>M</mml:mi>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mo>&#x22ef;</mml:mo>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2002;</mml:mtext>
<mml:msubsup>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf29">
<mml:math id="m32">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf30">
<mml:math id="m33">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the starting and ending mapped coordinates on the reference genome, respectively, and <inline-formula id="inf31">
<mml:math id="m34">
<mml:mrow>
<mml:msubsup>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the total number of reads covering the SV breakpoints (<inline-formula id="inf32">
<mml:math id="m35">
<mml:mrow>
<mml:msubsup>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 129).</p>
<p>With the ground truth breakpoints and the mapped coordinates obtained by a mapping tool, we can assess the number of ground truth breakpoints being recovered. For a certain read, a ground truth breakpoint, <inline-formula id="inf33">
<mml:math id="m36">
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>T</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:msubsup>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:msubsup>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msubsup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, is considered being recovered, only if there is at least one mapped coordinate, <inline-formula id="inf34">
<mml:math id="m37">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>M</mml:mi>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, meeting the following condition:<disp-formula id="e4">
<mml:math id="m38">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:mi>P</mml:mi>
<mml:msubsup>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3e;</mml:mo>
<mml:mi>P</mml:mi>
<mml:msubsup>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
</p>
<p>Finally, the number of ground truth breakpoints being recovered from our kngMap and other seven mapping programs is listed in <xref ref-type="table" rid="T2">Table 2</xref>, from which we can see that kngMap can map more reads with SVs on the genome than the other approaches, which demonstrate that our kngMap is capable to handle the SV-spanning reads.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Number of mapped reads that span SV breakpoints for different mapping methods.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="center">kngMap</th>
<th align="center">smsMap</th>
<th align="center">lordFAST</th>
<th align="center">BLASR</th>
<th align="center">BWA-MEM</th>
<th align="center">GraphMap</th>
<th align="center">minimap2</th>
<th align="center">NGMLR</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">&#x23;SVs</td>
<td align="char" char=".">122</td>
<td align="char" char=".">97</td>
<td align="char" char=".">119</td>
<td align="char" char=".">114</td>
<td align="char" char=".">87</td>
<td align="char" char=".">119</td>
<td align="char" char=".">115</td>
<td align="char" char=".">103</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Note: rHAT tool always appears as segmentation fault (core dumped) information for the simulated reads with SVs, so we did not show the results of rHAT.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec id="s3-2">
<title>3.2 Experiment on Three Real Datasets</title>
<p>In this experiment, three datasets (generating from PacBio SMRT and Oxford Nanopore platforms) of <italic>A. thaliana</italic>, <italic>E. coli UTI89</italic>, and <italic>H. sapiens</italic> (CHM1) were used to benchmark kngMap against other methods on real sequencing data. Among these datasets, <italic>A. thaliana</italic> and <italic>H. sapiens</italic> (CHM1) were generated from the PacBio SMRT platform, and <italic>E. coli UTI89</italic> was generated from the Oxford Nanopore MinION sequencer. The availability of these datasets and their reference genomes are provided in <xref ref-type="sec" rid="s11">Supplementary Tables S5 and S6</xref>, respectively, and the detail statistics (e.g., read number and length distributions) related to these datasets are given in <xref ref-type="sec" rid="s11">Supplementary Figure S7</xref>. In the absence of true mapping locations for these real-life datasets, the mappers were compared based on different metrics as follows.</p>
<p>First, the number of mapped reads, the number of mapped bases, the number of matched bases, and the alignment score are used to evaluate the performance of different mapping methods. The number of mapped reads (read-level) and bases (base-level) are two important metrics to evaluate mapping sensitivity for long read alignment since it could still be lack of information for downstream analysis if reads are unmapped or only partially aligned (<xref ref-type="bibr" rid="B35">Peng et al., 2015</xref>). The number of mapped bases is usually viewed as a metric in base-level sensitivity. The number of matched bases and the alignment score can reflect the quality of the reported alignments for each method (<xref ref-type="bibr" rid="B10">Haghshenas et al., 2018</xref>). Specifically, for each mapping of a read, the matched base is defined as the base which is mapped to the identical one in the reference genome, the alignment score is calculated with scoring parameters: match &#x3d; &#x2b;1, mismatch (including insertion, deletion, substitution, and unmapped/clipped cases) &#x3d; &#x2212;1, gap opening &#x3d; &#x2212;1, and gap extension &#x3d; &#x2212;1, and the sum of alignment scores of all mapped reads was calculated for each method. <xref ref-type="table" rid="T3">Table 3</xref> reports the mapping results of nine methods for the <italic>A. thaliana</italic> dataset. Obviously, it can be seen that kngMap mapped the most reads and aligned the most bases, that is, kngMap mapped 21,182 reads and 185,924,663 bases, improving sensitivity by 3.5% and 4.3% over the closest competitor (lordFAST) in terms of mapped reads and mapped bases, respectively, and even more compared with other tools. For the matched bases and alignment score, kngMap still performed the best; more precisely, kngMap reports a 6.02 million higher number of matched bases and 4.32 million higher alignment score compared to the best runner-up (i.e., lordFAST). These results indicate that kngMap can not only provide more sensitive alignments in read-level and base-level sensitivity than other mappers but also achieved the best quality of the alignments. Similar results (<xref ref-type="sec" rid="s11">Supplementary Tables S8 and S9</xref>) were replicated in two other real datasets (<italic>E. coli UTI89</italic> and <italic>H. sapiens</italic>) as well.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Number of mapped reads, mapped bases, mapped matched bases, and the alignment score of nine methods on the real <italic>A. thaliana</italic> dataset. The best results are labeled with a bold typeface.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Method</th>
<th align="center">Mapped reads</th>
<th align="center">Mapped bases</th>
<th align="center">Matched bases</th>
<th align="center">Alignment score</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">kngMap</td>
<td align="center">
<bold>21,182</bold>
</td>
<td align="center">
<bold>185,924,663</bold>
</td>
<td align="center">
<bold>170,116,753</bold>
</td>
<td align="center">
<bold>144,880,696</bold>
</td>
</tr>
<tr>
<td align="left">rHAT</td>
<td align="center">21,139</td>
<td align="center">161,930,990</td>
<td align="center">149,098,181</td>
<td align="center">104,998,955</td>
</tr>
<tr>
<td align="left">smsMap</td>
<td align="center">21,170</td>
<td align="center">185,937,988</td>
<td align="center">162,348,708</td>
<td align="center">131,169,015</td>
</tr>
<tr>
<td align="left">lordFAST</td>
<td align="center">21,156</td>
<td align="center">179,616,861</td>
<td align="center">165,787,986</td>
<td align="center">138,853,327</td>
</tr>
<tr>
<td align="left">BLASR</td>
<td align="center">20,951</td>
<td align="center">170,669,368</td>
<td align="center">156,421,685</td>
<td align="center">122,279,677</td>
</tr>
<tr>
<td align="left">BEA-MEM</td>
<td align="center">20,987</td>
<td align="center">168,716,088</td>
<td align="center">157,879,323</td>
<td align="center">125,312,363</td>
</tr>
<tr>
<td align="left">GraphMap</td>
<td align="center">20,088</td>
<td align="center">175,859,786</td>
<td align="center">161,472,579</td>
<td align="center">139,464,799</td>
</tr>
<tr>
<td align="left">minimap2</td>
<td align="center">20,748</td>
<td align="center">169,541,803</td>
<td align="center">158,229,925</td>
<td align="center">127,139,769</td>
</tr>
<tr>
<td align="left">NGMLR</td>
<td align="center">18,918</td>
<td align="center">155,650,064</td>
<td align="center">144,968,837</td>
<td align="center">117,202,862</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Next, we investigated the consecutive alignments, which can reflect the details of the alignments (<xref ref-type="bibr" rid="B25">Liu et al., 2015</xref>). To compare the consecutiveness of the alignments, we set four covering thresholds (i.e., <inline-formula id="inf35">
<mml:math id="m39">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>80</mml:mn>
<mml:mo>%</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf36">
<mml:math id="m40">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>85</mml:mn>
<mml:mo>%</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf37">
<mml:math id="m41">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>90</mml:mn>
<mml:mo>%</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf38">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>4</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>95</mml:mn>
<mml:mo>%</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>) to inspect the aligned proportion of the read which has at least one alignment covering at least <italic>c</italic>
<sub>
<italic>i</italic>
</sub> (<italic>i</italic> &#x3d; 1,2,3,4) proportion of the whole read. If one alignment covers more than <italic>c</italic>
<sub>
<italic>i</italic>
</sub> of the read, the alignment is defined as the consecutive alignment at threshold <italic>c</italic>
<sub>
<italic>i</italic>
</sub>. <xref ref-type="fig" rid="F3">Figure 3</xref> describes the results of the consecutive alignment for nine methods, and the detailed values shown in <xref ref-type="fig" rid="F3">Figure 3</xref> can be seen in <xref ref-type="sec" rid="s11">Supplementary Table S10</xref>. From <xref ref-type="fig" rid="F3">Figure 3</xref>, we can clearly see that kngMap and smsMap always achieved 100% consecutive alignments at all thresholds, indicating that kngMap and smsMap can achieve the end-to-end alignment for each read, which will facilitate downstream analysis in practice since each mapped read is completely aligned and not split aligned. GraphMap also achieved high consecutive alignment, while other methods, especially BWA-MEM and NGMLR, tend to produce more reads that have a large proportion of bases being clipped. Similar results can be found in <xref ref-type="sec" rid="s11">Supplementary Figures S9 and S10</xref> for the real dataset of <italic>E. coli</italic> and <italic>H. sapiens</italic>. The consecutive results in <xref ref-type="fig" rid="F3">Figure 3</xref> and <xref ref-type="sec" rid="s11">Supplementary Figures S9 and S10</xref> demonstrate that kngMap can provide better consecutive alignments, which is the main reason for the higher base-level sensitivity of kngMap.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Consecutiveness alignment of different methods on the <italic>A. thaliana</italic> dataset. The percentage values labeled on the horizontal axis are the thresholds of the proportion of covered bases, that is, <italic>c</italic>
<sub>
<italic>i</italic>
</sub> (<italic>i</italic> &#x3d; 1,2,3,4), for considering if the alignment result for a read is consecutive. The vertical axis bars indicate the proportions of reads consecutively mapped by different mappers. kngMap always achieved 100% consecutive alignments with all thresholds, while GraphMap achieved 99.91%, 99.88%, 99.84%, and 99.77% consecutive alignment at 80%, 85%, 90%, and 95% coverage thresholds, respectively. Therefore, kngMap is superior to GraphMap in providing better consecutive alignments. Each bar also corresponds to a value line in <xref ref-type="sec" rid="s11">Supplementary Table S10</xref>.</p>
</caption>
<graphic xlink:href="fgene-13-890651-g003.tif"/>
</fig>
<p>Furthermore, the agreement between different methods based on their alignment results was measured. Supposing that alignments <italic>x</italic>
<sub>
<italic>1</italic>
</sub> and <italic>x</italic>
<sub>
<italic>2</italic>
</sub> are two alignment results obtained by two mapping methods for a query read, we define that <italic>x</italic>
<sub>
<italic>1</italic>
</sub> has an agreement with <italic>x</italic>
<sub>
<italic>2</italic>
</sub> if and only if the mapped region on the reference genome covered by <italic>x</italic>
<sub>
<italic>1</italic>
</sub> overlaps with at least 90% of the mapped region on the reference genome covered by <italic>x</italic>
<sub>
<italic>2</italic>
</sub>. <xref ref-type="fig" rid="F4">Figure 4</xref> describes some toy examples to illustrate the covering and non-covering alignments. As a result, the agreement alignments between each pair methods on the <italic>A. thaliana</italic> dataset are shown in <xref ref-type="table" rid="T4">Table 4</xref>. Each row in <xref ref-type="table" rid="T4">Table 4</xref> denotes the percentage of the alignments produced by the corresponding method that covers alignments generated by other tools listed in the column. For example, among all mapped reads for kngMap and minimap2 in <xref ref-type="table" rid="T4">Table 4</xref>, kngMap covers 91.85% of the alignments produced by the minimap2 method, while minimap2 only covers 78.82% of the alignments generated by kngMap. <xref ref-type="sec" rid="s11">Supplementary Tables S12 and S13</xref> report the agreement between different methods on <italic>E. coli</italic> and <italic>H. sapiens</italic> datasets, respectively. We can see that the mapping results of kngMap have a high coverage of the alignments obtained by other mapping methods.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Examples of covering and non-covering alignment results. <italic>x</italic>
<sub>
<italic>1</italic>
</sub>, <italic>x</italic>
<sub>
<italic>2</italic>
</sub>, <italic>x</italic>
<sub>
<italic>3</italic>
</sub>, and <italic>x</italic>
<sub>
<italic>4</italic>
</sub> are different alignment results reported by four mapping methods for the same query read. The dotted line indicates the mapped region in the reference genome sequence for the corresponding alignment result. In this figure, <italic>x</italic>
<sub>
<italic>1</italic>
</sub> and <italic>x</italic>
<sub>
<italic>2</italic>
</sub> alignments cover each other as they span the mapped regions on the reference genome that have at least 90% overlap. The alignments <italic>x</italic>
<sub>
<italic>1</italic>
</sub> and <italic>x</italic>
<sub>
<italic>2</italic>
</sub> cover <italic>x</italic>
<sub>
<italic>3</italic>
</sub> alignment but not the <italic>x</italic>
<sub>
<italic>4</italic>
</sub> alignment, while <italic>x</italic>
<sub>
<italic>3</italic>
</sub> does not cover <italic>x</italic>
<sub>
<italic>1</italic>
</sub> and <italic>x</italic>
<sub>
<italic>2</italic>
</sub> alignments. On the other hand, the alignment <italic>x</italic>
<sub>
<italic>4</italic>
</sub> does not cover either alignment <italic>x</italic>
<sub>
<italic>1</italic>
</sub>, <italic>x</italic>
<sub>
<italic>2</italic>
</sub>, or <italic>x</italic>
<sub>
<italic>3</italic>
</sub>.</p>
</caption>
<graphic xlink:href="fgene-13-890651-g004.tif"/>
</fig>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Agreement of different alignment methods for the real <italic>A. thaliana</italic> dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="center">kngMap</th>
<th align="center">rHAT</th>
<th align="center">smsMap</th>
<th align="center">lordFAST</th>
<th align="center">BLASR</th>
<th align="center">BWA-MEM</th>
<th align="center">GraphMap</th>
<th align="center">minimap2</th>
<th align="center">NGMLR</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">kngMap</td>
<td align="char" char=".">-</td>
<td align="char" char=".">79.98</td>
<td align="char" char=".">76.72</td>
<td align="char" char=".">84.59</td>
<td align="char" char=".">72.16</td>
<td align="char" char=".">77.03</td>
<td align="char" char=".">87.59</td>
<td align="char" char=".">78.82</td>
<td align="char" char=".">74.21</td>
</tr>
<tr>
<td align="left">rHAT</td>
<td align="char" char=".">87.58</td>
<td align="char" char=".">-</td>
<td align="char" char=".">79.80</td>
<td align="char" char=".">85.49</td>
<td align="char" char=".">79.91</td>
<td align="char" char=".">84.39</td>
<td align="char" char=".">86.22</td>
<td align="char" char=".">85.85</td>
<td align="char" char=".">79.82</td>
</tr>
<tr>
<td align="left">smsMap</td>
<td align="char" char=".">89.07</td>
<td align="char" char=".">79.30</td>
<td align="char" char=".">-</td>
<td align="char" char=".">85.67</td>
<td align="char" char=".">78.23</td>
<td align="char" char=".">77.44</td>
<td align="char" char=".">86.38</td>
<td align="char" char=".">79.12</td>
<td align="char" char=".">74.36</td>
</tr>
<tr>
<td align="left">lordFAST</td>
<td align="char" char=".">88.32</td>
<td align="char" char=".">81.45</td>
<td align="char" char=".">78.29</td>
<td align="char" char=".">-</td>
<td align="char" char=".">75.25</td>
<td align="char" char=".">80.00</td>
<td align="char" char=".">86.63</td>
<td align="char" char=".">81.14</td>
<td align="char" char=".">75.58</td>
</tr>
<tr>
<td align="left">BLASR</td>
<td align="char" char=".">90.08</td>
<td align="char" char=".">84.27</td>
<td align="char" char=".">89.22</td>
<td align="char" char=".">89.41</td>
<td align="char" char=".">-</td>
<td align="char" char=".">90.11</td>
<td align="char" char=".">88.85</td>
<td align="char" char=".">91.96</td>
<td align="char" char=".">82.58</td>
</tr>
<tr>
<td align="left">BWA-MEM</td>
<td align="char" char=".">90.71</td>
<td align="char" char=".">83.85</td>
<td align="char" char=".">84.66</td>
<td align="char" char=".">89.99</td>
<td align="char" char=".">84.78</td>
<td align="char" char=".">-</td>
<td align="char" char=".">89.06</td>
<td align="char" char=".">91.36</td>
<td align="char" char=".">83.02</td>
</tr>
<tr>
<td align="left">GraphMap</td>
<td align="char" char=".">92.94</td>
<td align="char" char=".">83.99</td>
<td align="char" char=".">79.67</td>
<td align="char" char=".">88.28</td>
<td align="char" char=".">75.95</td>
<td align="char" char=".">80.82</td>
<td align="char" char=".">-</td>
<td align="char" char=".">82.44</td>
<td align="char" char=".">78.25</td>
</tr>
<tr>
<td align="left">minimap2</td>
<td align="char" char=".">91.85</td>
<td align="char" char=".">84.93</td>
<td align="char" char=".">85.06</td>
<td align="char" char=".">90.03</td>
<td align="char" char=".">86.47</td>
<td align="char" char=".">91.61</td>
<td align="char" char=".">89.68</td>
<td align="char" char=".">-</td>
<td align="char" char=".">83.75</td>
</tr>
<tr>
<td align="left">NGLMR</td>
<td align="char" char=".">95.62</td>
<td align="char" char=".">89.91</td>
<td align="char" char=".">89.42</td>
<td align="char" char=".">93.46</td>
<td align="char" char=".">90.55</td>
<td align="char" char=".">94.44</td>
<td align="char" char=".">94.59</td>
<td align="char" char=".">96.34</td>
<td align="char" char=".">-</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Finally, we compared the performance of the tools on reads for which their alignments do not agree. The number of disagreeing alignments and average identity difference between the two methods for the inconsistent alignments are listed in <xref ref-type="table" rid="T5">Table 5</xref>. The numbers in each row indicate the number of reads for which the corresponding method reports alignments that do not cover alignments outputted by the other methods listed in the column. The positive/negative percentage in parentheses means that the average identity of the disagreeing alignments for the corresponding method is higher/lower than the average identity of the disagreeing alignments for the other methods listed in the column. For instance, there are 1,997 reads for which BLASR does not cover the alignments of kngMap. For those reads, BLASR produced alignments with an average of 5.83% lower identity than kngMap. On the contrary, there are 5,497 reads for which kngMap does not cover BLASR&#x2019;s alignments. For those reads, on average, kngMap alignments have 24.74% higher identity than those of BLASR. This indicates that kngMap can find more similar alignments than BLASR. The performance of the tools on <italic>E. coli</italic> and <italic>H</italic>. <italic>sapiens</italic> reads is provided in <xref ref-type="sec" rid="s11">Supplementary Tables S13 and S14</xref>. With a lack of the true mappings for the real dataset, the information in <xref ref-type="table" rid="T5">Table 5</xref> and <xref ref-type="sec" rid="s11">Supplementary Tables S13 and S14</xref> are some extra support for the fact that the alignments of our kngMap are reliable.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Performance of each pair methods on the <italic>A. thaliana</italic> dataset for which their alignments do not agree.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="center">kngMap</th>
<th align="center">rHAT</th>
<th align="center">smsMap</th>
<th align="center">lordFAST</th>
<th align="center">BLASR</th>
<th align="center">BWA-MEM</th>
<th align="center">GraphMap</th>
<th align="center">minimap2</th>
<th align="center">NGMLR</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">kngMap</td>
<td align="center">-</td>
<td align="center">4,014 (90.15)</td>
<td align="center">4,852 (9.46)</td>
<td align="center">3,054 (11.66)</td>
<td align="center">5,497 (24.74)</td>
<td align="center">4,496 (31.60)</td>
<td align="center">1,414 (&#x2212;1.22)</td>
<td align="center">3,893 (31.91)</td>
<td align="center">3,101 (40.99)</td>
</tr>
<tr>
<td align="left">rHAT</td>
<td align="center">2,524 (&#x2212;30.04)</td>
<td align="center">-</td>
<td align="center">4,270 (&#x2212;14.97)</td>
<td align="center">2,917 (&#x2212;19.06)</td>
<td align="center">3,975 (&#x2212;7.57)</td>
<td align="center">3,070 (&#x2212;11.22)</td>
<td align="center">1809 (&#x2212;27.80)</td>
<td align="center">2,537 (15.07)</td>
<td align="center">2034 (15.03)</td>
</tr>
<tr>
<td align="left">smsMap</td>
<td align="center">2,230 (&#x2212;6.06)</td>
<td align="center">4,258 (67.14)</td>
<td align="center">-</td>
<td align="center">2,920 (7.33)</td>
<td align="center">4,299 (22.66)</td>
<td align="center">4,503 (21.20)</td>
<td align="center">1700 (&#x2212;6.05)</td>
<td align="center">3,907 (20.95)</td>
<td align="center">3,089 (30.72)</td>
</tr>
<tr>
<td align="left">lordFAST</td>
<td align="center">2,383 (1.60)</td>
<td align="center">3,812 (97.58)</td>
<td align="center">4,591 (10.61)</td>
<td align="center">-</td>
<td align="center">4,996 (28.68)</td>
<td align="center">4,016 (33.68)</td>
<td align="center">1742 (&#x2212;0.29)</td>
<td align="center">3,553 (36.91)</td>
<td align="center">2,921 (40.55)</td>
</tr>
<tr>
<td align="left">BLASR</td>
<td align="center">1997 (&#x2212;5.83)</td>
<td align="center">3,212 (50.74)</td>
<td align="center">2,258 (&#x2212;3.29)</td>
<td align="center">2,184 (9.02)</td>
<td align="center">-</td>
<td align="center">2004 (3.31)</td>
<td align="center">1,407 (&#x2212;3.33)</td>
<td align="center">1,508 (0.59)</td>
<td align="center">1,609 (35.75)</td>
</tr>
<tr>
<td align="left">BWA-MEM</td>
<td align="center">1869 (1.55)</td>
<td align="center">3,311 (56.90)</td>
<td align="center">3,218 (2.43)</td>
<td align="center">2053 (14.00)</td>
<td align="center">3,092 (6.06)</td>
<td align="center">-</td>
<td align="center">1,330 (2.40)</td>
<td align="center">1,570 (6.17)</td>
<td align="center">1,489 (38.46)</td>
</tr>
<tr>
<td align="left">GraphMap</td>
<td align="center">1,388 (1.79)</td>
<td align="center">3,164 (94.77)</td>
<td align="center">4,083 (9.49)</td>
<td align="center">2,336 (15.95)</td>
<td align="center">4,765 (25.09)</td>
<td align="center">3,785 (33.64)</td>
<td align="center">-</td>
<td align="center">3,341 (33.88)</td>
<td align="center">2,918 (39.92)</td>
</tr>
<tr>
<td align="left">minimap2</td>
<td align="center">1,630 (2.58)</td>
<td align="center">3,062 (57.86)</td>
<td align="center">3,098 (2.53)</td>
<td align="center">2041 (15.33)</td>
<td align="center">2,765 (2.83)</td>
<td align="center">1737 (5.80)</td>
<td align="center">1,249 (&#x2212;0.14)</td>
<td align="center">-</td>
<td align="center">1,490 (39.72)</td>
</tr>
<tr>
<td align="left">NGLMR</td>
<td align="center">819 (&#x2212;21.91)</td>
<td align="center">1899 (52.86)</td>
<td align="center">2000 (&#x2212;5.35)</td>
<td align="center">1,229 (5.48)</td>
<td align="center">1780 (&#x2212;8.57)</td>
<td align="center">1,407 (&#x2212;10.16)</td>
<td align="center">742 (&#x2212;19.50)</td>
<td align="center">641 (&#x2212;24.43)</td>
<td align="center">-</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s4">
<title>4 Discussion</title>
<p>For SMS read mapping, the seed-chain-align procedure (<xref ref-type="bibr" rid="B20">Lin and Hsu, 2020</xref>), as is used by most mapping methods, is a classical and effective strategy to obtain the superior alignment results since it can help to seek the alignment skeleton consisting of matched segment pairs between the long SMS read and the reference genome. Generally, the seed-chain-align procedure can be summarized as three phases: first, matched <italic>k</italic>-mers are found as seeds in one read and in the reference sequence. Then, a group of seeds that are colinear or close to each other is chained as the candidate alignment skeleton. Finally, the non-seed fragments within the alignment skeleton are being extended to generate the base-level alignment. In the stage of generating the alignment skeleton, these methods are usually difficult to capture the matches with long distance, especially for the read parts with SVs and high sequencing errors since that there are few or no matched <italic>k</italic>-mers in the parts. Thus, most methods failed to generate the alignment skeleton of the query read or just build an alignment skeleton that covers a small part of the read. As a result, these methods cannot obtain the alignment result or just report a local mapping result for the query read, other than obtaining the whole end-to-end alignment, leading to low mapping sensitivity and aligned coverage.</p>
<p>To address the aforementioned challenge, here, we developed kngMap to increase the mapping sensitivity and with 100% aligned coverage for long noisy alignment. kngMap is also a seed-chain-align method using the hashing index technique, compared with other methods. There are three main key features of kngMap: 1) kngMap proposes a scoring strategy in the chaining procedure to choose a group of anchors to form the initial alignment skeleton for each read, even in the situation that the read contains SVs or some regions that matched seeds are dispersedly distributed caused by high sequencing errors; 2) defining an increased credibility function to refine the alignment skeleton, and then a high quality of alignment skeleton is subsequently obtained; and 3) for each of the gaps within the skeleton, kngMap classifies it into one of three categories and implements a specific alignment strategy to fill the corresponding gaps. The scoring strategy and increased credibility function ensures that kngMap can effectively find the alignment skeleton for every query read, even in the situation that the matched seeds are dispersedly distributed in the reference genome. Thus, kngMap can obtain higher mapping sensitivity, that is, align more reads and bases. The last feature can guarantee that the whole end-to-end alignment is obtained, not local alignment achieved by other methods. Therefore, the aligned coverage of kngMap is higher than that of other methods. In addition, in order to further evaluate the performance of the kngMap for a higher error rate up to &#x223c;15%, we applied the PBSIM2 simulator to generate another two simulated datasets with 20% (parameter: --accuracy-mean 20) and 25% (parameter: --accuracy-mean 25) error rates; the other parameter settings are similar to those given in <xref ref-type="sec" rid="s11">Supplementary Table S2</xref>. <xref ref-type="sec" rid="s11">Supplementary Table S14</xref> shows the evaluation result of kngMap, rHAT, smsMap, lordFAST, BLASR, BWA-MEM, GraphMap, minimap2, and NGMLR on the simulated datasets with 20% error rate. As can be seen from <xref ref-type="sec" rid="s11">Supplementary Table S14</xref>, kngMap correctly mapped the maximum number of reads and bases to the reference genome. Importantly, kngMap achieved 100% aligned coverage, demonstrating that all mapped reads by kngMap are completely aligned. For the base sensitivity and precision in <xref ref-type="sec" rid="s11">Supplementary Table S14</xref>, we can see that kngMap still performed the best, indicating that most bases are correctly mapped in the alignments generated by kngMap. Similar mapping results can be found in <xref ref-type="sec" rid="s11">Supplementary Table S15</xref> with simulated dataset with 25% error rate. Therefore, these results in <xref ref-type="sec" rid="s11">Supplementary Tables S14 and S15</xref> demonstrate that our kngMap is more robust to sequencing errors, and it can obtain better mapping results for a higher error rate up to &#x223c;15%.</p>
<p>The massive amount of long noisy read data produced by SMS technologies brings some challenges to existing mapping approaches. In addition to accuracy, computational complexity is another important issue that needs to be considered. The time complexity of kngMap has four main components: 1) In the phase of constructing the reference genome index, it needs to extract all <italic>k</italic>-mers of the reference genome sequences to build the hash table index; thus, the maximum complexity is in the order of <italic>O</italic>(<italic>G</italic>), where <italic>G</italic> is the length of the genome sequences. 2) In the phase of generating <italic>k</italic>-mer <italic>d</italic>-neighborhood graph, it needs to retrieve all the matches of the <italic>k</italic>-mers for each query read, so the maximum complexity is <italic>O</italic>(<italic>N&#x2a;L</italic>), where <italic>N</italic> is the number of reads, and <italic>L</italic> is the average read length. 3) In the phase of generating the skeleton of alignment, it needs to recurrently calculate the score of each node, so the maximum complexity is <italic>O</italic>(<italic>N</italic>&#x2a;<italic>M</italic>), where <italic>M</italic> is the average number of matched <italic>k</italic>-mers for all reads. 4) In the phase of filling the gaps between anchors, it needs to obtain the detailed base-to-base alignment, so the maximum complexity is <italic>O</italic>(<italic>K</italic>&#x2a;<italic>Q</italic>
<sup>
<italic>2</italic>
</sup>), where <italic>K</italic> is the average number of gaps, <italic>Q</italic> is the average length of gaps, and <italic>Q&#x3c;&#x3c;L</italic>. In summary, the total complexity for kngMap is <italic>O</italic>(<italic>G&#x2b;N&#x2a;L&#x2b;N&#x2a;M&#x2b;K&#x2a;Q</italic>
<sup>
<italic>2</italic>
</sup>). Since <italic>Q&#x3c;&#x3c;L</italic>, kngMap has a time complexity of the order of <italic>G</italic> and <italic>L</italic>. In order to graphically evaluate the computational efficiency of our kngMap, we compared kngMap with other mapping tools on the real <italic>H</italic>. <italic>sapiens</italic> datasets. <xref ref-type="fig" rid="F5">Figure 5</xref> shows the relative running time (wall-time) by using the nine tools. In terms of computational efficiency, we can see that the speed of kngMap is almost as fast as minimap2 and is several folds faster than other mapping methods. For example, kngMap is overall about 2-folds faster than rHAT and 6- to 9-folds faster than smsMap, lordFAST, BLASR, BWA-MEM, and GraphMap mappers. The speed comparison result described in <xref ref-type="fig" rid="F5">Figure 5</xref> indicates that kngMap is efficient to align SMS reads.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Relative speed of the aligners benchmarking on the real <italic>H</italic>. <italic>sapiens</italic> datasets. The relative speed of one mapper is defined as <italic>T</italic>(aln)/<italic>T</italic>(<italic>kngMap</italic>), where <italic>T</italic>(<italic>aln</italic>) and <italic>T</italic>(<italic>kngMap</italic>) are the alignment times of a compared aligner and kngMap, respectively. The lower the height of the bar, the faster the speed of the corresponding method. We can see that kngMap is faster than rHAT, smsMap, lordFAST, BLASR, BWA-MEM, GraphMap, and NGLMR mappers.</p>
</caption>
<graphic xlink:href="fgene-13-890651-g005.tif"/>
</fig>
</sec>
<sec id="s5">
<title>5 Conclusion</title>
<p>Since the infancy of SMS technologies (e.g., PacBio and Oxford Nanopore MinION) that produce longer but higher sequencing error reads, mapping these reads to a reference genome is often the most basic and computing-expensive step for downstream genome sequence analysis. Developing novel long read mapping tools is on demand for improving the mapping sensitivity and effectiveness of SMS read alignment.</p>
<p>In this study, we present kngMap, a new, fast, and highly sensitive mapping algorithm for long noisy reads. Mainly, kngMap contains three key characteristics: 1) kngMap proposes a scoring strategy in the chaining procedure to choose a group of anchors to form the initial alignment skeleton for each read. 2) kngMap designs an increased credibility function to refine the alignment skeleton. The scoring strategy and increased credibility function progressively ensure that kngMap can locate the aligned region for every query read, even in the situation that the matched seeds are dispersedly distributed in the reference genome. 3) For each of the gaps within the skeleton, kngMap classifies it into one of three categories and implements a specific alignment strategy to fill the corresponding gaps, which can help to robustly handle potentially different types of SVs in the reads. kngMap was benchmarked on simulated and real datasets across various genomes with other state-of-the-art mappers. The experimental results demonstrated that kngMap has higher accuracy and sensitivity that can correctly map more sequences and bases to the reference genome and achieves 100% aligned read coverage ratio; meanwhile, it also has good ability to span different types of SVs within the reads.</p>
</sec>
</body>
<back>
<sec id="s6">
<title>Data Availability Statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s11">Supplementary Material</xref>, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s7">
<title>Author Contributions</title>
<p>Z-GW wrote the draft manuscript and programmed the C&#x2b;&#x2b; source codes. X-GF and HZ participated in the benchmarking experiments, prepared several figures, and collected some tables. X-DZ helped download the source codes of other compared mappers and install them. FL designed the overall study and reviewed the manuscript. YQ and S-WZ helped in improving and revising the manuscript. All authors contributed to the conception and design of the study, participated in the analysis of the experimental results, and edited the manuscript. All authors read and approved the final manuscript.</p>
</sec>
<sec id="s8">
<title>Funding</title>
<p>This study was supported by the Scientific Research Program Funded by Shaanxi Provincial Education Department (No. 21JK0486), the Natural Science Basic Research Plan in Shaanxi Province of China (Nos. 2021JQ-811 and 2022JZ-03), and the National Natural Science Foundation of China (Nos. 61873202, 61473232, and 91430111).</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors, and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s11">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2022.890651/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2022.890651/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.docx" id="SM1" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alser</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Rotman</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Deshpande</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Taraszka</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Baykal</surname>
<given-names>P. I.</given-names>
</name>
</person-group>, <article-title>Technology Dictates Algorithms: Recent Developments in Read Alignment</article-title>
<italic>.</italic> <year>2021</year>. <volume>22</volume>(<issue>1</issue>): p. <fpage>1</fpage>&#x2013;<lpage>34</lpage>.<pub-id pub-id-type="doi">10.1186/s13059-021-02443-7</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bartenhagen</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Dugas</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>RSVSim: an R/Bioconductor Package for the Simulation of Structural Variations</article-title>. <source>Bioinformatics</source> <volume>29</volume> (<issue>13</issue>), <fpage>1679</fpage>&#x2013;<lpage>1681</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btt198</pub-id> </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Berlin</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Koren</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chin</surname>
<given-names>C.-S.</given-names>
</name>
<name>
<surname>Drake</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Landolin</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Phillippy</surname>
<given-names>A. M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Assembling Large Genomes with Single-Molecule Sequencing and Locality-Sensitive Hashing</article-title>. <source>Nat. Biotechnol.</source> <volume>33</volume> (<issue>6</issue>), <fpage>623</fpage>&#x2013;<lpage>630</lpage>. <pub-id pub-id-type="doi">10.1038/nbt.3238</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cao</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Hou</surname>
<given-names>Y. F.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>Z. G.</given-names>
</name>
</person-group>, <article-title>EdClust: A Heuristic Sequence Clustering Method with Higher Sensitivity</article-title>
<italic>.</italic> <source>J. Bioinform Comput. Biol.</source>, <year>2021</year>: <volume>20</volume>p. <fpage>2150036</fpage>. <pub-id pub-id-type="doi">10.1142/S0219720021500360</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chaisson</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Tesler</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Mapping Single Molecule Sequencing Reads Using Basic Local Alignment with Successive Refinement (BLASR): Application and Theory</article-title>. <source>Bmc Bioinformatics</source> <volume>13</volume> (<issue>1</issue>), <fpage>238</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-13-238</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chakraborty</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bandyopadhyay</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>conLSH: Context Based Locality Sensitive Hashing for Mapping of Noisy SMRT Reads</article-title>. <source>Comput. Biol. Chem.</source> <volume>85</volume>, <fpage>107206</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiolchem.2020.107206</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chakraborty</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Morgenstern</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Bandyopadhyay</surname>
<given-names>S. J. B. b.</given-names>
</name>
</person-group>, <article-title>S-conLSH: Alignment-free Gapped Mapping of Noisy Long Reads</article-title>
<italic>.</italic> <year>2021</year>. <volume>22</volume>(<issue>1</issue>): p. <fpage>1</fpage>&#x2013;<lpage>18</lpage>.<pub-id pub-id-type="doi">10.1186/s12859-020-03918-3</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Nie</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>S. Q.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>Y. F.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Bray</surname>
<given-names>T.</given-names>
</name>
</person-group>, <article-title>Efficient Assembly of Nanopore Reads via Highly Accurate and Intact Error Correction</article-title>
<italic>.</italic> <year>2021</year>. <volume>12</volume>(<issue>1</issue>): p. <fpage>1</fpage>&#x2013;<lpage>10</lpage>.<pub-id pub-id-type="doi">10.1038/s41467-020-20236-7</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Faust</surname>
<given-names>G. G.</given-names>
</name>
<name>
<surname>Hall</surname>
<given-names>I. M.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>YAHA: Fast and Flexible Long-Read Alignment with Optimal Breakpoint Detection</article-title>. <source>Bioinformatics</source> <volume>28</volume> (<issue>19</issue>), <fpage>2417</fpage>&#x2013;<lpage>2424</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bts456</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Haghshenas</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Sahinalp</surname>
<given-names>S. C.</given-names>
</name>
<name>
<surname>Hach</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>lordFAST: Sensitive and Fast Alignment Search Tool for LOng Noisy Read Sequencing Data</article-title>. <source>Bioinformatics</source>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty544</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Hayashi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Taura</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2013</year>). &#x201c;<article-title>Parallel and Memory-Efficient Burrows-Wheeler Transform</article-title>,&#x201d; in <conf-name>2013 IEEE International Conference on Big Data</conf-name> (<publisher-name>IEEE</publisher-name>). <pub-id pub-id-type="doi">10.1109/bigdata.2013.6691757</pub-id> </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ivan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>&#x160;iki&#x107;</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wilm</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Fenlon</surname>
<given-names>S. N.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Nagarajan</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Fast and Sensitive Mapping of Nanopore Sequencing Reads with GraphMap</article-title>. <source>Nat. Commun.</source> <volume>7</volume>, <fpage>11307</fpage>. <pub-id pub-id-type="doi">10.1038/ncomms11307</pub-id> </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kolmogorov</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Pevzner</surname>
<given-names>P. A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Assembly of Long, Error-Prone Reads Using Repeat Graphs</article-title>. <source>Nat. Biotechnol.</source> <volume>37</volume> (<issue>5</issue>), <fpage>540</fpage>&#x2013;<lpage>546</lpage>. <pub-id pub-id-type="doi">10.1038/s41587-019-0072-8</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Langmead</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Salzberg</surname>
<given-names>S. L.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Fast Gapped-Read Alignment with Bowtie 2</article-title>. <source>Nat. Methods</source> <volume>9</volume> (<issue>4</issue>), <fpage>357</fpage>&#x2013;<lpage>359</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.1923</pub-id> </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Langmead</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Trapnell</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Pop</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Salzberg</surname>
<given-names>S. L.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Ultrafast and Memory-Efficient Alignment of Short DNA Sequences to the Human Genome</article-title>. <source>Genome Biol.</source> <volume>10</volume> (<issue>3</issue>), <fpage>R25</fpage>. <pub-id pub-id-type="doi">10.1186/gb-2009-10-3-r25</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Laver</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Harrison</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>O&#x2019;Neill</surname>
<given-names>P. A.</given-names>
</name>
<name>
<surname>Moore</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Farbos</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Paszkiewicz</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Assessing the Performance of the Oxford Nanopore Technologies MinION</article-title>. <source>Biomol. Detect. Quantification</source> <volume>3</volume>, <fpage>1</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1016/j.bdq.2015.02.001</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
</person-group>, <article-title>Aligning Sequence Reads, Clone Sequences and Assembly Contigs with BWA-MEM</article-title>
<italic>.</italic> <year>2013</year>. <fpage>1303</fpage>. </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Minimap and Miniasm: Fast Mapping and De Novo Assembly for Noisy Long Sequences</article-title>. <source>Bioinformatics</source> <volume>32</volume> (<issue>14</issue>), <fpage>2103</fpage>&#x2013;<lpage>2110</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btw152</pub-id> </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Minimap2: Pairwise Alignment for Nucleotide Sequences</article-title>. <source>Bioinformatics</source> <volume>34</volume> (<issue>18</issue>), <fpage>3094</fpage>&#x2013;<lpage>3100</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty191</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>H. N.</given-names>
</name>
<name>
<surname>Hsu</surname>
<given-names>W. L.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>GSAlign: an Efficient Sequence Alignment Tool for Intra-species Genomes</article-title>. <source>BMC Genomics</source> <volume>21</volume> (<issue>1</issue>), <fpage>182</fpage>&#x2013;<lpage>210</lpage>. <pub-id pub-id-type="doi">10.1186/s12864-020-6569-1</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lindner</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Friedel</surname>
<given-names>C. C.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>A Comprehensive Evaluation of Alignment Algorithms in the Context of RNA-Seq</article-title>. <source>PLoS ONE</source> <volume>7</volume> (<issue>12</issue>), <fpage>e52403</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0052403</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lippert</surname>
<given-names>R. A.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Space-Efficient Whole Genome Comparisons with Burrows-Wheeler Transforms</article-title>. <source>J. Comput. Biol.</source> <volume>12</volume> (<issue>4</issue>), <fpage>407</fpage>&#x2013;<lpage>415</lpage>. <pub-id pub-id-type="doi">10.1089/cmb.2005.12.407</pub-id> </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>LAMSA: Fast Split Read Alignment with Long Approximate Matches</article-title>. <source>Bioinformatics</source> <volume>33</volume> (<issue>2</issue>), <fpage>192</fpage>&#x2013;<lpage>201</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btw594</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>deSALT: fast and accurate long transcriptomic read alignment with de Bruijn graph-based index</article-title>. <source>Genome Biol.</source> <volume>20</volume> (<issue>1</issue>), <fpage>274</fpage>&#x2013;<lpage>314</lpage>. <pub-id pub-id-type="doi">10.1186/s13059-019-1895-9</pub-id> </citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Guan</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Teng</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>rHAT: Fast Alignment of Noisy Long Reads with Regional Hashing</article-title>. <source>Bioinformatics</source> <volume>32</volume> (<issue>11</issue>), <fpage>1625</fpage>&#x2013;<lpage>1631</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btv662</pub-id> </citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Brudno</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>deBGA: read alignment with de Bruijn graph-based seed and extension</article-title>. <source>Bioinformatics</source> <volume>32</volume> (<issue>21</issue>), <fpage>3224</fpage>&#x2013;<lpage>3232</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btw371</pub-id> </citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>C.-M.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Yiu</surname>
<given-names>S.-M.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>SOAP3: Ultra-fast GPU-Based Parallel Alignment Tool for Short Reads</article-title>. <source>Bioinformatics</source> <volume>28</volume> (<issue>6</issue>), <fpage>878</fpage>&#x2013;<lpage>879</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bts061</pub-id> </citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
</person-group> , <article-title>SKSV: Ultrafast Structural Variation Detection from Circular Consensus Sequencing Reads</article-title>
<italic>.</italic> <year>2021</year>. <volume>37</volume>(<issue>20</issue>): p. <fpage>3647</fpage>&#x2013;<lpage>3649</lpage>.<pub-id pub-id-type="doi">10.1093/bioinformatics/btab341</pub-id> </citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Marchet</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Lecompte</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Silva</surname>
<given-names>C. D.</given-names>
</name>
<name>
<surname>Cruaud</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Aury</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Nicolas</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>De Novo clustering of Long Reads by Gene from Transcriptomics Data</article-title>. <source>Nucleic Acids Res.</source> <volume>47</volume> (<issue>1</issue>), <fpage>e2</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gky834</pub-id> </citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Marco-Sola</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sammeth</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Guig&#xf3;</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Ribeca</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>The GEM Mapper: Fast, Accurate and Versatile Alignment by Filtration</article-title>. <source>Nat. Methods</source> <volume>9</volume> (<issue>12</issue>), <fpage>1185</fpage>&#x2013;<lpage>1188</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.2221</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Michael</surname>
<given-names>S. C.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>CloudBurst: Highly Sensitive Read Mapping with MapReduce</article-title>. <source>Bioinformatics</source> <volume>25</volume> (<issue>11</issue>), <fpage>1363</fpage>&#x2013;<lpage>1369</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp236</pub-id> </citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ning</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Cox</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Mullikin</surname>
<given-names>J. C.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>SSAHA: a Fast Search Method for Large DNA Databases</article-title>. <source>Genome Res.</source> <volume>11</volume> (<issue>10</issue>), <fpage>1725</fpage>&#x2013;<lpage>1729</lpage>. <pub-id pub-id-type="doi">10.1101/gr.194201</pub-id> </citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ono</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Asai</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hamada</surname>
<given-names>M. J. B.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>PBSIM2: a Simulator for Long-Read Sequencers with a Novel Generative Model of Quality Scores</article-title>. <source>Bioinformatics</source> <volume>37</volume> (<issue>5</issue>), <fpage>589</fpage>&#x2013;<lpage>595</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa835</pub-id> </citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ono</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Asai</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hamada</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>PBSIM: PacBio Reads Simulator-Toward Accurate Genome Assembly</article-title>. <source>Bioinformatics</source> <volume>29</volume> (<issue>1</issue>), <fpage>119</fpage>&#x2013;<lpage>121</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bts649</pub-id> </citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Re-alignment of the Unmapped Reads with Base Quality Score</article-title>. <source>Bmc Bioinformatics</source> <volume>6 Suppl 5</volume> (<issue>Suppl. 5</issue>), <fpage>S8</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-16-s5-s8</pub-id> </citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Prezza</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Vezzi</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>K&#xe4;ller</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Policriti</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Fast, Accurate, and Lightweight Analysis of BS-Treated Reads with ERNE 2</article-title>. <source>BMC Bioinformatics</source> <volume>17 Suppl 4</volume> (<issue>4</issue>), <fpage>69</fpage>&#x2013;<lpage>245</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-016-0910-3</pub-id> </citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ren</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chaisson</surname>
<given-names>M. J. P.</given-names>
</name>
<name>
<surname>lra</surname>
</name>
</person-group> (<year>2021</year>). <article-title>Lra: A Long Read Aligner for Sequences and Contigs</article-title>. <source>Plos Comput. Biol.</source> <volume>17</volume> (<issue>6</issue>), <fpage>e1009078</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1009078</pub-id> </citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rhoads</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Au</surname>
<given-names>K. F.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>PacBio Sequencing and its Applications</article-title>. <source>Genomics, proteomics &#x26; bioinformatics</source> <volume>13</volume> (<issue>5</issue>), <fpage>278</fpage>&#x2013;<lpage>289</lpage>. <pub-id pub-id-type="doi">10.1016/j.gpb.2015.08.002</pub-id> </citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schmieder</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Edwards</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Fast Identification and Removal of Sequence Contamination from Genomic and Metagenomic Datasets</article-title>. <source>Plos One</source> <volume>6</volume> (<issue>3</issue>), <fpage>e17288</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0017288</pub-id> </citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sedlazeck</surname>
<given-names>F. J.</given-names>
</name>
<name>
<surname>Rescheneder</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Smolka</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Nattestad</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>von Haeseler</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Accurate Detection of Complex Structural Variations Using Single-Molecule Sequencing</article-title>. <source>Nat. Methods</source> <volume>15</volume> (<issue>6</issue>), <fpage>461</fpage>&#x2013;<lpage>468</lpage>. <pub-id pub-id-type="doi">10.1038/s41592-018-0001-7</pub-id> </citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sedlazeck</surname>
<given-names>F. J.</given-names>
</name>
<name>
<surname>Rescheneder</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Von Haeseler</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>NextGenMap: Fast and Accurate Read Mapping in Highly Polymorphic Genomes</article-title>. <source>Bioinformatics</source> <volume>29</volume> (<issue>21</issue>), <fpage>2790</fpage>&#x2013;<lpage>2791</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btt468</pub-id> </citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>St&#xf6;cker</surname>
<given-names>B. K.</given-names>
</name>
<name>
<surname>K&#xf6;ster</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Rahmann</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>SimLoRD: Simulation of Long Read Data</article-title>. <source>Bioinformatics</source> <volume>32</volume> (<issue>17</issue>), <fpage>2704</fpage>&#x2013;<lpage>2706</lpage>. </citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>Z.-G.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.-W.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>smsMap: Mapping Single Molecule Sequencing Reads by Locating the Alignment Starting Positions</article-title>. <source>BMC Bioinformatics</source> <volume>21</volume> (<issue>1</issue>), <fpage>341</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-020-03698-w</pub-id> </citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>Z.-G.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.-W.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>NPBSS: a New PacBio Sequencing Simulator for Generating the Continuous Long Reads with an Empirical Model</article-title>. <source>BMC Bioinformatics</source> <volume>19</volume> (<issue>1</issue>), <fpage>177</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-018-2208-0</pub-id> </citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Chu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Warren</surname>
<given-names>R. L.</given-names>
</name>
<name>
<surname>Birol</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>NanoSim: Nanopore Sequence Read Simulator Based on Statistical Characterization</article-title>. <source>Gigascience</source> <volume>6</volume> (<issue>4</issue>), <fpage>1</fpage>&#x2013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1093/gigascience/gix010</pub-id> </citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Schmidt</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Fast and Efficient Short Read Mapping Based on a Succinct Hash index</article-title>. <source>Bmc Bioinformatics</source> <volume>19</volume> (<issue>1</issue>), <fpage>92</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-018-2094-5</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>