<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Cell Dev. Biol.</journal-id>
<journal-title>Frontiers in Cell and Developmental Biology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Cell Dev. Biol.</abbrev-journal-title>
<issn pub-type="epub">2296-634X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fcell.2021.643645</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Cell and Developmental Biology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Fast and Accurate Classification of Meta-Genomics Long Reads With deSAMBA</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Li</surname> <given-names>Gaoyang</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1176755/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Liu</surname> <given-names>Yongzhuang</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x02020;</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Li</surname> <given-names>Deying</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x02020;</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Liu</surname> <given-names>Bo</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Li</surname> <given-names>Junyi</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/944248/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Hu</surname> <given-names>Yang</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/534353/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Wang</surname> <given-names>Yadong</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x0002A;</sup></xref>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Center for Bioinformatics, School of Computer Science and Technology, Harbin Institute of Technology</institution>, <addr-line>Harbin</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Internal Medicine, General Hospital of Heilongjiang Province Land Reclamation Bureau</institution>, <addr-line>Harbin</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>School of Computer Science and Technology, Harbin Institute of Technology (Shenzhen)</institution>, <addr-line>Shenzhen</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>School of Life Science and Technology, Harbin Institute of Technology</institution>, <addr-line>Harbin</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Lei Deng, Central South University, China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Xin Gao, King Abdullah University of Science and Technology, Saudi Arabia; Le Zhang, Sichuan University, China</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Yang Hu <email>huyang&#x00040;hit.edu.cn</email></corresp>
<corresp id="c002">Yadong Wang <email>ydwang&#x00040;hit.edu.cn</email></corresp>
<fn fn-type="other" id="fn001"><p>This article was submitted to Molecular Medicine, a section of the journal Frontiers in Cell and Developmental Biology</p></fn>
<fn fn-type="other" id="fn002"><p>&#x02020;These authors share first authorship</p></fn></author-notes>
<pub-date pub-type="epub">
<day>28</day>
<month>04</month>
<year>2021</year>
</pub-date>
<pub-date pub-type="collection">
<year>2021</year>
</pub-date>
<volume>9</volume>
<elocation-id>643645</elocation-id>
<history>
<date date-type="received">
<day>18</day>
<month>12</month>
<year>2020</year>
</date>
<date date-type="accepted">
<day>09</day>
<month>02</month>
<year>2021</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2021 Li, Liu, Li, Liu, Li, Hu and Wang.</copyright-statement>
<copyright-year>2021</copyright-year>
<copyright-holder>Li, Liu, Li, Liu, Li, Hu and Wang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license></permissions>
<abstract><p>There is still a lack of fast and accurate classification tools to identify the taxonomies of noisy long reads, which is a bottleneck to the use of the promising long-read metagenomic sequencing technologies. Herein, we propose de Bruijn graph-based Sparse Approximate Match Block Analyzer (deSAMBA), a tailored long-read classification approach that uses a novel pseudo alignment algorithm based on sparse approximate match block (SAMB). Benchmarks on real sequencing datasets demonstrate that deSAMBA enables to achieve high yields and fast speed simultaneously, which outperforms state-of-the-art tools and has many potentials to cutting-edge metagenomics studies.</p></abstract>
<kwd-group>
<kwd>long read</kwd>
<kwd>pseudo alignment</kwd>
<kwd>de Bruijn graph-based index</kwd>
<kwd>metagenomics 16S</kwd>
<kwd>read classification</kwd>
</kwd-group>
<counts>
<fig-count count="3"/>
<table-count count="0"/>
<equation-count count="7"/>
<ref-count count="29"/>
<page-count count="9"/>
<word-count count="5652"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>Introduction</title>
<p>Metagenomic sequencing is ubiquitously applied to comprehensively study environmental samples (Meth&#x000E9; et al., <xref ref-type="bibr" rid="B24">2012</xref>; Gilbert et al., <xref ref-type="bibr" rid="B8">2014</xref>; Cheng et al., <xref ref-type="bibr" rid="B6">2020</xref>). It enables to reveal the compositions of microbial communities in various environments and study the functions of microbial communities and their interactions to environments. Furthermore, many new species can be discovered without cultivation in laboratories. With the rapid development of high-throughput sequencing technologies, metagenomic sequencing is promising for the analysis of microbiome. Especially due to its ability of real-time and portable sequencing of the samples (Quick et al., <xref ref-type="bibr" rid="B26">2016</xref>), long-read sequencing technologies have enormous potential to metagenomic studies. However, with the characteristics of long-read sequencing data, analytical challenges still remain.</p>
<p>In metagenomic studies, a fundamental task is to recognize the composition of the microbial community of the sequenced sample. With the ever-increasing number of sequenced genomes, it is feasible to accomplish this task by using the libraries of assembled genomes [e.g., RefSeq (Pruitt et al., <xref ref-type="bibr" rid="B25">2014</xref>)] as reference to implement the taxonomy classification of sequencing reads. A common approach is to align the reads against the reference (Altschul et al., <xref ref-type="bibr" rid="B1">1990</xref>; Huson et al., <xref ref-type="bibr" rid="B16">2007</xref>; Cheng, <xref ref-type="bibr" rid="B5">2019</xref>); however, this is not viable to handle a large amount of metagenomic reads (hundreds of gigabases) due to a low processing speed. Moreover, there are several specific technical issues in the classification of metagenomic long reads, which makes it an even more difficult computational task. Firstly, most of the long reads produced by mainstream platforms (such as ONT and PacBio platforms) are error-prone, which requires read classifiers to be noise-robust. Secondly, the reference is usually incomplete, i.e., many reads could be from unknown genomes, which requires read classifiers to handle the divergences between the sequenced genomes and their related genomes in the reference well. Thirdly, there are many common sequences among closely related genomes (e.g., various strains of bacteria species), which require read classifiers to handle the ubiquitous repeats in the reference well. Most state-of-the-art tools, such as Kraken (Wood and Salzberg, <xref ref-type="bibr" rid="B29">2014</xref>; Wood et al., <xref ref-type="bibr" rid="B28">2019</xref>), Centrifuge (Kim et al., <xref ref-type="bibr" rid="B17">2016</xref>), Kaiju (Menzel et al., <xref ref-type="bibr" rid="B23">2016</xref>), MetaOthello (Liu et al., <xref ref-type="bibr" rid="B22">2018</xref>), are designed for short reads. Generally, they use pseudo alignments, i.e., the exact or approximate matches from reads to reference as signals to achieve fast speed without loss of accuracy on short reads. However, most of them rely on the assumptions on the long exact matches and/or low divergences between reads and reference, which might fail at the issues mentioned above. Long-read aligners (Li, <xref ref-type="bibr" rid="B19">2018</xref>) can be used as alternatives; however, they are still time-consuming, which could not be well-suited for large-scale datasets and/or real-time tasks.</p>
<p>Herein, we present de Bruijn graph-based Sparse Approximate Match Block Analyzer (deSAMBA), a novel approximate match-based pseudo alignment approach for the classification of long reads. deSAMBA is motivated by the fact (Chaisson and Tesler, <xref ref-type="bibr" rid="B4">2012</xref>) that sequencing errors are unevenly distributed along the reads. Many long-read aligners (Chaisson and Tesler, <xref ref-type="bibr" rid="B4">2012</xref>; Li, <xref ref-type="bibr" rid="B18">2013</xref>, <xref ref-type="bibr" rid="B19">2018</xref>; Sedlazeck et al., <xref ref-type="bibr" rid="B27">2018</xref>; Hu et al., <xref ref-type="bibr" rid="B15">2019</xref>, <xref ref-type="bibr" rid="B14">2020</xref>, <xref ref-type="bibr" rid="B13">2021</xref>; Govindaraj et al., <xref ref-type="bibr" rid="B9">2020</xref>; Hasan et al., <xref ref-type="bibr" rid="B12">2020a</xref>,<xref ref-type="bibr" rid="B11">b</xref>) also take advantage of this model to find short exact matches (i.e., &#x0201C;seeds&#x0201D;); however, deSAMBA looks for longer approximate match blocks between reads and reference. Previous studies (Liu et al., <xref ref-type="bibr" rid="B20">2017</xref>) indicate that such blocks can be specifically mapped to reference under the circumstance of sequencing noise, so it is possible for them to become noise-robust features for read classification.</p></sec>
<sec sec-type="results" id="s2">
<title>Results</title>
<sec>
<title>Overview of de Bruijn Graph-Based Sparse Approximate Match Block Analyzer Approach</title>
<p>deSAMBA is composed of some tailored designs and implementations to achieve high yields and fast speed simultaneously. Basically, it uses Unitig&#x02013;Burrows&#x02013;Wheeler transform (BWT) data structure (Guan et al., <xref ref-type="bibr" rid="B10">2018</xref>) to index the de Bruijn graph of reference sequences and finds highly similar approximate match blocks through the index. These blocks are called sparse approximate match blocks (SAMBs), as they are usually sparsely placed along reads. Mainly, deSAMBA recognizes the taxonomy of a give read in the following four major steps.</p>
<list list-type="order">
<list-item><p>deSAMBA partitions the read into a series of segments. For each of the segments, it finds the local region having highest number of consecutive k-mer matches as a seed block.</p></list-item>
<list-item><p>For each of the seeding blocks, deSAMBA retrieves a set of maximal exact matches (MEMs) to the unitigs of the reference and extends the MEMs to generate SAMBs by local alignment.</p></list-item>
<list-item><p>deSAMBA greedily merges the SAMBs and extends the merged SAMBs by sparse dynamic programming (SDP)-based pseudo alignment against local reference sequences.</p></list-item>
<list-item><p>deSAMBA scores the extended SAMBs and identifies the taxonomy of the read by the highest scored SAMB. Moreover, it also supports to output the SAMBs as pseudo alignment results.</p></list-item>
</list>
<p>A schematics illustration is in <xref ref-type="fig" rid="F1">Figure 1</xref>.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>A schematics illustration of de Bruijn graph-based Sparse Approximate Match Block Analyzer (deSAMBA) approach. <bold>(A)</bold> The generation of seed block. A read is partitioned into fixed length segments. For a segment, all its <italic>l</italic>-mers (marked by blue round dots) are matched to reference through a bloom filter-based index, and the local region having the largest number of consecutive <italic>l</italic>-mer matches (marked by a blue dashed rectangle) is determined as a &#x0201C;seed block.&#x0201D; <bold>(B)</bold> The generation of initial sparse approximate match blocks (SAMBs). The seed blocks are matched to the unitigs of the RdBG through the Unitig&#x02013;Burrows&#x02013;Wheeler transform (BWT) index. Each of the matches is extended to a U-MEM. If a U-MEM is distanced from the two ends of the matched unitig by at least 12 bp (like the U-MEM in blue color), deSAMBA extends it to an approximate match by aligning the corresponding read part against the unitig. Further, the generated approximate match is mapped to the various copies of the unitig to be initial SAMBs (the blue unitig has 3 copies in this case). Moreover, if a U-MEM is within 12 bp of either end of the unitig (like the U-MEM in yellow color), deSAMBA maps the U-MEM to the various copies of the unitig at first (the yellow unitig has three copies in this case), and the mapped matches are as R-MEMs. For each of the R-MEMs, deSAMBA separately aligns the corresponding read part against the local reference sequence to generate a distinct initial SAMB. <bold>(C)</bold> The generation of extended SAMB. deSAMBA merges nearby initial SAMBs to generate a SAMB chain. In the figure, three initial SAMBs are chained, and the read part of the corresponding SAMB chain is extracted. Meanwhile, a local reference sequence is also extracted by extending the upstream and downstream boundaries of the SAMB chain on reference (1,000 bp are extended for both upstream and downstream). A hash table is then built for read to find all the short matches between the local reference and the read part. Further, a sparse dynamic programming (SDP)-based pseudo alignment is implemented between the local reference and the read part to generate an extended SAMB.</p></caption>
<graphic xlink:href="fcell-09-643645-g0001.tif"/>
</fig></sec>
<sec>
<title>Benchmark on Pseudo Metagenomic Datasets</title>
<p>We benchmarked deSAMBA with a series of pseudo metagenomic datasets and a real mock metagenome dataset. At first, we employed 145 real datasets from single genomes as pseudo metagenomic datasets (86 ONT datasets and 59 PacBio datasets; <xref ref-type="supplementary-material" rid="SM1">Supplementary Table 1</xref>) and implemented deSAMBA, Centrifuge, Kaiju, and Minimap2 on them. Here, 16,284 complete genomes (8,621 bacterial, 251 archaea, and 7,412 viral genomes, totaling &#x0007E;35 Gbp) downloaded from NCBI RefSeq were used as reference. There are 49 ONT and 22 PacBio datasets (called &#x0201C;WR datasets&#x0201D;) whose ground truth genomes are in the reference, and for each of the remaining datasets (called &#x0201C;NR datasets&#x0201D;), there is at least one genome in the reference having common ancestry at species or genus level to its ground truth genome. The sensitivity, accuracy, F1 score, and speed of read classification were assessed. Refer to <xref ref-type="supplementary-material" rid="SM1">Supplementary Notes</xref> for more details about the implementation of the benchmark.</p>
<p>Primarily, four issues are observed from the results (<xref ref-type="fig" rid="F2">Figure 2</xref>, <xref ref-type="supplementary-material" rid="SM1">Supplementary Figure 1</xref>):</p>
<list list-type="order">
<list-item><p>deSAMBA has good classification yields. deSAMBA had slightly higher F1 score than that of Minimap2 and outperformed Centrifuge and Kaiju on both of sensitivity and accuracy (<xref ref-type="fig" rid="F2">Figure 2A</xref>, <xref ref-type="supplementary-material" rid="SM1">Supplementary Figures 1A,B</xref>). Furthermore, we assessed the classifications on the WR datasets (<xref ref-type="fig" rid="F2">Figure 2B</xref>, <xref ref-type="supplementary-material" rid="SM1">Supplementary Figures 1C,D</xref>), and similar trends were observed. These results indicate that Centrifuge and Kaiju are easier to be affected by sequencing errors, even if the reads are from known genomes. This is mainly because such short read toward approaches rely on the assumptions of long exact matches and/or few divergences between reads and reference, which does not stand for long reads. However, as approximate matches, SAMBs are much more noise-robust, which helps find the signatures to implement a precise classification.</p></list-item>
<list-item><p>deSAMBA has good ability to classify the reads from unknown genomes. We assessed the classifications on the NR datasets (<xref ref-type="fig" rid="F2">Figure 2C</xref>, <xref ref-type="supplementary-material" rid="SM1">Supplementary Figures 1E,F</xref>) and observed that deSAMBA also had the highest F1 score. We investigated the intermediate results and found that this is because the generated SAMBs usually had relatively large lengths and low edit distances, so that most of them were specifically mapped to the reference genomes closely related to the ground truth genomes of NR datasets. Moreover, all the approaches have reduced sensitivities and accuracies on NR datasets mainly due to the divergences between the ground truth genomes and their related genomes in the reference.</p></list-item>
<list-item><p>deSAMBA enables to identify the reads from various strains, which is an on-demand function. To assess this ability, we did an assessment with six ONT datasets having strain-level labels (lines 13, 14, 19, 24, 31, 48 of <xref ref-type="supplementary-material" rid="SM1">Supplementary Table 1</xref>) that a read was considered as correctly classified only if it was assigned to its ground truth strain. deSAMBA outperformed Minimaps2 slightly and Centrifuge and Kaiju significantly (<xref ref-type="supplementary-material" rid="SM1">Supplementary Figure 2</xref>).</p></list-item>
<list-item><p>deSAMBA has fast speed. deSAMBA is about 4 and 2 times faster than Minimap2 and Kaiju, respectively, and slower than Centrifuge (<xref ref-type="fig" rid="F2">Figure 2D</xref>, <xref ref-type="supplementary-material" rid="SM1">Supplementary Figures 1G,H</xref>). Moreover, deSAMBA also had a nearly linear speedup with increasing number of CPU threads (<xref ref-type="supplementary-material" rid="SM1">Supplementary Figure 3</xref>). Besides, the memory footprint of deSAMBA is 69 GB. Comparing to that of Centrifuge (9 GB), Kaiju (31 GB), and Minimap2 (76 GB), this is acceptable, especially for modern servers.</p></list-item>
</list>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>The results of various approaches on the 145 pseudo-metagenomic datasets. <bold>(A&#x02013;C)</bold> The average sensitivity, accuracy, and F1 score on all the 145 pseudo-metagenomic datasets <bold>(A)</bold>, the 71 WR datasets <bold>(B)</bold>, and the 74 NR datasets <bold>(C)</bold>. The sensitivity, accuracy, and F1 score are defined as S &#x0003D; <italic>N</italic><sub><italic>TP</italic></sub>/<italic>N</italic><sub><italic>T</italic></sub>, A &#x0003D; <italic>N</italic><sub><italic>TP</italic></sub>/<italic>N</italic><sub><italic>C</italic></sub>, and F1 &#x0003D; 2<italic>SA</italic>/ (<italic>S</italic>&#x0002B;<italic>A</italic>), respectively, where <italic>N</italic><sub><italic>T</italic></sub>, <italic>N</italic><sub><italic>C</italic></sub>, and <italic>N</italic><sub><italic>TP</italic></sub> are, respectively, the total numbers of all the reads, the reads being classified, and the reads being correctly classified. <bold>(D)</bold> The speed of the approaches, which was assessed by Kbp processed per second with eight CPU threads.</p></caption>
<graphic xlink:href="fcell-09-643645-g0002.tif"/>
</fig>
<p>Overall, deSAMBA achieved a better balance between yields and speed than state-of-the-art pseudo alignment-based approaches, and its fast speed more suited to large-scale datasets and real-time tasks than long-read aligners. It is also observed that deSAMBA had &#x0003C;80% sensitivities on seven WR datasets (<xref ref-type="supplementary-material" rid="SM1">Supplementary Figure 4</xref>), indicating that they were still not handled well. We further aligned those reads directly to their reference by BLASTN (Boratyn et al., <xref ref-type="bibr" rid="B3">2013</xref>). The results (<xref ref-type="supplementary-material" rid="SM1">Supplementary Figure 5</xref>) indicated that, on average, BLASTN failed to align nearly 30% of the reads, similar to the classification results of deSAMBA. So, we realized that these datasets could have relatively poor sequencing quality, which affect the yields of deSAMBA.</p></sec>
<sec>
<title>Benchmark on Real Mock Community Dataset</title>
<p>Further, we benchmarked deSAMBA, Minimap2, Centrifuge, Kaiju, and MetaMaps (Dilthey et al., <xref ref-type="bibr" rid="B7">2019</xref>) with a real ONT dataset (SRA accession number: <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="ERR3200811">ERR3200811</ext-link>, 367173 reads and 2.36G bases in total) from a mock community [GIS20 (Bertrand et al., <xref ref-type="bibr" rid="B2">2019</xref>)] that consists of 20 species with abundances range from 0.1 to 30% (<xref ref-type="supplementary-material" rid="SM1">Supplementary Table 2</xref>). We also composed a set of ground truth taxonomies (GTTs) for the assessment (<xref ref-type="supplementary-material" rid="SM1">Supplementary Table 2</xref>). That is, if a GIS20 species has its own genome in the reference library, the taxonomy ID of the genome was used as the GTT of the species; otherwise, if there are genomes in the reference library that have a common ancestry at species or genus level to the GIS20 species, the taxonomy ID of the lowest common ancestry was used as the GTT. It is also worth noting that two GIS20 species did not have GTTs, since there was no genome in the reference that has a common ancestry to them at the genus or lower level.</p>
<p>The sensitivities and false discovery rates (FDRs) of the approaches were assessed. The results (<xref ref-type="fig" rid="F3">Figures 3A,B</xref>) indicate that deSAMBA had the highest sensitivity and lowest FDR, indicating that overall, it had the best yields on the GIS20 dataset. It is worth noting that the FDRs of all the approaches are quite high (&#x0003E;10%). This is mainly because 14 of the GIS20 species had strain-level GTTs, i.e., there were not only their own genomes but also other strains of the same species in the reference. Under such circumstance, the classification is considered as correct only if the bases are classified to the correct strains. However, there are ubiquitous long common sequences among the various strains of the species, so that it was extremely difficult to recognize the strain-level taxonomy entities of the reads. We further used the corresponding species-level taxonomy IDs as GTTs for the 14 GIS20 species and reevaluated the classification results and observed much lower FDRs (<xref ref-type="fig" rid="F3">Figure 3C</xref>).</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>The results of various approaches on GIS20 mock metagenome dataset. <bold>(A&#x02013;C)</bold> The sensitivity, false discovery rate (&#x0201C;FDR&#x0201D;), false discovery rate at species or higher level (&#x0201C;FDR-species&#x0201D;) of various approaches on GIS20 mock metagenome dataset. Herein, the sensitivity is defined as <italic>N</italic><sub><italic>G</italic></sub>/<italic>N</italic><sub><italic>T</italic></sub>, and the FDR is defined as <italic>N</italic><sub><italic>F</italic></sub>/<italic>N</italic><sub><italic>C</italic></sub>, where <italic>N</italic><sub><italic>T</italic></sub> is the total number of bases in the dataset, <italic>N</italic><sub><italic>G</italic></sub> is the number of bases being classified to ground truth taxonomies (GTTs), <italic>N</italic><sub><italic>C</italic></sub> is the number of bases being classified, and <italic>N</italic><sub><italic>F</italic></sub> is the number of bases being classified to non-GTT taxonomies. It is worth noting that species-level taxonomy IDs were used as GTTs instead of strain-level ones when calculating FDR-species. <bold>(D)</bold> Log-transformed Pearson&#x00027;s correlation (&#x0201C;LTP&#x0201D;) between the proportions of bases classified to GTTs and the corresponding expected abundances. It is worth noting that the proportions of 18 GTTs were used to calculate the correlations, since two of the GIS20 species did not have their GTTs in the reference. <bold>(E)</bold> The speed of the approaches, which was assessed by Kbp processed per second with eight CPU threads.</p></caption>
<graphic xlink:href="fcell-09-643645-g0003.tif"/>
</fig>
<p>We further assessed the proportions of bases being classified to various GTTs. Log-transformed Pearson&#x00027;s correlations between the proportions of classified bases to GTTs and the corresponding expected abundances were calculated. The results (<xref ref-type="fig" rid="F3">Figure 3D</xref>) suggested that deSAMBA achieved the highest correlations, indicating that its classifications mostly coincide with the ground truths of the dataset. Moreover, no large divergence was observed between the proportions of classified bases and their corresponding GTTs, which suggests that deSAMBA has the ability to handle various species well. The correlation of Kaiju was quite low mainly because it has poor ability to produce correct strain-level classifications.</p>
<p>The speed of the approaches was also assessed (<xref ref-type="fig" rid="F3">Figure 3E</xref>), and Centrifuge was still the fastest (5,870 Kbp/s). deSAMBA (2,860 Kbp/s) was the best runner-up and the fastest long read-toward approach, i.e., about 3.5 and 18.5 times faster than that of Minimap2 (803 Kbp/s) and MetaMaps (155 Kbp/s).</p></sec></sec>
<sec sec-type="discussion" id="s3">
<title>Discussion</title>
<p>Due to the combination of the high sequencing errors, the large sequence divergences between sequenced unknown genomes and reference genomes and the large size of reference sequences, it is still a non-trivial task to implement fast and accurate long-read classification. With the rapid growth of long-read sequencing metagenomics data, it has become a pressing need to develop more advanced computational approaches to break through this bottleneck with the use of metagenomic long reads. Herein, we present deSAMBA to show how to use SAMBs as a kind of useful signal to implement fast and accurate metagenomic read classification. Mainly, deSAMBA has three advantages to long-read classification as follows:</p>
<p>Firstly, as approximate matches, SAMBs enable to better handle the sequencing noise and the divergences between reference and related genomes, and this feature helps to achieve higher sensitivity than that of short read toward algorithms that usually use exact matches or only allow low divergences between reads and reference.</p>
<p>Secondly, as longer matches, SAMBs are feasible to handle the ubiquitous repeats in reference, and this feature helps improve the specificity of the matches and effectively narrow down the searching space during read classification, which paves the way to implement accurate classification. Moreover, the narrowed searching space also helps accelerate read classification speed.</p>
<p>Thirdly, deSAMBA has several tailored implementations, especially on the Unitig&#x02013;BWT index and pseudo alignment method. They help to achieve high performance with moderate cost of computational resources that is affordable to modern servers and high-performance clusters (HPCs). This is well-suited to large-scale datasets and real-time tasks.</p>
<p>The benchmark on a series of real sequencing datasets suggests that deSAMBA improves the yields of long-read classification substantially, comparing to state-of-the-art pseudo alignment-based read classification tools. Meanwhile, deSAMBA can produce equally good classifications to state-of-the-art long aligners, while it is times faster. Considering its yields and performance, deSAMBA achieves a good balance, and it is a promising productivity tool in metagenomic data analysis. We believe that deSAMBA has enormous potential to cutting-edge metagenomic studies.</p></sec>
<sec sec-type="methods" id="s4">
<title>Methods</title>
<sec>
<title>The Indexing of Reference Sequences</title>
<p>deSAMBA organizes the reference sequences in a de Bruijn graph-based approach, which is initially proposed in deBGA (Liu et al., <xref ref-type="bibr" rid="B21">2016</xref>). To reduce memory use, we used Unitig&#x02013;BWT data structure (Guan et al., <xref ref-type="bibr" rid="B10">2018</xref>) to index the unitig of the de Bruijn graph of the given reference sequences. More precisely, deSAMBA constructs the de Bruijn graph of reference sequences at first and extracts the unitigs. The BWT is then constructed for the concatenated unitigs as the Unitig&#x02013;BWT of the reference sequences. This indexing approach has good balance between retrieval speed and RAM space cost. It is also worth noting that instead of assigning a unique taxonomy ID like the previous study (Guan et al., <xref ref-type="bibr" rid="B10">2018</xref>), deSAMBA maintains a position list for each of the unitigs. For a certain unitig, each item of the position list records a genomic position in reference, which represents the location of a specific copy of the unitig.</p>
<p>In addition to the Unitig&#x02013;BWT index, deSAMBA also builds a bloom filter-based index, which is used as an auxiliary index to the generation of SAMBs. The bloom filter-based index enables to give a quick answer whether a k-mer in a given read appears in the reference, and it helps to fast find the candidate positions that are likely to be within the read blocks highly similar to local reference sequences.</p>
<p>With the Unitig&#x02013;BWT index and the auxiliary bloom filter-based index, deSAMBA classifies a given read in the following four steps.</p></sec>
<sec>
<title>The Generation of Seed Blocks</title>
<p>deSAMBA initially partitions a given read into 100-bp-long segments. For each of the segments, deSAMBA extracts all its l-mers and separately tests each of the l-mer with the bloom filter-based index. The l-mers that passed the bloom filter are recorded, and deSAMBA finds the local region in the segment having the most consecutive passed l-mers, i.e., the longest passed l-mer chain, as a &#x0201C;seed block.&#x0201D; It is also worth noting that the l parameter is automatically configured in advance according to the size of the reference, and it is set as 15&#x02013;19 bp in most of the cases.</p>
<p>This idea derives from the characteristics of bloom filter. With a bloom filter-based index, there could be a proportion of false positives, i.e., a passed k-mer could be a false-positive &#x0201C;hit&#x0201D; to reference. However, the false-positive rate is relatively low, so that a seed block is more likely to be a true positive &#x0003E;l bp long exact match to the reference than false positives. Therefore, deSAMBA assumes that the seed blocks are long exact matches and uses them as candidates to generate SAMBs. Moreover, it is not problematic if a seed block is a false-positive match, since deSAMBA would retrieve the corresponding sequence of the seed block through the Unitig&#x02013;BWT index and filter it out if no such sequence can be found.</p></sec>
<sec>
<title>The Generation of Initial Sparse Approximate Match Blocks</title>
<p>For each of the seed blocks, deSAMBA extracts all the suffixes of the seed block and efficiently retrieves the maximal exact matches between the suffixes and the unitigs of the reference through the Unitig&#x02013;BWT index. All the retrieved matches are called U-MEMs, and deSAMBA separately checks each of them. If a U-MEM is fully covered by another one, deSAMBA would filter it out. After the filtration, the longest eight remaining U-MEMs are used to generate initial SAMBs.</p>
<p>For each of the U-MEMs, deSAMBA checks if the match is distanced from both of the two ends of the located unitig by at least 12 bp. If so, deSAMBA uses Landau&#x02013;Vishkin algorithm to compose an alignment between the read part and unitig to extend the U-MEM to a longer approximate match block. The extension is limited to the flanking 12 bp of the located unitig of the R-MEM, and it is expected that the alignment has a low edit distance, and a quality score is assigned to the generated approximate match block. The quality score is calculated based on all the matched and mismatched bases in the alignment with the following equations:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>S</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>c</mml:mi><mml:mi>h</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>b</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>c</mml:mi><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>c</mml:mi><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>c</mml:mi><mml:mi>h</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>b</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>f</mml:mi><mml:mtext>_</mml:mtext><mml:mi>C</mml:mi><mml:mi>o</mml:mi><mml:mi>m</mml:mi><mml:mi>p</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mi>x</mml:mi><mml:mtext>_</mml:mtext><mml:mi>P</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>t</mml:mi><mml:mi>y</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E3"><label>(2)</label><mml:math id="M3"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>c</mml:mi><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mn>10</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:mo class="qopname">lg</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>25</mml:mn></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mtext>E</mml:mtext></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E4"><label>(3)</label><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mn>10</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:mo class="qopname">lg</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>75</mml:mn></mml:mrow><mml:mrow><mml:mtext>E</mml:mtext></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E5"><label>(4)</label><mml:math id="M5"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>f</mml:mi><mml:mtext>_</mml:mtext><mml:mi>C</mml:mi><mml:mi>o</mml:mi><mml:mi>m</mml:mi><mml:mi>p</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mi>x</mml:mi><mml:mtext>_</mml:mtext><mml:mi>P</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>t</mml:mi><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mn>10</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:mo class="qopname">lg</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mtext>L</mml:mtext></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>f</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>c</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Herein, <italic>S</italic><sub><italic>match</italic></sub> and <italic>S</italic><sub><italic>mis</italic></sub> are the scores of matched and mismatched bases, <italic>S</italic><sub><italic>Ref</italic>_<italic>Complex</italic>_<italic>Penalty</italic></sub> is a reference size-based penalty that is related to the total length of the reference L<sub><italic>reference</italic></sub>, <italic>N</italic><sub><italic>match</italic></sub> and <italic>N</italic><sub><italic>mis</italic></sub> are, respectively, the numbers of matched bases and mismatched bases in the alignment, and E is a parameter representing the expected sequencing error rate (default value: 0.15). After the extension, deSAMBA maps the alignment block to all the copies of the unitig to produce a series of &#x0201C;generated approximate match blocks.&#x0201D;</p>
<p>If the distance between the U-MEM is within 12 bp of either end of its located unitig, deSAMBA produces approximate match blocks in a different way. That is, deSAMBA maps the U-MEM to all the copies of its matched unitig at first, i.e., the U-MEMs are converted to one or more MEMs to local reference sequences (each of them is called an &#x0201C;R-MEM&#x0201D;). And then, deSAMBA separately extends the R-MEMs by Landau&#x02013;Vishkin algorithm and scores the generated approximate match blocks in a similar approach.</p>
<p>After scoring, the generated approximate match blocks having &#x0003E;30 quality scores remained as &#x0201C;initial SAMBs,&#x0201D; and other ones are discarded. Moreover, each of the SAMBs can be written as a 4-tuple: <inline-formula><mml:math id="M6"><mml:mi>S</mml:mi><mml:mi>A</mml:mi><mml:mi>M</mml:mi><mml:msub><mml:mrow><mml:mi>B</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>, where <inline-formula><mml:math id="M7"><mml:msubsup><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> and <inline-formula><mml:math id="M8"><mml:msubsup><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> are the start and end positions on the read, and <inline-formula><mml:math id="M9"><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> and <inline-formula><mml:math id="M10"><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> are the start and end positions on the reference, respectively.</p></sec>
<sec>
<title>The Generation of Extended Sparse Approximate Match Blocks</title>
<p>deSAMBA greedily merges initial SAMBs from upstream to downstream. Two SAMBs are combined if their distance is &#x0003C;300 bp on the reference, and the difference between their distances on the read and on the reference is &#x0003C;30 bp. After this processing, the initial SAMBs are combined as a series of SAMB chains. Each SAMB chain can be written as a series of SAMBs:</p>
<disp-formula id="E6"><label>(5)</label><mml:math id="M11"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>S</mml:mi><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>S</mml:mi><mml:mi>A</mml:mi><mml:mi>M</mml:mi><mml:msub><mml:mrow><mml:mi>B</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x02026;</mml:mo><mml:mo stretchy="false">|</mml:mo><mml:mi>S</mml:mi><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where |<italic>SC</italic><sub><italic>i</italic></sub>| is the number of SAMBs of <italic>SC</italic><sub><italic>i</italic></sub>. Moreover, it can be derived that the read part and the local reference sequence covered by a SAMB chain <italic>SC</italic><sub><italic>i</italic></sub> are <inline-formula><mml:math id="M12"><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo stretchy="false">|</mml:mo><mml:mi>S</mml:mi><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M13"><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo stretchy="false">|</mml:mo><mml:mi>S</mml:mi><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula>, respectively.</p>
<p>deSAMBA further extends the SAMB chains by a SDP-based pseudo alignment approach. For a SAMB chain, <italic>SC</italic><sub><italic>i</italic></sub>, this is done in the following four sub-steps.</p>
<list list-type="order">
<list-item><p>deSAMBA extracts all the 9-mers within <inline-formula><mml:math id="M14"><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msubsup><mml:mo>-</mml:mo><mml:mi>L</mml:mi><mml:mi>R</mml:mi><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo stretchy="false">|</mml:mo><mml:mi>S</mml:mi><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:mi>L</mml:mi><mml:mi>R</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula> and indexes them with a hash-table-based data structure. The parameter LR (default value: 1,000) defines an extended local region in reference.</p></list-item>
<list-item><p>deSAMBA retrieves all the 9-bp matches between <inline-formula><mml:math id="M15"><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo stretchy="false">|</mml:mo><mml:mi>S</mml:mi><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M16"><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msubsup><mml:mo>-</mml:mo><mml:mi>L</mml:mi><mml:mi>R</mml:mi><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo stretchy="false">|</mml:mo><mml:mi>S</mml:mi><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:mi>L</mml:mi><mml:mi>R</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula> through the 9-mer hash table and combines all the consecutive 9-mer matches into one or more longer exact matches.</p></list-item>
<list-item><p>The remaining matches are chained in an SDP approach with the following function:
<disp-formula id="E61"><label>(6)</label><mml:math id="M17"><mml:mrow><mml:mi>f</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:msub><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>max</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:munder><mml:mrow><mml:mi>max</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mo>&#x0003E;</mml:mo><mml:mi>q</mml:mi><mml:mo>&#x02265;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:munder><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>f</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:msub><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mi>L</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:msub><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mi>&#x003B8;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mo>,</mml:mo><mml:mi>q</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mi>L</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:msub><mml:mi>t</mml:mi><mml:mi>p</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:math></disp-formula>
<disp-formula id="E7"><label>(7)</label><mml:math id="M18"><mml:mrow><mml:mi>&#x003B8;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mo>,</mml:mo><mml:mi>q</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:mtable columnalign='left'><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:mn>0.1</mml:mn><mml:mo>&#x000D7;</mml:mo></mml:mrow></mml:mtd><mml:mtd columnalign='left'><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:msubsup><mml:mi>t</mml:mi><mml:mi>p</mml:mi><mml:mi>R</mml:mi></mml:msubsup><mml:mo>&#x02212;</mml:mo><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:msubsup><mml:mi>t</mml:mi><mml:mi>q</mml:mi><mml:mi>R</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:mrow></mml:mrow></mml:mtd></mml:mtr><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mtext>&#x000A0;</mml:mtext></mml:mtd><mml:mtd columnalign='left'><mml:mrow><mml:mrow><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:msubsup><mml:mi>t</mml:mi><mml:mi>p</mml:mi><mml:mi>r</mml:mi></mml:msubsup><mml:mo>&#x02212;</mml:mo><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:msubsup><mml:mi>t</mml:mi><mml:mi>q</mml:mi><mml:mi>r</mml:mi></mml:msubsup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mtd><mml:mtd><mml:mrow><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mtext>&#x000A0;&#x000A0;</mml:mtext><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:msubsup><mml:mi>t</mml:mi><mml:mi>p</mml:mi><mml:mi>R</mml:mi></mml:msubsup><mml:mo>&#x02212;</mml:mo><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:msubsup><mml:mi>t</mml:mi><mml:mi>q</mml:mi><mml:mi>R</mml:mi></mml:msubsup><mml:mo>&#x0003C;</mml:mo><mml:mn>600</mml:mn></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>&#x0221E;</mml:mi></mml:mtd><mml:mtd><mml:mrow><mml:mi>o</mml:mi><mml:mi>t</mml:mi><mml:mi>h</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mi>w</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mrow></mml:mrow></mml:math></disp-formula></p></list-item>
<list-item><p>where Mat<sub><italic>p</italic></sub> and Mat<sub><italic>q</italic></sub> are the p-th and q-th matches (sorted by reference position) and they are not overlapped, <inline-formula><mml:math id="M19"><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:msubsup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>R</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> and <inline-formula><mml:math id="M20"><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:msubsup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow><mml:mrow><mml:mi>R</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> are their positions on the reference, <inline-formula><mml:math id="M21"><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:msubsup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> and <inline-formula><mml:math id="M22"><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:msubsup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> are their positions on the read, respectively; <italic>f</italic> (Mat<sub><italic>p</italic></sub>) is the scoring function for the Mat<sub><italic>p</italic></sub>, <italic>L</italic> (Mat<sub><italic>p</italic></sub>) is the length of Mat<sub><italic>p</italic></sub>, and &#x003B8; (<italic>p, q</italic>) is a penalty for the two linked matches, Mat<sub><italic>p</italic></sub> and Mat<sub><italic>q</italic></sub>.</p></list-item>
<list-item><p>After the SDP, the optimal chain of matches is obtained through backtracking, and it recorded as the &#x0201C;extended SAMB&#x0201D; generated based on <italic>SC</italic><sub><italic>i</italic></sub>.</p></list-item>
</list></sec>
<sec>
<title>The Classification of the Reads</title>
<p>deSAMBA collects all the generated extended SAMBs and sorts them by their scores calculated in the SDP process. deSAMBA then determines the primary classification of the reads by the taxonomy entity of the reference genome corresponding to the extended SAMB with highest score. The taxonomy entities of other extended SAMBs are as secondary classifications. Moreover, the SAMBs are also output as the partial pseudo alignments of the read.</p></sec>
<sec>
<title>Implementation of Benchmark</title>
<p>All the benchmarks were carried out on a server with four Intel E7-4820 CPUs (32 cores) and 1 TB RAM running Ubuntu Linux OS. All the benchmarked classification tools were run in eight CPU threads. Some detailed information about employed reference sequences, the real sequencing datasets, and the command lines used for read classification is as follows.</p>
<p>We downloaded all reference sequences from NCBI RefSeq database. A genome sequence from RefSeq database was employed only if it is marked as &#x0201C;complete genome.&#x0201D; There are totally 8,621 bacterial, 251 archaea, and 7,412 viral genomes being used. The RefSeq ID and Taxonomy ID are described in &#x0201C;reference describe.txt.&#x0201D; For kaiju, the reference index was built using NCBI protein database due to its specifically designed read classification approach. We downloaded the 145 real sequencing pseudo-metagenomic datasets from NCBI Sequence Read Archive (SRA). The datasets are from various bacterial, viral, or archaeal genomes. It is also worth noting that for all the datasets, only the reads longer than 1,000 bp were used for the benchmark.</p></sec></sec>
<sec sec-type="data-availability-statement" id="s5">
<title>Data Availability Statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="SM1">Supplementary Material</xref>, further inquiries can be directed to the corresponding author/s.</p></sec>
<sec id="s6">
<title>Author Contributions</title>
<p>GL wrote the paper and did the experiments. YL and BL provided ideas of this work. YW and DL provided important suggestions of performance improvements. JL and DL revised this manuscript and guided how to do experiments. YH and YW supervised this work. All authors contributed to the article and approved the submitted version.</p></sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p></sec>
</body>
<back>
<sec sec-type="supplementary-material" id="s7">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fcell.2021.643645/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fcell.2021.643645/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/></sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Altschul</surname> <given-names>S. F.</given-names></name> <name><surname>Gish</surname> <given-names>W.</given-names></name> <name><surname>Miller</surname> <given-names>W.</given-names></name> <name><surname>Myers</surname> <given-names>E. W.</given-names></name> <name><surname>Lipman</surname> <given-names>D. J.</given-names></name></person-group> (<year>1990</year>). <article-title>Basic local alignment search tool</article-title>. <source>J. Mol. Biol</source>. <volume>215</volume>, <fpage>403</fpage>&#x02013;<lpage>410</lpage>. <pub-id pub-id-type="doi">10.1016/S0022-2836(05)80360-2</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bertrand</surname> <given-names>D.</given-names></name> <name><surname>Shaw</surname> <given-names>J.</given-names></name> <name><surname>Kalathiyappan</surname> <given-names>M.</given-names></name> <name><surname>Ng</surname> <given-names>A. H. Q.</given-names></name> <name><surname>Kumar</surname> <given-names>M. S.</given-names></name> <name><surname>Li</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Hybrid metagenomic assembly enables high-resolution analysis of resistance determinants and mobile elements in human microbiomes</article-title>. <source>Nat. Biotechnol</source>. <volume>37</volume>, <fpage>937</fpage>&#x02013;<lpage>944</lpage>. <pub-id pub-id-type="doi">10.1038/s41587-019-0191-2</pub-id><pub-id pub-id-type="pmid">31359005</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Boratyn</surname> <given-names>G. M.</given-names></name> <name><surname>Camacho</surname> <given-names>C.</given-names></name> <name><surname>Cooper</surname> <given-names>P. S.</given-names></name> <name><surname>Coulouris</surname> <given-names>G.</given-names></name> <name><surname>Fong</surname> <given-names>A.</given-names></name> <name><surname>Ma</surname> <given-names>N.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>BLAST: a more efficient report with usability improvements</article-title>. <source>Nucleic Acids Res</source>. <volume>41</volume>, <fpage>W29</fpage>&#x02013;<lpage>W33</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkt282</pub-id><pub-id pub-id-type="pmid">23609542</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chaisson</surname> <given-names>M. J.</given-names></name> <name><surname>Tesler</surname> <given-names>G.</given-names></name></person-group> (<year>2012</year>). <article-title>Mapping single molecule sequencing reads using basic local alignment with successive refinement (BLASR): application and theory</article-title>. <source>BMC Bioinformatics</source> <volume>13</volume>:<fpage>238</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-13-238</pub-id><pub-id pub-id-type="pmid">22988817</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cheng</surname> <given-names>L.</given-names></name></person-group> (<year>2019</year>). <article-title>Computational and biological methods for gene therapy</article-title>. <source>Curr. Gene Ther</source>. <volume>19</volume>:<fpage>210</fpage>. <pub-id pub-id-type="doi">10.2174/156652321904191022113307</pub-id><pub-id pub-id-type="pmid">31762421</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cheng</surname> <given-names>L.</given-names></name> <name><surname>Qi</surname> <given-names>C.</given-names></name> <name><surname>Zhuang</surname> <given-names>H.</given-names></name> <name><surname>Fu</surname> <given-names>T.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name></person-group> (<year>2020</year>). <article-title>gutMDisorder: a comprehensive database for dysbiosis of the gut microbiota in disorders and interventions</article-title>. <source>Nucleic Acids Res</source>. <volume>48</volume>, <fpage>D554</fpage>&#x02013;<lpage>D560</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkz843</pub-id><pub-id pub-id-type="pmid">32515792</pub-id></citation></ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dilthey</surname> <given-names>A. T.</given-names></name> <name><surname>Jain</surname> <given-names>C.</given-names></name> <name><surname>Koren</surname> <given-names>S.</given-names></name> <name><surname>Phillippy</surname> <given-names>A. M.</given-names></name></person-group> (<year>2019</year>). <article-title>Strain-level metagenomic assignment and compositional estimation for long reads with MetaMaps</article-title>. <source>Nat. Commun</source>. <volume>10</volume>:<fpage>3066</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-019-10934-2</pub-id><pub-id pub-id-type="pmid">31296857</pub-id></citation></ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gilbert</surname> <given-names>J. A.</given-names></name> <name><surname>Jansson</surname> <given-names>J. K.</given-names></name> <name><surname>Knight</surname> <given-names>R.</given-names></name></person-group> (<year>2014</year>). <article-title>The earth microbiome project: successes and aspirations</article-title>. <source>BMC Biol</source>. <volume>12</volume>:<fpage>69</fpage>. <pub-id pub-id-type="doi">10.1186/s12915-014-0069-1</pub-id><pub-id pub-id-type="pmid">25184604</pub-id></citation></ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Govindaraj</surname> <given-names>R. G.</given-names></name> <name><surname>Subramaniyam</surname> <given-names>S.</given-names></name> <name><surname>Manavalan</surname> <given-names>B.</given-names></name></person-group> (<year>2020</year>). <article-title>Extremely-randomized-tree-based Prediction of N6-Methyladenosine Sites in Saccharomyces cerevisiae</article-title>. <source>Curr. Genomics</source> <volume>21</volume>, <fpage>26</fpage>&#x02013;<lpage>33</lpage>. <pub-id pub-id-type="doi">10.2174/1389202921666200219125625</pub-id><pub-id pub-id-type="pmid">32655295</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Guan</surname> <given-names>D.</given-names></name> <name><surname>Liu</surname> <given-names>B.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name></person-group> (<year>2018</year>). <article-title>deSPI: efficient classification of metagenomics reads with lightweight de Bruijn graph-based reference indexing</article-title>, in <source>IEEE International Conference on Bioinformatics and Biomedicine (BIBM) (IEEE)</source>, (<publisher-loc>Madrid</publisher-loc>), <fpage>265</fpage>&#x02013;<lpage>269</lpage>. <pub-id pub-id-type="doi">10.1109/BIBM.2018.8621235</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hasan</surname> <given-names>M. M.</given-names></name> <name><surname>Basith</surname> <given-names>S.</given-names></name> <name><surname>Khatun</surname> <given-names>M. S.</given-names></name> <name><surname>Lee</surname> <given-names>G.</given-names></name> <name><surname>Manavalan</surname> <given-names>B.</given-names></name> <name><surname>Kurata</surname> <given-names>H.</given-names></name></person-group> (<year>2020b</year>). <article-title>Meta-i6mA: an interspecies predictor for identifying DNA N6-methyladenine sites of plant genomes by exploiting informative features in an integrative machine-learning framework</article-title>. <source>Brief. Bioinformatics.</source> bbaa202. <pub-id pub-id-type="doi">10.1093/bib/bbaa202</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hasan</surname> <given-names>M. M.</given-names></name> <name><surname>Manavalan</surname> <given-names>B.</given-names></name> <name><surname>Khatun</surname> <given-names>M. S.</given-names></name> <name><surname>Kurata</surname> <given-names>H.</given-names></name></person-group> (<year>2020a</year>). <article-title>i4mC-ROSE, a bioinformatics tool for the identification of DNA N4-methylcytosine sites in the rosaceae genome</article-title>. <source>Int. J. Biol. Macromol</source>. <volume>157</volume>, <fpage>752</fpage>&#x02013;<lpage>758</lpage>. <pub-id pub-id-type="doi">10.1016/j.ijbiomac.2019.12.009</pub-id><pub-id pub-id-type="pmid">31805335</pub-id></citation></ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>Y.</given-names></name> <name><surname>Sun</surname> <given-names>J.-y.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Gao</surname> <given-names>S.</given-names></name> <name><surname>Wang</surname> <given-names>T.</given-names></name> <name><surname>Han</surname> <given-names>Z.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>rs1990622 variant associates with Alzheimer&#x00027;s disease and regulates TMEM106B expression in human brain tissues</article-title>. <source>BMC Med</source>. <volume>19</volume>:<fpage>11</fpage>. <pub-id pub-id-type="doi">10.1186/s12916-020-01883-5</pub-id><pub-id pub-id-type="pmid">33461566</pub-id></citation></ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Liu</surname> <given-names>B.</given-names></name> <name><surname>Gao</surname> <given-names>S.</given-names></name> <name><surname>Wang</surname> <given-names>T.</given-names></name> <name><surname>Han</surname> <given-names>Z.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>rs34331204 regulates TSPAN13 expression and contributes to Alzheimer&#x00027;s disease with sex differences</article-title>. <source>Brain</source> <volume>143</volume>:<fpage>e95</fpage>. <pub-id pub-id-type="doi">10.1093/brain/awaa302</pub-id><pub-id pub-id-type="pmid">33175954</pub-id></citation></ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>Y.</given-names></name> <name><surname>Zhao</surname> <given-names>T.</given-names></name> <name><surname>Zhang</surname> <given-names>N.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Cheng</surname> <given-names>L.</given-names></name></person-group> (<year>2019</year>). <article-title>A review of recent advances and research on drug target identification methods</article-title>. <source>Curr. Drug Metab</source>. <volume>20</volume>, <fpage>209</fpage>&#x02013;<lpage>216</lpage>. <pub-id pub-id-type="doi">10.2174/1389200219666180925091851</pub-id><pub-id pub-id-type="pmid">30251599</pub-id></citation></ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huson</surname> <given-names>D. H.</given-names></name> <name><surname>Auch</surname> <given-names>A. F.</given-names></name> <name><surname>Qi</surname> <given-names>J.</given-names></name> <name><surname>Schuster</surname> <given-names>S. C.</given-names></name></person-group> (<year>2007</year>). <article-title>MEGAN analysis of metagenomic data</article-title>. <source>Genome Res</source>. <volume>17</volume>, <fpage>377</fpage>&#x02013;<lpage>386</lpage>. <pub-id pub-id-type="doi">10.1101/gr.5969107</pub-id></citation></ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname> <given-names>D.</given-names></name> <name><surname>Song</surname> <given-names>L.</given-names></name> <name><surname>Breitwieser</surname> <given-names>F. P.</given-names></name> <name><surname>Salzberg</surname> <given-names>S. L.</given-names></name></person-group> (<year>2016</year>). <article-title>Centrifuge: rapid and sensitive classification of metagenomic sequences</article-title>. <source>Genome Res</source>. <volume>26</volume>, <fpage>1721</fpage>&#x02013;<lpage>1729</lpage>. <pub-id pub-id-type="doi">10.1101/gr.210641.116</pub-id><pub-id pub-id-type="pmid">27852649</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>H.</given-names></name></person-group> (<year>2013</year>). <article-title>Aligning sequence reads, clone sequences and assembly contigs with BWA-MEM</article-title>. <source>arXiv[Preprint].arXiv:13033997</source>.</citation></ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>H.</given-names></name></person-group> (<year>2018</year>). <article-title>Minimap2: pairwise alignment for nucleotide sequences</article-title>. <source>Bioinformatics</source> <volume>34</volume>, <fpage>3094</fpage>&#x02013;<lpage>3100</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty191</pub-id><pub-id pub-id-type="pmid">29750242</pub-id></citation></ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>B.</given-names></name> <name><surname>Gao</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name></person-group> (<year>2017</year>). <article-title>LAMSA: fast split read alignment with long approximate matches</article-title>. <source>Bioinformatics</source> <volume>33</volume>, <fpage>192</fpage>&#x02013;<lpage>201</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btw594</pub-id><pub-id pub-id-type="pmid">27667793</pub-id></citation></ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>B.</given-names></name> <name><surname>Guo</surname> <given-names>H.</given-names></name> <name><surname>Brudno</surname> <given-names>M.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name></person-group> (<year>2016</year>). <article-title>deBGA: read alignment with de Bruijn graph-based seed and extension</article-title>. <source>Bioinformatics</source> <volume>32</volume>, <fpage>3224</fpage>&#x02013;<lpage>3232</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btw371</pub-id><pub-id pub-id-type="pmid">27378303</pub-id></citation></ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Yu</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name> <name><surname>Elliott</surname> <given-names>C. F.</given-names></name> <name><surname>Qian</surname> <given-names>C.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name></person-group> (<year>2018</year>). <article-title>A novel data structure to support ultra-fast taxonomic classification of metagenomic sequences with k-mer signatures</article-title>. <source>Bioinformatics</source> <volume>34</volume>, <fpage>171</fpage>&#x02013;<lpage>178</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btx432</pub-id><pub-id pub-id-type="pmid">29036588</pub-id></citation></ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Menzel</surname> <given-names>P.</given-names></name> <name><surname>Ng</surname> <given-names>K. L.</given-names></name> <name><surname>Krogh</surname> <given-names>A.</given-names></name></person-group> (<year>2016</year>). <article-title>Fast and sensitive taxonomic classification for metagenomics with Kaiju</article-title>. <source>Nat. Commun</source>. <volume>7</volume>:<fpage>11257</fpage>. <pub-id pub-id-type="doi">10.1038/ncomms11257</pub-id><pub-id pub-id-type="pmid">27071849</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Meth&#x000E9;</surname> <given-names>B. A.</given-names></name> <name><surname>Nelson</surname> <given-names>K. E.</given-names></name> <name><surname>Pop</surname> <given-names>M.</given-names></name> <name><surname>Creasy</surname> <given-names>H. H.</given-names></name> <name><surname>Giglio</surname> <given-names>M. G.</given-names></name> <name><surname>Huttenhower</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>A framework for human microbiome research</article-title>. <source>Nature</source> <volume>486</volume>, <fpage>215</fpage>&#x02013;<lpage>221</lpage>. <pub-id pub-id-type="doi">10.1038/nature11209</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pruitt</surname> <given-names>K. D.</given-names></name> <name><surname>Brown</surname> <given-names>G. R.</given-names></name> <name><surname>Hiatt</surname> <given-names>S. M.</given-names></name> <name><surname>Thibaud-Nissen</surname> <given-names>F.</given-names></name> <name><surname>Astashyn</surname> <given-names>A.</given-names></name> <name><surname>Ermolaeva</surname> <given-names>O.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>RefSeq: an update on mammalian reference sequences</article-title>. <source>Nucleic Acids Res</source>. <volume>42</volume>, <fpage>D756</fpage>&#x02013;<lpage>D763</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkt1114</pub-id><pub-id pub-id-type="pmid">24259432</pub-id></citation></ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Quick</surname> <given-names>J.</given-names></name> <name><surname>Loman</surname> <given-names>N. J.</given-names></name> <name><surname>Duraffour</surname> <given-names>S.</given-names></name> <name><surname>Simpson</surname> <given-names>J. T.</given-names></name> <name><surname>Severi</surname> <given-names>E.</given-names></name> <name><surname>Cowley</surname> <given-names>L.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>Real-time, portable genome sequencing for Ebola surveillance</article-title>. <source>Nature</source> <volume>530</volume>, <fpage>228</fpage>&#x02013;<lpage>232</lpage>. <pub-id pub-id-type="doi">10.1038/nature16996</pub-id><pub-id pub-id-type="pmid">26840485</pub-id></citation></ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sedlazeck</surname> <given-names>F. J.</given-names></name> <name><surname>Rescheneder</surname> <given-names>P.</given-names></name> <name><surname>Smolka</surname> <given-names>M.</given-names></name> <name><surname>Fang</surname> <given-names>H.</given-names></name> <name><surname>Nattestad</surname> <given-names>M.</given-names></name> <name><surname>Von Haeseler</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Accurate detection of complex structural variations using single-molecule sequencing</article-title>. <source>Nat. Methods</source> <volume>15</volume>, <fpage>461</fpage>&#x02013;<lpage>468</lpage>. <pub-id pub-id-type="doi">10.1038/s41592-018-0001-7</pub-id><pub-id pub-id-type="pmid">29713083</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wood</surname> <given-names>D. E.</given-names></name> <name><surname>Lu</surname> <given-names>J.</given-names></name> <name><surname>Langmead</surname> <given-names>B.</given-names></name></person-group> (<year>2019</year>). <article-title>Improved metagenomic analysis with Kraken 2</article-title>. <source>Genome Biol</source>. <volume>20</volume>:<fpage>257</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-019-1891-0</pub-id><pub-id pub-id-type="pmid">31779668</pub-id></citation></ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wood</surname> <given-names>D. E.</given-names></name> <name><surname>Salzberg</surname> <given-names>S. L.</given-names></name></person-group> (<year>2014</year>). <article-title>Kraken: ultrafast metagenomic sequence classification using exact alignments</article-title>. <source>Genome Biol</source>. <volume>15</volume>:<fpage>R46</fpage>. <pub-id pub-id-type="doi">10.1186/gb-2014-15-3-r46</pub-id><pub-id pub-id-type="pmid">26812576</pub-id></citation></ref>
</ref-list>
<fn-group>
<fn fn-type="financial-disclosure"><p><bold>Funding.</bold> This work was supported by the National Key R&#x00026;D Program of China (2017YFC1201201, 2018YFC0910504, and 2017YFC0907503), the Natural Science Foundation of China (61801147 and 82003553), and Heilongjiang Postdoctoral Science Foundation (LBH-Z6064).</p>
</fn>
</fn-group>
</back>
</article>