<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">739054</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2021.739054</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>SeekFusion - A Clinically Validated Fusion Transcript Detection Pipeline for PCR-Based Next-Generation Sequencing of RNA</article-title>
<alt-title alt-title-type="left-running-head">Balan et&#x20;al.</alt-title>
<alt-title alt-title-type="right-running-head">SeekFusion Fusion Transcript Detection Pipeline</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Balan</surname>
<given-names>Jagadheshwar</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1379216/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Jenkinson</surname>
<given-names>Garrett</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/889464/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Nair</surname>
<given-names>Asha</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/277240/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Saha</surname>
<given-names>Neiladri</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1406541/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Koganti</surname>
<given-names>Tejaswi</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Voss</surname>
<given-names>Jesse</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zysk</surname>
<given-names>Christopher</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Barr Fritcher</surname>
<given-names>Emily G.</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ross</surname>
<given-names>Christian A.</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Giannini</surname>
<given-names>Caterina</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Raghunathan</surname>
<given-names>Aditya</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Kipp</surname>
<given-names>Benjamin R.</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Jenkins</surname>
<given-names>Robert</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ida</surname>
<given-names>Cris</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Halling</surname>
<given-names>Kevin C.</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1402721/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Blackburn</surname>
<given-names>Patrick R.</given-names>
</name>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Dasari</surname>
<given-names>Surendra</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/311634/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Oliver</surname>
<given-names>Gavin R.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/578052/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Klee</surname>
<given-names>Eric W.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/135135/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<label>
<sup>1</sup>
</label>Quantitative Health Sciences, Mayo Clinic, <addr-line>Rochester</addr-line>, <addr-line>MN</addr-line>, <country>United&#x20;States</country>
</aff>
<aff id="aff2">
<label>
<sup>2</sup>
</label>Division of Laboratory Genetics and Genomics, Mayo Clinic, <addr-line>Rochester</addr-line>, <addr-line>MN</addr-line>, <country>United&#x20;States</country>
</aff>
<aff id="aff3">
<label>
<sup>3</sup>
</label>Applied Genomics Division, Perkin Elmer, <addr-line>Waltham</addr-line>, <addr-line>MA</addr-line>, <country>United&#x20;States</country>
</aff>
<aff id="aff4">
<label>
<sup>4</sup>
</label>Division of Anatomic Pathology, Mayo Clinic, <addr-line>Rochester</addr-line>, <addr-line>MN</addr-line>, <country>United&#x20;States</country>
</aff>
<aff id="aff5">
<label>
<sup>5</sup>
</label>Information Technology, Mayo Clinic, <addr-line>Rochester</addr-line>, <addr-line>MN</addr-line>, <country>United&#x20;States</country>
</aff>
<aff id="aff6">
<label>
<sup>6</sup>
</label>Department of Pathology, St. Jude Children&#x2019;s Research Hospital, <addr-line>Memphis</addr-line>, <addr-line>TN</addr-line>, <country>United&#x20;States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/529645/overview">Andrew J.&#x20;Mungall</ext-link>, Canada&#x2019;s Michael Smith Genome Sciences Centre, Canada</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/55577/overview">Prashanth N. Suravajhala</ext-link>, Amrita Vishwa Vidyapeetham University, India</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1326293/overview">Ka Ming Nip</ext-link>, Canada&#x2019;s Michael Smith Genome Sciences Centre, Canada</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Eric W. Klee, <email>Klee.Eric@mayo.edu</email>
</corresp>
<fn fn-type="other">
<p>This article was submitted to Genomic Assay Technology, a section of the journal Frontiers in Genetics</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>22</day>
<month>10</month>
<year>2021</year>
</pub-date>
<pub-date pub-type="collection">
<year>2021</year>
</pub-date>
<volume>12</volume>
<elocation-id>739054</elocation-id>
<history>
<date date-type="received">
<day>21</day>
<month>07</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>24</day>
<month>09</month>
<year>2021</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2021 Balan, Jenkinson, Nair, Saha, Koganti, Voss, Zysk, Barr Fritcher, Ross, Giannini, Raghunathan, Kipp, Jenkins, Ida, Halling, Blackburn, Dasari, Oliver and Klee.</copyright-statement>
<copyright-year>2021</copyright-year>
<copyright-holder>Balan, Jenkinson, Nair, Saha, Koganti, Voss, Zysk, Barr Fritcher, Ross, Giannini, Raghunathan, Kipp, Jenkins, Ida, Halling, Blackburn, Dasari, Oliver and Klee</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these&#x20;terms.</p>
</license>
</permissions>
<abstract>
<p>Detecting gene fusions involving driver oncogenes is pivotal in clinical diagnosis and treatment of cancer patients. Recent developments in next-generation sequencing (NGS) technologies have enabled improved assays for bioinformatics-based gene fusions detection. In clinical applications, where a small number of fusions are clinically actionable, targeted polymerase chain reaction (PCR)-based NGS chemistries, such as the QIAseq RNAscan assay, aim to improve accuracy compared to standard RNA sequencing. Existing informatics methods for gene fusion detection in NGS-based RNA sequencing assays traditionally use a transcriptome-based spliced alignment approach or a <italic>de-novo</italic> assembly approach. Transcriptome-based spliced alignment methods face challenges with short read mapping yielding low quality alignments. <italic>De-novo</italic> assembly-based methods yield longer contigs from short reads that can be more sensitive for genomic rearrangements, but face performance and scalability challenges. Consequently, there exists a need for a method to efficiently and accurately detect fusions in targeted PCR-based NGS chemistries. We describe SeekFusion, a highly accurate and computationally efficient pipeline enabling identification of gene fusions from PCR-based NGS chemistries. Utilizing biological samples processed with the QIAseq RNAscan assay and in-silico simulated data we demonstrate that SeekFusion gene fusion detection accuracy outperforms popular existing methods such as STAR-Fusion, TOPHAT-Fusion and JAFFA-hybrid. We also present results from 4,484 patient samples tested for neurological tumors and sarcoma, encompassing details on some novel fusions identified.</p>
</abstract>
<kwd-group>
<kwd>gene fusion</kwd>
<kwd>RNA</kwd>
<kwd>bioinformatics</kwd>
<kwd>UMI consensus</kwd>
<kwd>QiaSeq</kwd>
<kwd>sarcoma</kwd>
<kwd>neuro-oncolgy</kwd>
<kwd>neuro-oncological disease</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>Gene fusions are potentially pathogenic events that result from genomic structural rearrangements including inversions, translocations, and interstitial deletions. Gene fusions are frequently observed in cancer, and while their prevalence varies by tumor type they have been estimated to account for 20% of all cancer morbidity (<xref ref-type="bibr" rid="B27">Mitelman et&#x20;al., 2007</xref>). Identification of gene fusions in a tumor can help guide therapeutic decision-making since fused protein products can represent targets of small molecule inhibitors or other novel treatments. Examples include the treatment of recurrent PTPRZ1-MET fusion-positive glioblastoma using the MET kinase inhibitor crizotinib, KIAA1549-BRAF fusion driven pediatric pilocytic astrocytoma treatment using the MEK inhibitor trametinib, and ALK-EML4 fusion-positive lung cancer using tyrosine kinase inhibitor lorlatinib (<xref ref-type="bibr" rid="B3">Bender et&#x20;al., 2016</xref>; <xref ref-type="bibr" rid="B18">Jain et&#x20;al., 2017</xref>; <xref ref-type="bibr" rid="B33">Shaw et&#x20;al., 2019</xref>).</p>
<p>RNA sequencing (RNA-Seq) assays for gene fusion detection provide improvements in throughput, sensitivity and specificity over traditional DNA and protein-based approaches like fluorescence <italic>in situ</italic> hybridization (FISH) and immunohistochemistry (IHC) (<xref ref-type="bibr" rid="B39">Wang et&#x20;al., 2009</xref>; <xref ref-type="bibr" rid="B1">Abel et&#x20;al., 2014</xref>; <xref ref-type="bibr" rid="B28">Moskalev et&#x20;al., 2014</xref>; <xref ref-type="bibr" rid="B30">Pekar-Zlotin et&#x20;al., 2015</xref>; <xref ref-type="bibr" rid="B16">Haynes et&#x20;al., 2019</xref>). RNA-based strategies to detect gene fusions use transcriptome-wide sequencing with ribosomal RNA-depleted, fractionated messenger RNA or target specific genes of interest using polymerase chain reaction (PCR) based amplicon RNA-Seq, or bait hybridization (capture and ligation) using assays such as QIAGEN&#x2019;s QIAseq RNAscan panel or Illumina&#x2019;s SureSelect RNA capture (<xref ref-type="bibr" rid="B39">Wang et&#x20;al., 2009</xref>; <xref ref-type="bibr" rid="B5">Blomquist et&#x20;al., 2013</xref>; <xref ref-type="bibr" rid="B12">Drilon et&#x20;al., 2015</xref>). The PCR-based amplicon RNA-Seq methods offer inexpensive and highly accurate sequencing of fusion transcripts, simultaneously assessing dozens to thousands of targets (<xref ref-type="bibr" rid="B5">Blomquist et&#x20;al., 2013</xref>) and enabling detection of lowly expressed fusions at a high sequencing depth. While PCR-based approaches have traditionally been limited due to stochastic errors that propagate to all PCR cycles (<xref ref-type="bibr" rid="B35">Thilly, 1993</xref>), the addition of unique molecular indexes (UMI) in ligation adapters has recently been used to alleviate the impact of such errors (<xref ref-type="bibr" rid="B22">Kivioja et&#x20;al., 2012</xref>), and can improve the sensitivity of gene fusion detection assays. QIAseq NGS assay panels have been demonstrated to provide robust correlation with RT-PCR and low PCR-bias (<xref ref-type="bibr" rid="B41">Wong et&#x20;al., 2019</xref>), and so was the chemistry used in this study (see <italic>Methods</italic>, <italic>Library Preparation and Sequencing</italic> section).</p>
<p>Several bioinformatics tools are available for detecting gene fusions that use wide variety of approaches. Brian et&#x20;al. and Trung et&#x20;al. have each outlined benchmarking of 15 gene fusion identification tools in their studies and detailed the methods and performance (<xref ref-type="bibr" rid="B38">Vu et&#x20;al., 2018</xref>; <xref ref-type="bibr" rid="B15">Haas et&#x20;al., 2019</xref>). One common approach uses transcriptome-based spliced alignment. This method is reliant on the accurate mapping of short reads to the transcriptome, which can be challenging due to genome repetitiveness, sequence homology, incomplete transcriptome annotations, and novel patient sequences not well-represented by the reference genome. Matteo et&#x20;al. elaborate on limitations of several gene fusion identification tools using short read alignment technologies and report key factors such as read length, quality scores and number of reads supporting each fusion call (<xref ref-type="bibr" rid="B6">Carrara et&#x20;al., 2013a</xref>). <italic>De-novo</italic> assembly-based approaches yield longer contigs from short reads and address some of the limitations of short-read alignment but are computationally intensive and often not scalable to large scale clinical testing (<xref ref-type="bibr" rid="B7">Carrara et&#x20;al., 2013b</xref>; <xref ref-type="bibr" rid="B10">Davidson et&#x20;al., 2015</xref>; <xref ref-type="bibr" rid="B15">Haas et&#x20;al., 2019</xref>). Hybrid approaches such as the method implemented in JAFFA (<xref ref-type="bibr" rid="B10">Davidson et&#x20;al., 2015</xref>) combine the approaches described above, but so far, no hybrid method achieves a balance of scalability and accuracy.</p>
<p>Herein we describe SeekFusion (<ext-link ext-link-type="uri" xlink:href="https://hub.docker.com/repository/docker/jagadhesh89/seekfusion">https://hub.docker.com/repository/docker/jagadhesh89/seekfusion</ext-link>), a time-efficient pipeline that leverages <italic>de-novo</italic> assembly and alignment based approaches to accurately identify gene fusions utilizing PCR-UMI-based amplicon RNA-Seq. SeekFusion performs rapid alignment to gene sequences, then groups and filters aligned reads for <italic>de-novo</italic> assembly. Assembled contigs are realigned to a reference genome and fusion genes are identified, annotated and reported in a VCF format. Using verified clinical cases and synthetic controls, we demonstrate that SeekFusion outperforms existing pipelines by balancing high analytical sensitivity and specificity with computational efficiency. The algorithm is written using Python and bash and the functions are wrapped in WDL and processed using Cromwell (<xref ref-type="bibr" rid="B37">Voss, 2017</xref>) for ease of deployment by end-users in their own compute environments.</p>
</sec>
<sec sec-type="methods" id="s2">
<title>Methods</title>
<sec id="s2-1">
<title>Methods Overview</title>
<p>The study (<xref ref-type="fig" rid="F1">Figure&#x20;1</xref>) involves developing a bioinformatics method for neurological cancer and sarcoma cancer clinical NGS assays aimed at targeting gene fusions. Towards development of the assay, few positive samples, negative samples and in-silico samples were selected, sequenced and analyzed across multiple fusion callers. The fusion callers were benchmarked carefully, results were summarized and compared to orthogonal assay results to identify the winning method. SeekFusion, an internally developed pipeline is highly optimized and accurate compared to the other methods that included STAR-Fusion, JAFFA and Tophat-Fusion. SeekFusion was then used towards verification of the developed assay, the assays were then implemented clinically for gene fusion identification post New York State (NYS) approval for the laboratory developed&#x20;tests.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Methods overview. The study involves developing Neuorlogical oncology and Sarcoma clinical NGS assays aimed at targeting gene fusions. Towards development of the assay, few positive samples, negative samples and in-silico samples were selected, sequenced and analyzed across multiple fusion callers. The fusion calls were examined to identify the winning method, and the winning method was SeekFusion, which was internally developed and optimized for the assay. The method was then used towards verification of the assay for NYS approvals and was deployed clinically for gene fusion identification.</p>
</caption>
<graphic xlink:href="fgene-12-739054-g001.tif"/>
</fig>
</sec>
<sec id="s2-2">
<title>Panel Design for Gene Fusion Detection in Neurological Cancers and Sarcomas</title>
<p>A neurological oncology panel was designed to target fusions in 80 genes (<xref ref-type="sec" rid="s11">Supplementary Table S2</xref>) utilizing the Qiagen QIAseq RNAscan Custom Panel (<xref ref-type="bibr" rid="B4">Blessing et&#x20;al., 2019</xref>). Targeted rearrangements were selected based on association with a variety of adult and pediatric central nervous system (CNS) tumors and potential utility in the differential diagnosis of these tumors or in the differentiation of molecularly defined tumor subtypes (e.g., ependymoma RELA fusion-positive) (<xref ref-type="bibr" rid="B21">Kim et&#x20;al., 2017</xref>). A sarcoma assay was designed to diagnose specific soft tissue and bone tumors (sarcoma) based on the observed gene fusion in 138 genes (<xref ref-type="sec" rid="s11">Supplementary Table S3</xref>). Targeted rearrangements were selected based on associations with a variety of sarcoma types such as rhabdomyosarcoma, synovial sarcoma, and Ewing&#x2019;s sarcoma. Both neurological oncology and sarcoma assays&#x2019; chemistry is designed to target a list of genes known to be involved in rearrangements using gene specific primers but uses a universal primer on the partner gene, making it a partner-agnostic chemistry capable of novel gene fusion identification. Specimen requirements and types for the assays have been detailed in <xref ref-type="sec" rid="s11">Supplementary Methods</xref> section.</p>
</sec>
<sec id="s2-3">
<title>Sample Selection for Benchmarking</title>
<p>Twenty-one neurological tumor samples were obtained following IRB-approved protocols. Of the 21 samples, 12 were fusion positive and 9 were fusion negative (<xref ref-type="table" rid="T1">Table&#x20;1</xref>). In addition to these clinical samples of known gene fusion status, we included a positive control, created from fusion-negative samples to which 13 gene fusion oligonucleotides were spiked in. Finally, a normal control sample was sequenced, and 27 unique fusions were added in-silico as positive controls. In total, benchmarking was performed using 52 known positive gene fusions. All samples were processed using the 80-gene QIAseq RNAscan neurological oncology NGS&#x20;panel.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>List of samples for benchmarking.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Sample ID</th>
<th align="center">Tumor type</th>
<th align="center">Expected fusion</th>
<th align="center">Fusion positive/negative</th>
<th align="center">Confirmed method</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">pos_S1</td>
<td align="left">Pilocytic astrocytoma (WHO Grade I)</td>
<td align="left">
<italic>KIAA1549-BRAF</italic>
</td>
<td align="left">positive</td>
<td align="left">RT-PCR with Sanger sequencing</td>
</tr>
<tr>
<td align="left">pos_S2</td>
<td align="left">High grade astrocytoma consistent with glioblastoma small cell type, IDH-wildtype</td>
<td align="left">
<italic>EGFR-SEPT14</italic>
</td>
<td align="left">positive</td>
<td align="left">RT-PCR with Sanger sequencing</td>
</tr>
<tr>
<td align="left">pos_S3</td>
<td align="left">Granular cell astrocytoma</td>
<td align="left">
<italic>EGFR-EGFR (EGFRvIII)</italic>
</td>
<td align="left">positive</td>
<td align="left">RT-PCR with Sanger sequencing</td>
</tr>
<tr>
<td align="left">pos_S4</td>
<td align="left">Anaplastic astrocytoma, IDH-mutant</td>
<td align="left">
<italic>EWSR1-FLI1</italic>
</td>
<td align="left">positive</td>
<td align="left">RT-PCR</td>
</tr>
<tr>
<td align="left">pos_S5</td>
<td align="left">Pilocytic astrocytoma</td>
<td align="left">
<italic>FAM131B-BRAF</italic>
</td>
<td align="left">positive</td>
<td align="left">RT-PCR with Sanger sequencing</td>
</tr>
<tr>
<td align="left">pos_S6</td>
<td align="left">Glioblastoma, IDH-wildtype</td>
<td align="left">
<italic>FGFR3-TACC3</italic>
</td>
<td align="left">positive</td>
<td align="left">RT-PCR with Sanger sequencing</td>
</tr>
<tr>
<td align="left">pos_S7</td>
<td align="left">Glioblastoma, IDH-mutant</td>
<td align="left">
<italic>MN1-MOB3B</italic>
</td>
<td align="left">positive</td>
<td align="left">RT-PCR with Sanger sequencing</td>
</tr>
<tr>
<td align="left">pos_S8</td>
<td align="left">Pilocytic astrocytoma</td>
<td align="left">
<italic>SRGAP3-RAF1</italic>
</td>
<td align="left">positive</td>
<td align="left">CMA</td>
</tr>
<tr>
<td align="left">pos_S9</td>
<td align="left">Anaplastic ependymoma (WHO grade III)</td>
<td align="left">
<italic>C11ORF95-RELA</italic>
</td>
<td align="left">positive</td>
<td align="left">RT-PCR with Sanger sequencing</td>
</tr>
<tr>
<td align="left">pos_S10</td>
<td align="left">Pilocytic astrocytoma</td>
<td align="left">
<italic>PDE4B-NTRK2</italic>
</td>
<td align="left">positive</td>
<td align="left">RT-PCR with Sanger sequencing</td>
</tr>
<tr>
<td align="left">pos_S11</td>
<td align="left">Pleomorphic xanthoastrocytoma</td>
<td align="left">
<italic>QKI-RAF1</italic>
</td>
<td align="left">positive</td>
<td align="left">RT-PCR with Sanger sequencing</td>
</tr>
<tr>
<td align="left">pos_S12</td>
<td align="left">Solitary fibrous tumor</td>
<td align="left">
<italic>NAB2-STAT6</italic>
</td>
<td align="left">positive</td>
<td align="left">RT-PCR with Sanger sequencing</td>
</tr>
<tr>
<td rowspan="13" align="left">pos_S13</td>
<td rowspan="13" align="left">Negative sample spiked in with fusion oligos</td>
<td align="left">
<italic>SRGAP3-RAF1</italic>
</td>
<td rowspan="13" align="left">positive</td>
<td rowspan="13" align="left">Specific synthetic oligonucleotide design</td>
</tr>
<tr>
<td align="left">
<italic>KIAA1549-BRAF</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>MYB-QKI</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>FAM131B-BRAF</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>PVT1-MYC</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>FGFR1-TACC1</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>DDX31-GFI1B</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>SLC44A1-PRKCA</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>MYB-ESR1</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>PTPRZ1-MET</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>TPM3-NTRK1</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>MYB-PCDHGA1</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>FGFR1-FGFR1 (E9-E19)</italic>
</td>
</tr>
<tr>
<td rowspan="27" align="left">pos_S14</td>
<td rowspan="27" align="left">Negative cell line with insilico spiked in fusions</td>
<td align="left">
<italic>FGFR3-FAM184B</italic>
</td>
<td rowspan="27" align="left">positive</td>
<td rowspan="27" align="left">Insilico spike in</td>
</tr>
<tr>
<td align="left">
<italic>MET-ST7</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>NTRK1-TPR</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>ZSCAN21-MET</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>ATG7-RAF1</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>MET-PTPRZ1</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>ETV6-NTRK3</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>PACRG-QKI</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>KIAA1549-BRAF</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>TFG-MET</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>FN1-FGFR1</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>NTRK2-VCL</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>FGFR3-TACC3</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>KIAA1549-BRAF</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>QKI-RAF1</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>QKI-NTRK2</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>NAB2-STAT6</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>EGFR-EGFR</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>KIF21B-NTRK1</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>YAP1-FAM118B</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>C11orf95-RELA</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>PDE4B-NTRK1</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>EWSR1-FLI1</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>FAM131B-BRAF</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>FGFR1-TACC1</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>MYB-QKI</italic>
</td>
</tr>
<tr>
<td align="left">
<italic>SRGAP3-RAF1</italic>
</td>
</tr>
<tr>
<td align="left">neg_S1</td>
<td align="left">CNS negative</td>
<td align="left">None</td>
<td align="left">negative</td>
<td align="left">CNS negative</td>
</tr>
<tr>
<td align="left">neg_S2</td>
<td align="left">CNS negative</td>
<td align="left">None</td>
<td align="left">negative</td>
<td align="left">CNS negative</td>
</tr>
<tr>
<td align="left">neg_S3</td>
<td align="left">CNS negative</td>
<td align="left">None</td>
<td align="left">negative</td>
<td align="left">CNS negative</td>
</tr>
<tr>
<td align="left">neg_S4</td>
<td align="left">CNS negative</td>
<td align="left">None</td>
<td align="left">negative</td>
<td align="left">CNS negative</td>
</tr>
<tr>
<td align="left">neg_S5</td>
<td align="left">CNS negative</td>
<td align="left">None</td>
<td align="left">negative</td>
<td align="left">CNS negative</td>
</tr>
<tr>
<td align="left">neg_S6</td>
<td align="left">CNS negative</td>
<td align="left">None</td>
<td align="left">negative</td>
<td align="left">CNS negative</td>
</tr>
<tr>
<td align="left">neg_S7</td>
<td align="left">CNS negative</td>
<td align="left">None</td>
<td align="left">negative</td>
<td align="left">CNS negative</td>
</tr>
<tr>
<td align="left">neg_S8</td>
<td align="left">CNS negative</td>
<td align="left">None</td>
<td align="left">negative</td>
<td align="left">CNS negative</td>
</tr>
<tr>
<td align="left">neg_S9</td>
<td align="left">CNS negative</td>
<td align="left">None</td>
<td align="left">negative</td>
<td align="left">CNS negative</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-4">
<title>Orthogonal Validation</title>
<p>Gene fusions detected while benchmarking was confirmed using OncoScan FFPE Assay Kit (Thermo Fisher Scientific, Waltham, MA) chromosomal microarray (CMA) and reverse transcription polymerase chain reaction (RT-PCR) with or without subsequent direct Sanger sequencing as previously described (<xref ref-type="bibr" rid="B14">Gliem and Aypar, 2017</xref>).</p>
</sec>
<sec id="s2-5">
<title>Library Preparation and Sequencing</title>
<p>RNA was extracted from macrodissected, unstained slides using the QIAamp miRNeasy FFPE kit and nucleic acid was quantitated with the NanoDrop 2.0 system (<xref ref-type="bibr" rid="B4">Blessing et&#x20;al., 2019</xref>). Library preparation was performed with the QIAseq RNAscan chemistry using 200&#xa0;ng of RNA per manufacturer&#x2019;s protocol recommendations. Final libraries were quantitated using a Qubit fluorometer, with 8 equimolar samples pooled and sequenced on an Illumina MiSeq instrument. Raw sequencing data was de-multiplexed into FASTQ files and processed through the SeekFusion pipeline.</p>
</sec>
<sec id="s2-6">
<title>SeekFusion Fusion Detection Workflow</title>
<p>SeekFusion was designed in Cromwell/WDL to enable standalone or server mode workflow execution. Server mode performs tasks in parallel for every gene targeted (<xref ref-type="fig" rid="F2">Figure&#x20;2</xref>). The pipeline is configurable for each module, and parameters can be fine-tuned through the profile settings file for the pipeline. Details of the individual steps comprising the pipeline are described in the following sections.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Workflow of SeekFusion pipeline. The pipeline begins with deduping the reads utilizing the UMIs; the deduped reads are then aligned to target genes. The aligned reads are pooled based on the genes and for each gene-based read pool a <italic>de-novo</italic> assembly is performed to construct longer contigs. The contigs are aligned to the reference genome to identify potential breakpoints. The potential breakpoints are annotated, and reads are aligned to the breakpoints to visualize the fusions.</p>
</caption>
<graphic xlink:href="fgene-12-739054-g002.tif"/>
</fig>
</sec>
<sec id="s2-7">
<title>UMI Based PCR Duplicate Removal (UMI Consensus)</title>
<p>PCR duplication is a byproduct of amplification steps in sequencing protocols that produces multiple copies of individual sequence fragments and artificially inflates their abundance. The QIAseq RNAscan chemistry utilizes UMIs that are random nucleotide sequences incorporated into sequence fragments prior to amplification to enable the identification of PCR duplicate reads originating from the same sequence fragment. Barring technical and sequencing errors, each of these duplicated reads will share identical nucleotide sequence. In reality, however, sequencing errors introduce differences in duplicate reads. The &#x201c;deduping&#x201d; module accounts for these errors and attempts to identify the PCR duplicates irrespective of sequencing error. It then builds a single consensus read from identified PCR duplicates and produces an updated Phred quality score for each base that accurately represents the summary evidence when estimating the probability of an erroneous base call. This approach produces higher confidence base calls that facilitate accuracy of downstream processing and improves the ability to discern true&#x20;calls.</p>
<p>The deduping algorithm assumes a prior probability for the nucleotides at a particular position and uses Bayes theorem to update the probability of the base in a particular position of a UMI using conditional independence (See <xref ref-type="sec" rid="s11">Supplementary Methods</xref>).</p>
<p>FASTQ files are used as input to the deduping module. The algorithm is written in C and based on a highly efficient and customized data structure. Efficient binary representation of k-mers (up to length 16) enables perfect hashing and rapid computations of hamming distances. The algorithms also make use of BGZF compressed FASTQ files (this is the same compression method used in bam files) which enables indexing and fast random read access within the compressed FASTQ&#x20;files.</p>
</sec>
<sec id="s2-8">
<title>Trimming Adapters</title>
<p>Fastp (<xref ref-type="bibr" rid="B9">Chen et&#x20;al., 2018</xref>) was implemented with default settings to trim the adapters from reads produced by the QIAseq RNAscan chemistry, with adapter sequences provided in a FASTA file using the--adapter_fasta option.</p>
</sec>
<sec id="s2-9">
<title>Alignment-Based Read Reduction and <italic>De-Novo</italic> Assembly</title>
<p>
<italic>De-novo</italic> assembly is a computationally intensive process that has the potential to overwhelm computational resources (<xref ref-type="bibr" rid="B34">Sze et&#x20;al., 2017</xref>). To alleviate this potential bottleneck, a pre-processing step was formulated to reduce the number of identical or high similarity reads entering the <italic>de-novo</italic> assembly stage. A target reference database for read alignment was created using the full nucleotide sequence of the longest coding transcript for all genes targeted by the neuro oncology NGS panel, identified by HGNC (<xref ref-type="bibr" rid="B42">Yates et&#x20;al., 2017</xref>) gene names. BWA-mem (<xref ref-type="bibr" rid="B23">Li and Durbin, 2009</xref>) from Sentieon (Sentieon, Mountain View, CA, United&#x20;States) was used to enable rapid alignment to the gene-based reference database. BWA-mem can be substituted with the open source BWA-mem tool using profile settings for the pipeline. Default parameters for paired-end alignment were used. Best alignments were selected for each read and binned into gene-specific SAM files utilizing the SAMTOOLS (<xref ref-type="bibr" rid="B24">Li et&#x20;al., 2009</xref>) view functionality with the aligned BAM file. A custom python module was developed to identify alignments sharing identical length and chromosomal start and end coordinates and filter the subsequent output to five copies or fewer. Reads passing this stage were output in FASTA format for <italic>de-novo</italic> assembly. Following alignment-based reduction, reads were assembled using the CAP3 (<xref ref-type="bibr" rid="B17">Huang and Madan, 1999</xref>) <italic>de-novo</italic> assembly tool with default settings (<xref ref-type="sec" rid="s11">Supplementary Table S1</xref>). Assembly was performed individually on each gene-specific SAM file to improve the specificity of the <italic>de-novo</italic> assembly.</p>
</sec>
<sec id="s2-10">
<title>Aligning Contigs to the Reference Genome</title>
<p>
<italic>De-novo</italic> assembled contigs were aligned to the human reference genome (hg19) using BLAT (<xref ref-type="bibr" rid="B19">Kent, 2002</xref>). Those generating multiple non-contiguous alignments were retained as potential fusion candidates or splice-variants for downstream characterization. The parameters for BLAT are specified in <xref ref-type="sec" rid="s11">Supplementary Table S1</xref>. Custom modules were developed to remove multiple contigs mapping to identical genomic regions from BLAT results, reducing redundancy in the putative gene fusion breakpoints.</p>
</sec>
<sec id="s2-11">
<title>Filtering Modules</title>
<p>Filtering modules were developed to reduce false positive calls occurring due to mononucleotide repeats, homology, and other recurrent artifacts. To filter the false positives due to mononucleotide repeats, the filtering script looks for repetitive regions using a Mononucleotide Repeat Ratio (MRR). The mononucleotide repeat ratio calculation is described in the <xref ref-type="sec" rid="s11">Supplementary Methods</xref>.</p>
<p>For filtering highly homologous regions, SeekFusion flags calls that are recurrent in every sample due to homology/low complexity using a blocklist. The blocklist calls are presented in the <xref ref-type="sec" rid="s11">Supplementary Data</xref>. Events involving a single transcript are filtered from the output by default. An inclusion feature is provided to enable detection of clinically relevant single transcript events, such as epidermal growth factor receptor variant III (EGFRvIII) in glioblastoma (<xref ref-type="bibr" rid="B2">An et&#x20;al., 2018</xref>).</p>
</sec>
<sec id="s2-12">
<title>Annotation of Contigs Representing Potential Fusions</title>
<p>A custom script was developed to annotate identified contig breakpoints and provide genomic context. Annotations generated include the gene, exon and coding frame status of the fusion, i.e.,&#x20;if the fusion is likely protein-coding. One of the key annotations provided by the annotation module is if a fusion is In-frame, which implies that there was no frame shift in the 3&#x2032;-gene, regardless of single amino acid mutation or insertion events at the fusion junction. <xref ref-type="table" rid="T2">Table&#x20;2</xref> contains examples of fusion annotations generated by the pipeline.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Fusion annotation generated by the pipeline for documented fusions.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Fusion</th>
<th align="center">Fusion location</th>
<th align="center">Frame status</th>
<th align="center">5&#x2032; exon annotation</th>
<th align="center">3&#x2032;_exon_annotation</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">FGFR3-TACC3</td>
<td align="center">Exon-exon_boundary</td>
<td align="left">In-frame</td>
<td align="left">&#x2b;&#x7c;End_E17&#x7c;FGFR3&#x7c;NM_001163213</td>
<td align="left">&#x2b;&#x7c;Start_E11&#x7c;TACC3&#x7c;NM_006342</td>
</tr>
<tr>
<td align="left">FN1-FGFR1</td>
<td align="center">Exon-exon_boundary</td>
<td align="left">In-frame</td>
<td align="left">-&#x7c;End_E27&#x7c;FN1&#x7c;NM_212482</td>
<td align="left">-&#x7c;Start_E5&#x7c;FGFR1&#x7c;NM_023110</td>
</tr>
<tr>
<td align="left">ETV6-NTRK3</td>
<td align="center">Exon-exon_boundary</td>
<td align="left">In-frame</td>
<td align="left">&#x2b;&#x7c;End_E5&#x7c;ETV6&#x7c;NM_001987</td>
<td align="left">-&#x7c;Start_E15&#x7c;NTRK3&#x7c;NM_001012338</td>
</tr>
<tr>
<td align="left">SSX1-SSX8</td>
<td align="center">Exon-exon_boundary</td>
<td align="left">In-frame</td>
<td align="left">&#x2b;&#x7c;End_E5&#x7c;SSX1&#x7c;NM_005635</td>
<td align="left">&#x2b;&#x7c;Start_E6&#x7c;SSX8&#x7c;NR_027250</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-13">
<title>Putative Fusion Reference Generation and Alignment-Based Event Confirmation</title>
<p>All reads originally passing the UMI-based deduping and adapter trimming steps were retrieved and aligned to the hg19 human reference transcriptome from Ensembl and Refseq using BWA-mem. Reads that produced perfect alignments to known gene transcripts were removed from further consideration as they likely represent normal transcriptional events. For each gene fusion candidate identified by the alignment of assembled contigs to the human genome, a chimeric construct of the purported fusion sequence was created (150 base pairs bidirectional beyond the putative breakpoint). Constructs annotated as occurring within a single known transcript were removed from further consideration, with the exception of select known aberrant single-gene events. The retained reads were aligned against the potentially chimeric gene fusion reference sequences using BWA-mem. Only reads spanning at least 10 bases beyond a breakpoint were considered as evidence of a potential fusion. A single supporting read was required as evidence of a fusion candidate and all candidates were output in a standardized variant call format (VCF) file. While minimal read support was required to facilitate downstream benchmarking, the supporting read threshold is user configurable. Finally, the fusion-construct aligned BAM file was used to create an Integrative Genomics Viewer (IGV) (<xref ref-type="bibr" rid="B31">Robinson et&#x20;al., 2011</xref>) session to enable visual inspection of the evidence supporting each fusion candidate (<xref ref-type="fig" rid="F3">Figure&#x20;3</xref>).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>An example of the C11ORF95-RELA positive fusion on IGV. Multiple reads generate high-confidence alignments and span the breakpoint in support of the putative fusion transcript.</p>
</caption>
<graphic xlink:href="fgene-12-739054-g003.tif"/>
</fig>
</sec>
<sec id="s2-14">
<title>Packaging of Pipeline for General Use</title>
<p>The pipeline is available in a Docker container (<xref ref-type="bibr" rid="B29">Docker, 2014</xref>) image and is available to download from <ext-link ext-link-type="uri" xlink:href="https://hub.docker.com/repository/docker/jagadhesh89/seekfusion">https://hub.docker.com/repository/docker/jagadhesh89/seekfusion</ext-link>. The usage details are explained in the README file. In addition to the easy to use docker image, the source code is also made available via github (<ext-link ext-link-type="uri" xlink:href="https://github.com/jagadhesh89/seekfusion">https://github.com/jagadhesh89/seekfusion</ext-link>).</p>
</sec>
<sec id="s2-15">
<title>Benchmarking</title>
<p>Benchmarking was performed for SeekFusion, JAFFA-hybrid (<xref ref-type="bibr" rid="B10">Davidson et&#x20;al., 2015</xref>), STAR-Fusion (<xref ref-type="bibr" rid="B15">Haas et&#x20;al., 2019</xref>) and TOPHAT-Fusion (<xref ref-type="bibr" rid="B20">Kim and Salzberg, 2011</xref>), mirroring benchmarking analyses performed by the STAR-Fusion study (<xref ref-type="bibr" rid="B15">Haas et&#x20;al., 2019</xref>). Selected samples were processed through the four pipelines for benchmarking in two separate analyses. In Analysis 1, FASTQs were trimmed with FASTP (<xref ref-type="bibr" rid="B9">Chen et&#x20;al., 2018</xref>), subsequently deduped using UMIs, and supplied as inputs to the pipelines. In Analysis 2, the raw FASTQs were provided as pipeline inputs for the 12 gene fusion positive cases, the 9 gene fusion negative cases, and the positive controls. Analysis 2 was used to assess tool performance with raw sample FASTQs, without deduping, however the adapters and UMIs were trimmed to prune the reads for specific tools (see <xref ref-type="sec" rid="s11">Supplementary Section S2.5</xref>). In Analysis 2, JAFFA-Hybrid posed a scalability challenge; the pipeline ran for over 24&#xa0;h per sample before failure due to memory requirements. Due to the scalability issue, JAFFA-Hybrid was excluded from Analysis 2. However for the analysis 2, we benchmarked JAFFA-Direct to circumvent the scalability issues of JAFFA-hybrid.</p>
<p>Accuracy was measured based on putative gene fusion calls using the following equation at varying levels of fusion read support thresholds.<disp-formula id="equ1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ2">
<mml:math id="m2">
<mml:mrow>
<mml:mi mathvariant="normal">Algorithm&#xa0;sensitivity&#xa0;was&#xa0;also&#xa0;measured&#xa0;using</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>For in-silico generated fusions, supporting reads are a known quantity. Mean Absolute Percentage Error (<xref ref-type="bibr" rid="B11">de Myttenaere et&#x20;al., 2016</xref>) (MAPE) and Symmetric Mean Absolute Percentage Error (<xref ref-type="bibr" rid="B36">Tofallis, 2015</xref>) (SMAPE) values were calculated as a measure of known supporting reads detected by the four algorithms. A Wilcoxon test was performed on the SMAPE values between SeekFusion and other tools. Key considerations on determining the winning tool from benchmarking involved high accuracy, high true positive rate, ease of installation, ease of usage in a clinical setting and turnaround time for the pipeline.</p>
</sec>
<sec id="s2-16">
<title>Clinical Testing on Neurological Oncology and Sarcoma Samples</title>
<p>The neurological oncology assay was developed, verified, validated, New York (NY) State approved and implemented clinically using the SeekFusion algorithm. In the verification study, 63 total unique samples were processed (59 CNS tumor samples, 1 CNS tumor sample spiked with 13 known synthetic oligonucleotides and 3 brain gliosis samples). The NGS assay results were confirmed by RT-PCR and CMA tests. The overall accuracy of the NGS assay was 96.1% and no fusion transcripts were detected in negative control samples indicating an overall specificity of 100%. Following the clinical go-live of the assay, 2,979 clinical patient samples ordered for neurological oncology testing and were processed using the SeekFusion algorithm. A sarcoma assay was developed, verified, validated and approved for clinical testing by NY state. The assay uses the SeekFusion algorithm and evaluates 138 gene targets for presence of somatic gene fusions and common BCOR internal tandem duplications (ITDs) using the same QIAseq RNAscan chemistry. In the verification study, this next-generation sequencing (NGS) assay was performed in 111 sarcoma formalin-fixed, paraffin-embedded (FFPE) and cytology samples (86 fusion positive and 25 fusion negative). The NGS assay results were confirmed by RT-PCR and FISH tests. The overall accuracy of the NGS assay was 95.5%. No targeted gene fusions were detected in 20 negative control samples (100% specificity), A total of 1,505 clinical cases were processed using the SeekFusion algorithm. An IRB request was submitted and approved to share the clinically reported gene fusions.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<sec id="s3-1">
<title>Accuracy and Sensitivity-Based Performance</title>
<p>The accuracy of SeekFusion was highest among the benchmarked tools, followed by JAFFA-hybrid, STAR-Fusion and TOPHAT-Fusion (<xref ref-type="fig" rid="F4">Figure&#x20;4A</xref>). True positive rate was measured as a surrogate of clinical utility and SeekFusion consistently performed best among the tools at all levels of read support cut-off (<xref ref-type="fig" rid="F4">Figure&#x20;4B</xref>). Read support cut-off is commonly applied to fusion detection algorithms for reducing false positive rate and improving accuracy (<xref ref-type="bibr" rid="B25">Liu et&#x20;al., 2016</xref>). Read cut-offs for calling a fusion were assessed for all tools at levels of 1, 2, 3, 4, 5, 10, 20, 30, 40, 50, 60, 70, 80, 90 and 100 supporting reads, and accuracy and true-positive rates were calculated. Intuitively if an algorithm consistently reports gene fusions with fewer reads supporting the event, a higher read cut-off will result in a lower accuracy and true positive rate. JAFFA hybrid was ranked second, followed by STAR-Fusion; however, both JAFFA and STAR-Fusion produced higher numbers of calls beyond the known fusion events. These calls were closely examined and are documented in <xref ref-type="sec" rid="s11">Supplementary File S1</xref>. Manual curation was performed, and the extra calls were classified as false positives; all these calls were due to homology, low complexity, or identified with very low frequency (<xref ref-type="sec" rid="s11">Supplementary Table S5</xref>). Tophat Fusion performed the worst, with a high rate of false negatives. SeekFusion was the only tool observed to be 100% concordant with the expected positive gene fusions whereas STAR-Fusion, JAFFA-Hybrid and TOPHAT-Fusion detected 83, 77 and 1% respectively.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>
<bold>(A)</bold> Accuracy of the tools was assessed for the tools JAFFA, STAR-Fusion, Tophat Fusion and SeekFusion at varying read support thresholds. SeekFusion demonstrated highest accuracy in detecting the gene fusions at all levels of read support. The median accuracy of SeekFusion was 82% followed by JAFFA-Hybrid at 60%, STAR-Fusion at 39% and Tophat-Fusion at 20% for Mode1. SeekFusion was the only pipeline that demonstrated high accuracy in detecting fusions from raw FASTQs. <bold>(B)</bold> True Positive Rate of the tools was assessed for the tools JAFFA, STAR-Fusion, Tophat Fusion and SeekFusion at varying read support thresholds. SeekFusion demonstrated the highest sensitivity in detecting the gene fusions at all levels of read support. For the deduped and trimmed FASTQs, we observed that the median true positive rate of SeekFusion was 89% followed by STAR-Fusion at 69% and JAFFA-Hybrid at 65%. Tophat Fusion performed least favorably among the tools assessed with a median true positive rate of 1%. SeekFusion was the only tool that successfully detected true positives in the raw FASTQs.</p>
</caption>
<graphic xlink:href="fgene-12-739054-g004.tif"/>
</fig>
</sec>
<sec id="s3-2">
<title>Levels of Read Support for Detected Fusions</title>
<p>The number of gene fusion-supporting reads detected by the four tools was assessed for the true positive gene fusions. Deduped and trimmed FASTQs were utilized in the analysis since only SeekFusion demonstrated the ability to successfully detect the true positive fusions from raw FASTQs. SeekFusion reported more gene fusion supporting reads than the other tools in 83% of cases (<xref ref-type="fig" rid="F5">Figure&#x20;5</xref>). SeekFusion was followed by JAFFA, STAR-Fusion and Tophat Fusion for the number of fusion supporting&#x20;reads.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>The gene fusion read support in log scale for each fusion detected by SeekFusion, STAR-Fusion, Tophat Fusion and JAFFA. In 83% of cases, SeekFusion demonstrated the highest read support amongst the algorithms tested.</p>
</caption>
<graphic xlink:href="fgene-12-739054-g005.tif"/>
</fig>
<p>MAPE and SMAPE are widely used measures of forecast accuracy; they aid in assessing accuracy based on percentage errors from the actual. MAPE (<xref ref-type="bibr" rid="B11">de Myttenaere et&#x20;al., 2016</xref>) and SMAPE (<xref ref-type="bibr" rid="B36">Tofallis, 2015</xref>) values indicate that SeekFusion performed the best in terms of observed to expected gene fusion-supporting reads amongst the four tools (<xref ref-type="table" rid="T3">Table&#x20;3</xref>). A Wilcoxon test (<xref ref-type="bibr" rid="B40">Wilcoxon, 1945</xref>) on the SMAPE values between SeekFusion and other tools supported the hypothesis that SeekFusion retrieved the highest proportion of known fusion-supporting reads in the in-silico samples (<xref ref-type="table" rid="T3">Table&#x20;3</xref>). Each tool considered has a different method of fusion transcript identification and these account for differences in levels of read support.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>MAPE and SMAPE values for the 4 tools and the <italic>p</italic>-values based on the Wilcoxon test on SMAPE values between SeekFusion and other&#x20;tools.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Tool</th>
<th align="center">MAPE</th>
<th align="center">SMAPE</th>
<th align="center">
<italic>p</italic>-value from Wilcoxon test on SMAPE values between SeekFusion and other tools</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">SeekFusion</td>
<td align="center">0.21077</td>
<td align="center">0.12887</td>
<td align="center">Not applicable</td>
</tr>
<tr>
<td align="left">STAR-fusion</td>
<td align="center">0.78855</td>
<td align="center">0.67441</td>
<td align="center">1.490116e-08</td>
</tr>
<tr>
<td align="left">TOPHAT-fusion</td>
<td align="center">1</td>
<td align="center">1</td>
<td align="center">1.490116e-08</td>
</tr>
<tr>
<td align="left">JAFFA hybrid</td>
<td align="center">0.54454</td>
<td align="center">0.47727</td>
<td align="center">0.0009347796</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-3">
<title>Computational Time Requirements and Peak Memory Utilization</title>
<p>In addition to analytical performance, clinical bioinformatics pipelines are required to perform in a time-efficient manner to maintain necessary assay turnaround time. Benchmarking of runtimes was performed to assess the suitability of the algorithms for real-world clinical use. The single in-silico sample was excluded from the benchmarking since it does not represent a true patient sample. For both Analysis 1, JAFFA exhibited time-requirements that exceeded all other tools (<xref ref-type="fig" rid="F6">Figure&#x20;6</xref>). The remaining three tools were comparable in their runtimes and were all considered amenable to routine clinical use, with STAR-Fusion performing most favorably in terms of time-requirements alone. In Analysis 2 due to scalability challenges posed by JAFFA-hybrid mode, analysis was performed in JAFFA-Direct mode and it was found that JAFFA-Direct mode was most favorable in terms of time-requirements&#x20;alone.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Pipeline completion time in minutes for the 4 tools. For Analysis 1, STAR-Fusion was the fastest among the 4 tools with median completion time of 6&#xa0;min followed by Tophat Fusion at 20&#xa0;min, SeekFusion at 45&#xa0;min and JAFFA-Hybrid at 196&#xa0;min. For Analysis 2, STAR-Fusion had a median completion time of 29&#xa0;min, SeekFusion had a median completion time of 36&#xa0;min, Tophat-Fusion had a median completion time of 53&#xa0;min and JAFFA-Hybrid exhibited scalability issues withe pipeline failing due to memory issues after running for over 24&#xa0;h (so the data was capped at 400&#xa0;min for JAFFA Hybrid in the raw FASTQ mode). To circumvent the scalability issues, analysis in direct mode of JAFFA was performed and this mode demonstrated the fastest median completion time of 20&#xa0;min.</p>
</caption>
<graphic xlink:href="fgene-12-739054-g006.tif"/>
</fig>
<p>Peak memory utilization of the individual tools benchmarked were assessed. The maximum input memory for each of the pipelines were capped at 180&#xa0;GB. For each sample, the maximum memory used by the tool is used for calculating a per tool median memory utilization for analysis 1 and analysis 2. The median is rounded to the nearest GB. JAFFA utilized the most memory with a median memory utilization of 171&#xa0;GB in analysis 1 and 172&#xa0;GB in analysis 2. SeekFusion peaked at a median memory of 58&#xa0;GB in analysis 1 and 64&#xa0;GB in analysis 2, with CAP3 assembly being the most memory consuming module in the SeekFusion pipeline. STAR-Fusion peaked at a median memory of 30&#xa0;GB in analysis 1 and 34&#xa0;GB in analysis 2. Tophat-Fusion used the least memory; with peak of median memory in analysis 1 and analysis 2 of 2&#xa0;GB.</p>
<p>Taken together, while considering high accuracy, high true positive rate, ease of installation, ease of usage in a clinical setting and turnaround time for the pipeline, SeekFusion was identified as the method for use of clinical assay development. Neurological oncology assay and sarcoma assay were developed, verified and approved as detailed in the <italic>Methods</italic>, <italic>Clinical Testing On Neurological Oncology and Sarcoma Samples</italic> section.</p>
</sec>
<sec id="s3-4">
<title>Clinical Testing Using Neurological and Oncology Assays</title>
<p>Of the 2,979 cases tested clinically using the neurological oncology assay, 569 cases were reported with clinically relevant gene fusions after careful variant review and orthogonal confirmation using RT-PCR (<xref ref-type="fig" rid="F7">Figure&#x20;7A</xref>). Out of the 1,505 cases tested clinically using the sarcoma assay, 476 cases were reported (<xref ref-type="fig" rid="F7">Figure&#x20;7B</xref>).</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>
<bold>(A)</bold> Clinically reported gene fusions in neurological oncology cases indicating frequency of common occurring events in the tested cohort. It is observed that the EGFRvIII is the most observed event followed by the KIAA1549-BRAF and the FGFR3-TACC3 fusion. <bold>(B)</bold> Clinically reported gene fusions in sarcoma assay indicating frequency of common occurring events in the tested cohort. EWSR1 related fusions are the most commonly observed in our tested cohort.</p>
</caption>
<graphic xlink:href="fgene-12-739054-g007.tif"/>
</fig>
<p>In the neurological oncology assay, EGFRvIII was the most commonly found rearrangement, 195 out of 2,979 patients tested were identified with this rearrangement. EGFRvIII has been established to be present in up to 28&#x2013;30% of the glioblastoma cells and constitutes to be a therapeutic target (<xref ref-type="bibr" rid="B32">Rutkowska et&#x20;al., 2019</xref>). KIAA1549-BRAF was identified in 120 patients and this rearrangement has been established to be frequently found in pilocytic astrocytoma (<xref ref-type="bibr" rid="B8">Chen et&#x20;al., 2019</xref>). In addition, our pipeline picked up several established fusions related to glioblastoma such as FGFR3-TACC3 and PTPRZ1-MET. Confirming the presence of NTRK gene fusions can guide treatment of the solid tumors (<xref ref-type="bibr" rid="B13">Filippi et&#x20;al., 2021</xref>). Using our SeekFusion method on this assay, we reported NTRK gene fusions for 46&#x20;cases.</p>
<p>In the sarcoma assay, EWSR1 was the most commonly rearranged gene in our case series, with 3&#x2032; partners including FLI1, ERG, WT1, CREB1, ATF, PBX3, CREB3L1, TFCP, NR4A3, NFATC2, POU5F1, DDIT3, CREM, PATZ1, and COLCA2 in a wide variety of tumor types. SS18-SSX1 was reported in 32 synovial sarcoma cases. PAX3 was the 5&#x2032; fusion partner in 29 cases with 3&#x2032; partners being FOXO1 fusions identified in alveolar rhabdomyosarcoma cases, MAML3 and NCOA1 fusions in two different biphenotypic sinonasal sarcoma&#x20;cases.</p>
</sec>
<sec id="s3-5">
<title>Novel Fusion Discovery</title>
<p>As described previously the assay chemistry is gene-partner agnostic, enabling identification of novel gene fusions, and many novel fusions were detected in the clinical patient cohort. We have summarized all the neurological cancer and sarcoma gene fusions identified by our method in the <xref ref-type="sec" rid="s11">Supplementary Table S7</xref>. We present two cases here in detail.</p>
<p>Case 1 is a 45-year-old male who presented with a lower left extremity mass suggestive of a low-grade mesenchymal neoplasm. By immunohistochemistry (IHC) the neoplastic cells were focally positive for desmin, epithelial membrane antigen and STAT6 and showed limited expression of TRK. ALK, OSCAR keratin and TLE1 were negative. Histologically the neoplasm was a very unusual appearing, multicystic proliferation of rather bland small round epithelioid cells, in areas showing the presence of an apparent second cell population, consisting of flattened cells with more abundant eosinophilic cytoplasm. This morphologic appearance was suggestive of at some level an adnexal lesion however a variety of epithelial markers are essentially negative. It also appeared to be related to angiomatoid fibrous histiocytoma although the morphologic features of this lesion are atypical and although it showed very limited coexpression of desmin and EMA it was negative for the EWSR1 and FUS gene fusions that characterize this tumor. Based on the very bland morphology of this lesion and the extremely low mitotic rate it is believed that the lesion is either entirely benign or at most has some limited capacity for local recurrence. NGS testing identified a novel TFG-ZBTB10 fusion (<xref ref-type="fig" rid="F8">Figure&#x20;8A</xref>). The specific fusion TFG-ZBTB10 has not been described as a recurrent oncogenic event. However, both genes have been involved in fusion events with several other partners. TFG (Trafficking for Endoplasmic Reticulum to Golgi Regulator) has been reported as a 5&#x2032; partner in several fusions genes in acute leukemias, sarcomas and carcinomas. In contrast, ZBTB10 (Zinc Finger and BTB Domain-Containing Protein 10), which encodes for a zinc finger protein, has been found in isolated fusion events with other partner genes in breast and colonic adenocarcinomas.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>
<bold>(A)</bold> Case 1 showing the reads spanning the TFG-ZBTB10 fusion. <bold>(B)</bold> Case 2 showing spanning reads for the LRRFIP1-ALK fusion.</p>
</caption>
<graphic xlink:href="fgene-12-739054-g008.tif"/>
</fig>
<p>Case 2 is a 23-year-old male who presented with a 14&#x20;&#xd7; 4.8 &#xd7; 7.8&#xa0;cm right chest mass which was eroding the underlying clavicle. Microscopically, this is a spindle cell neoplasm showing a whorling, fascicular growth pattern with scattered chronic inflammatory cells and foci of hyalinized fibrosis. The somewhat enlarged nuclei raise concern for the possibility of a low-grade sarcoma, but they maintain a rather uniform appearance and lack significant atypia. IHC studies show the tumor cells to be positive for FLI-1 and negative for S100 protein, SOX10, desmin, myogenin, WT1, SMA, CD34 and pan-cytokeratin. Given these findings, the differential diagnosis included follicular dendritic sarcoma and angiomatoid fibrous histiocytoma. Additional stains performed locally showed the cells to be negative for CD31, CD35 and CD21. NGS testing revealed a novel RRFIP1-ALK gene fusion (<xref ref-type="fig" rid="F8">Figure&#x20;8B</xref>). ALK fusions have been identified in a variety of neoplasms, including inflammatory myofibroblastic tumor, many subtypes of lymphomas and leukemias, adenocarcinomas, benign fibrous histiocytoma, and several other tumors. LRRFIP1 encodes the leucine-rich repeat flightless-interacting protein 1 (FLI-1 interacting protein), a DNA binding transcriptional repressor and may regulate expression of TNF, EGFR and PDGFA. Additionally, it may regulate smooth muscle cell proliferation following arterial damage through PDGFA repression. Previously, an LRRFIP1-MET fusion was identified in an atypical Spitz tumor (PMID: 26013381). There is also another report of an LRRFIP1-FGFR1 fusion in a myeloid neoplasm (PMID: 19369959). While the LRRFIP1-ALK gene fusion observed in this case has not been previously reported in the literature, tumors with ALK fusions have been shown to respond to ALK-inhibitor therapies. Given the histologic findings, it was felt that the tumor was within the spectrum of an inflammatory myofibroblastic&#x20;tumor.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<p>We have developed a novel bioinformatics pipeline to enable accurate and precise gene fusion detection for commonly used RNA UMI-based amplicon NGS assays. A combination of traditional alignment and <italic>de-novo</italic> assembly-based approaches enables SeekFusion to increase accuracy of fusion calling over other published and widely used gene fusion-calling algorithms, while maintaining reasonable computational expense. We have demonstrated that SeekFusion exhibits analytical sensitivity with minimal false positive calls and produces higher read support per gene fusion call when compared to other benchmarked bioinformatics tools. Furthermore, SeekFusion is computationally efficient and capable of facilitating realistic turnaround times in a clinical setting. The bioinformatics pipeline and quality check turnaround times are less than a day for the neurological oncology assay and the sarcoma assays mentioned in our study, which enables an analytic time of 14&#xa0;days for the assay, from sample collection, until clinical report.</p>
<p>Although STAR-Fusion, JAFFA-Hybrid and TOPHAT-Fusion have previously been benchmarked to perform adequately for gene fusion detection using standard RNA-Seq assays (<xref ref-type="bibr" rid="B15">Haas et&#x20;al., 2019</xref>), our study demonstrates these tools lack in performance for RNA UMI-based amplicon NGS assays. SeekFusion provides a tailored solution capable of sensitive and specific fusion gene detection from commonly used assays such as QIAseq RNAscan or Archer FusionPlex, thus making it widely applicable for clinical use. SeekFusion pipeline&#x2019;s applicability to routine clinical use is further aided by its DOCKER-based deployment. Despite the pipeline&#x2019;s complex multi-step operation, this packaging is compatible with a wide range of deployment systems including high performance clusters and cloud computing platforms. The implementation conceals the minutiae of the pipeline architecture from end-users and ensures ease of configuration with limited requirements for computational expertise.</p>
<p>In terms of computational run-time, all tested algorithms performed at a level that is amenable to clinical turnaround times, with the possible exception of JAFFA, which demonstrated obvious scalability issues and required as much as 6&#xa0;hours per sample to complete a single analysis. STAR-Fusion, Tophat-Fusion and SeekFusion performed similarly, although STAR-Fusion showed overall lower run-times than the other tools. Considered in totality, however, the overall performance characteristics of SeekFusion combined with clinically favorable run-times demonstrate it to be the optimal solution for fusion transcript identification.</p>
<p>SeekFusion pipeline&#x2019;s use of common output file formats represent another advantage for its implementation. Most gene fusion calling algorithms utilize arbitrary formats which present compatibility issues with downstream tools. The use of a standard VCF output for candidate events enables more seamless integration with existing tools and workflows. The generation of an IGV session file further facilitates the utilization of SeekFusion&#x2019;s outputs with common downstream components. The ability to readily visualize the level of support and alignment quality for fusion-supporting reads is a key functionality. The visualization functionality is currently unavailable in the other tested gene fusion calling software and allows end-users to visually judge the quality of the supporting reads and infer confidence of a gene fusion&#x20;call.</p>
<p>A limitation of the previously published tools profiled in this study, beyond core performance metrics, is the lack of ability to detect clinically relevant single gene events (<xref ref-type="sec" rid="s11">Supplementary Table S4</xref>), such as the EGFR vIII transcript variant that is a relatively frequent event in glioblastomas (<xref ref-type="bibr" rid="B2">An et&#x20;al., 2018</xref>). The ability of SeekFusion to detect aberrant single-gene events expands its clinical utility and offers additional advantages over pre-existing algorithms. Customization of filter settings in the SeekFusion pipeline enables the inclusion of specific single gene events as required by clinical end-users, while maintaining a low rate of false positives by excluding normal transcript variants. Furthermore, we have demonstrated SeekFusion&#x2019;s ability to detect novel fusions leveraging the gene-partner agnostic Qiaseq RNA, or similar chemistry, further evidencing its clinical utility. It should be noted that SeekFusion has been tailored to utilize the outputs from common PCR and UMI-based RNA assays such as the Qiaseq assay. It is therefore not likely to natively generalize to alternative, clinically utilized, capture-based methods. While these alternative approaches may be fundamentally similar, further customization and benchmarking will be required to assess the potential applicability of SeekFusion to other chemistries and represents a foundation for future development.</p>
<p>A potential limitation of our study was the relatively small number of clinical samples that were available for initial benchmarking. This reflected the difficulties in obtaining broader research consent, which we overcame by supplementing real clinical samples with oligonucleotide spike-ins and in-silico simulated data. Despite the diversity of the test data utilized, SeekFusion demonstrated high analytical sensitivity and specificity across all datatypes when compared to other available tools, further supporting SeekFusion pipeline&#x2019;s accuracy and versatility. SeekFusion pipeline&#x2019;s packaging allows it to be installed and deployed in a clinical setting with minimal efforts. The pipeline enables customization of thresholds, tools, and settings through a configuration file. The setup of SeekFusion for another clinical assay is easily enabled through simple user configuration, as mentioned in the readme instructions in <ext-link ext-link-type="uri" xlink:href="https://hub.docker.com/repository/docker/jagadhesh89/seekfusion">https://hub.docker.com/repository/docker/jagadhesh89/seekfusion</ext-link>. To validate SeekFusion and establish analytical sensitivity and specificity for a similar clinical assay, it is recommended to run the pipeline on a robust set of known positives and negatives, including positive in-silico controls and a positive control with spiked in oligonucleotides.</p>
<p>In summary, SeekFusion represents an accurate, time-efficient and versatile solution for the clinical detection of fusion transcripts from common RNA UMI-based amplicon NGS assays. It is our hope that the tool&#x2019;s public availability and cross-platform capabilities will enable ready deployment and facilitate accurate gene fusion identification in a range of clinical laboratory and disease settings.</p>
</sec>
</body>
<back>
<sec id="s5">
<title>Data Availability Statement</title>
<p>Datasets used for benchmarking analyses in this study is available in the NCBI&#x2019;s SRA website. The docker shared in article has test samples with real fusions and the entire code base. Any fusion sequence or breakpoint information for identified fusion transcripts in clinical cases can be shared on request. Requests to access the datasets should be directed to <email>balan.jagadheshwar@mayo.edu</email>.</p>
</sec>
<sec id="s6">
<title>Ethics Statement</title>
<p>The studies involving human participants were reviewed and approved by Minimal risk IRB at the Mayo Clinic. The study involves no patient interaction or counseling. The study involves using clinical sample data and processing it through pipeline developed during the study and reporting results. There is no PHI or personal identifiers revealed in the study and the results are completely anonymized. The IRB approved study is titled &#x201c;Investigation of residual samples using next-generation sequencing, conventional karyotyping, FISH, Array, PCR and Sanger sequencing.&#x201d; The IRB number is 15-007359. Written informed consent for participation was not required for this study in accordance with the national legislation and the institutional requirements.</p>
</sec>
<sec id="s7">
<title>Author Contributions</title>
<p>All authors listed have made a substantial, direct, and intellectual contribution to the work and approved it for publication.</p>
</sec>
<sec id="s8">
<title>Funding</title>
<p>This work was funded by the Department of Laboratory Medicine and Pathology, Mayo Clinic, Rochester, Minnesota.</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ack>
<p>We thank Dr. Jaime Davila from QHS (Health Sciences Research) - Biomedical Statistics and Informatics, Mayo Clinic, Rochester, Minnesota for valuable advice for the pipeline development. We also thank Jason Sinnwell and Poulami Barman from QHS (Health Sciences Research) - Biomedical Statistics and Informatics, Mayo Clinic, Rochester, Minnesota for their valuable advice for statistics.</p>
</ack>
<sec id="s11">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2021.739054/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2021.739054/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.PDF" id="SM1" mimetype="application/PDF" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abel</surname>
<given-names>H. J.</given-names>
</name>
<name>
<surname>Al-Kateb</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Cottrell</surname>
<given-names>C. E.</given-names>
</name>
<name>
<surname>Bredemeyer</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Pritchard</surname>
<given-names>C. C.</given-names>
</name>
<name>
<surname>Grossmann</surname>
<given-names>A. H.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Detection of Gene Rearrangements in Targeted Clinical Next-Generation Sequencing</article-title>. <source>J.&#x20;Mol. Diagn.</source> <volume>16</volume> (<issue>4</issue>), <fpage>405</fpage>&#x2013;<lpage>417</lpage>. <pub-id pub-id-type="doi">10.1016/j.jmoldx.2014.03.006</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>An</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Aksoy</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>Q.-W.</given-names>
</name>
<name>
<surname>Weiss</surname>
<given-names>W. A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Epidermal Growth Factor Receptor and EGFRvIII in Glioblastoma: Signaling Pathways and Targeted Therapies</article-title>. <source>Oncogene</source> <volume>37</volume> (<issue>12</issue>), <fpage>1561</fpage>&#x2013;<lpage>1575</lpage>. <pub-id pub-id-type="doi">10.1038/s41388-017-0045-7</pub-id> </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bender</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gronych</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Warnatz</surname>
<given-names>H-J.</given-names>
</name>
<name>
<surname>Hutter</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Gr&#xf6;bner</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ryzhova</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Recurrent MET Fusion Genes Represent a Drug Target in Pediatric Glioblastoma</article-title>. <source>Nat. Med.</source> <volume>22</volume> (<issue>11</issue>), <fpage>1314</fpage>&#x2013;<lpage>1320</lpage>. <pub-id pub-id-type="doi">10.1038/nm.4204</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Blessing</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Blackburn</surname>
<given-names>P. R.</given-names>
</name>
<name>
<surname>Krishnan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Harrod</surname>
<given-names>V. L.</given-names>
</name>
<name>
<surname>Barr Fritcher</surname>
<given-names>E. G.</given-names>
</name>
<name>
<surname>Zysk</surname>
<given-names>C. D.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Desmoplastic Infantile Ganglioglioma: A MAPK Pathway-Driven and Microglia/Macrophage-Rich Neuroepithelial Tumor</article-title>. <source>J.&#x20;Neuropathol. Exp. Neurol.</source> <volume>78</volume> (<issue>11</issue>), <fpage>1011</fpage>&#x2013;<lpage>1021</lpage>. <pub-id pub-id-type="doi">10.1093/jnen/nlz086</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Blomquist</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Crawford</surname>
<given-names>E. L.</given-names>
</name>
<name>
<surname>Lovett</surname>
<given-names>J.&#x20;L.</given-names>
</name>
<name>
<surname>Yeo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Stanoszek</surname>
<given-names>L. M.</given-names>
</name>
<name>
<surname>Levin</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Targeted RNA-Sequencing with Competitive Multiplex-PCR Amplicon Libraries</article-title>. <source>PLoS One</source> <volume>8</volume> (<issue>11</issue>). <fpage>e79120</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0079120</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Carrara</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Beccuti</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Cavallo</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Donatelli</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lazzarato</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Cordero</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>State of Art Fusion-Finder Algorithms Are Suitable to Detect Transcription-Induced Chimeras in normal Tissues</article-title>. <source>BMC Bioinformatics</source> <volume>14 Suppl 7</volume> (<issue>Suppl. 7Suppl 7</issue>), <fpage>S2</fpage>&#x2013;<lpage>S</lpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-14-S7-S2</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Carrara</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Beccuti</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lazzarato</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Cavallo</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Cordero</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Donatelli</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>State-of-the-art Fusion-Finder Algorithms Sensitivity and Specificity</article-title>. <source>Biomed. Res. Int.</source> <volume>2013</volume>, <fpage>340620</fpage>. <pub-id pub-id-type="doi">10.1155/2013/340620</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Keoni</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Waker</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Lober</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.-H.</given-names>
</name>
<name>
<surname>Gutmann</surname>
<given-names>D. H.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>KIAA1549-BRAF Expression Establishes a Permissive Tumor Microenvironment through NF&#x3ba;B-Mediated CCL2 Production</article-title>. <source>Neoplasia</source> <volume>21</volume> (<issue>1</issue>), <fpage>52</fpage>&#x2013;<lpage>60</lpage>. <pub-id pub-id-type="doi">10.1016/j.neo.2018.11.007</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Fastp: an Ultra-fast All-In-One FASTQ Preprocessor</article-title>. <source>Bioinformatics</source> <volume>34</volume> (<issue>17</issue>), <fpage>i884</fpage>&#x2013;<lpage>i890</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty560</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Davidson</surname>
<given-names>N. M.</given-names>
</name>
<name>
<surname>Majewski</surname>
<given-names>I. J.</given-names>
</name>
<name>
<surname>Oshlack</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>JAFFA: High Sensitivity Transcriptome-Focused Fusion Gene Detection</article-title>. <source>Genome Med.</source> <volume>7</volume> (<issue>1</issue>), <fpage>43</fpage>. <pub-id pub-id-type="doi">10.1186/s13073-015-0167-x</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>de Myttenaere</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Golden</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Le Grand</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Rossi</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Mean Absolute Percentage Error for Regression Models</article-title>. <source>Neurocomputing</source> <volume>192</volume>, <fpage>38</fpage>&#x2013;<lpage>48</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2015.12.114</pub-id> </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Drilon</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Arcila</surname>
<given-names>M. E.</given-names>
</name>
<name>
<surname>Balasubramanian</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Greenbowe</surname>
<given-names>J.&#x20;R.</given-names>
</name>
<name>
<surname>Ross</surname>
<given-names>J.&#x20;S.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Broad, Hybrid Capture-Based Next-Generation Sequencing Identifies Actionable Genomic Alterations in Lung Adenocarcinomas Otherwise Negative for Such Alterations by Other Genomic Testing Approaches</article-title>. <source>Clin. Cancer Res.</source> <volume>21</volume> (<issue>16</issue>), <fpage>3631</fpage>&#x2013;<lpage>3639</lpage>. <pub-id pub-id-type="doi">10.1158/1078-0432.ccr-14-2683</pub-id> </citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Docker</surname>
<given-names>Merkel D.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Lightweight Linux Containers for Consistent Development and Deployment</article-title>. <source>Linux Journal</source> <volume>2014</volume> (<issue>239</issue>), <fpage>2</fpage>. </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Filippi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Depetris</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Satolli</surname>
<given-names>M. A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Evaluating Larotrectinib for the Treatment of Advanced Solid Tumors Harboring an NTRK Gene Fusion</article-title>. <source>Expert Opin. Pharmacother.</source> <volume>22</volume> (<issue>6</issue>), <fpage>677</fpage>&#x2013;<lpage>684</lpage>. <pub-id pub-id-type="doi">10.1080/14656566.2021.1876664</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gliem</surname>
<given-names>T. J.</given-names>
</name>
<name>
<surname>Aypar</surname>
<given-names>U.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Development of a Chromosomal Microarray Test for the Detection of Abnormalities in Formalin-Fixed, Paraffin-Embedded Products of Conception Specimens</article-title>. <source>J.&#x20;Mol. Diagn.</source> <volume>19</volume> (<issue>6</issue>), <fpage>843</fpage>&#x2013;<lpage>847</lpage>. <pub-id pub-id-type="doi">10.1016/j.jmoldx.2017.07.001</pub-id> </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Haas</surname>
<given-names>B. J.</given-names>
</name>
<name>
<surname>Dobin</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Stransky</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Pochet</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Regev</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Accuracy Assessment of Fusion Transcript Detection via Read-Mapping and De Novo Fusion Transcript Assembly-Based Methods</article-title>. <source>Genome Biol.</source> <volume>20</volume>, <fpage>213</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-019-1842-9</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Haynes</surname>
<given-names>B. C.</given-names>
</name>
<name>
<surname>Blidner</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Cardwell</surname>
<given-names>R. D.</given-names>
</name>
<name>
<surname>Zeigler</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gokul</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Thibert</surname>
<given-names>J.&#x20;R.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>An Integrated Next-Generation Sequencing System for Analyzing DNA Mutations, Gene Fusions, and RNA Expression in Lung Cancer</article-title>. <source>Translational Oncol.</source> <volume>12</volume> (<issue>6</issue>), <fpage>836</fpage>&#x2013;<lpage>845</lpage>. <pub-id pub-id-type="doi">10.1016/j.tranon.2019.02.012</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Madan</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>1999</year>). <article-title>CAP3: A DNA Sequence Assembly Program</article-title>. <source>Genome Res.</source> <volume>9</volume> (<issue>9</issue>), <fpage>868</fpage>&#x2013;<lpage>877</lpage>. <pub-id pub-id-type="doi">10.1101/gr.9.9.868</pub-id> </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jain</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Silva</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>H. J.</given-names>
</name>
<name>
<surname>Lang</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Boucher</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Overcoming Resistance to Single-Agent Therapy for Oncogenic BRAF Gene Fusions via Combinatorial Targeting of MAPK and PI3K/mTOR Signaling Pathways</article-title>. <source>Oncotarget</source> <volume>8</volume> (<issue>No 49</issue>), <fpage>84697</fpage>&#x2013;<lpage>84713</lpage>. <pub-id pub-id-type="doi">10.18632/oncotarget.20949</pub-id> </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kent</surname>
<given-names>W. J.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>BLAT---The BLAST-like Alignment Tool</article-title>. <source>Genome Res.</source> <volume>12</volume> (<issue>4</issue>), <fpage>656</fpage>&#x2013;<lpage>664</lpage>. <pub-id pub-id-type="doi">10.1101/gr.229202</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Salzberg</surname>
<given-names>S. L.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>TopHat-Fusion: an Algorithm for Discovery of Novel Fusion Transcripts</article-title>. <source>Genome Biol.</source> <volume>12</volume> (<issue>8</issue>), <fpage>R72</fpage>. <pub-id pub-id-type="doi">10.1186/gb-2011-12-8-r72</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>S.-I.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>S. K.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>H. J.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>S.-H.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Aggressive Supratentorial Ependymoma, RELA Fusion-Positive with Extracranial Metastasis: A Case Report</article-title>. <source>J.&#x20;Pathol. Transl Med.</source> <volume>51</volume> (<issue>6</issue>), <fpage>588</fpage>&#x2013;<lpage>593</lpage>. <pub-id pub-id-type="doi">10.4132/jptm.2017.08.10</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kivioja</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>V&#xe4;h&#xe4;rautio</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Karlsson</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Bonke</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Enge</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Linnarsson</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Counting Absolute Numbers of Molecules Using Unique Molecular Identifiers</article-title>. <source>Nat. Methods</source> <volume>9</volume> (<issue>1</issue>), <fpage>72</fpage>&#x2013;<lpage>74</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.1778</pub-id> </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Durbin</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Fast and Accurate Short Read Alignment with Burrows-Wheeler Transform</article-title>. <source>Bioinformatics</source> <volume>25</volume> (<issue>14</issue>), <fpage>1754</fpage>&#x2013;<lpage>1760</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp324</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Handsaker</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wysoker</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Fennell</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Ruan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Homer</surname>
<given-names>N.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>The Sequence Alignment/Map Format and SAMtools</article-title>. <source>Bioinformatics</source> <volume>25</volume> (<issue>16</issue>), <fpage>2078</fpage>&#x2013;<lpage>2079</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp352</pub-id> </citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tsai</surname>
<given-names>W.-H.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Huo</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Comprehensive Evaluation of Fusion Transcript Detection Algorithms and a Meta-Caller to Combine Top Performing Methods in Paired-End RNA-Seq Data</article-title>. <source>Nucleic Acids Res.</source> <volume>44</volume> (<issue>5</issue>), <fpage>e47</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkv1234</pub-id> </citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mitelman</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Johansson</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Mertens</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>The Impact of Translocations and Gene Fusions on Cancer Causation</article-title>. <source>Nat. Rev. Cancer</source> <volume>7</volume> (<issue>4</issue>), <fpage>233</fpage>&#x2013;<lpage>245</lpage>. <pub-id pub-id-type="doi">10.1038/nrc2091</pub-id> </citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moskalev</surname>
<given-names>E. A.</given-names>
</name>
<name>
<surname>Frohnauer</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Merkelbach-Bruse</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Schildhaus</surname>
<given-names>H.-U.</given-names>
</name>
<name>
<surname>Dimmler</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Schubert</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Sensitive and Specific Detection of EML4-ALK Rearrangements in Non-small Cell Lung Cancer (NSCLC) Specimens by Multiplex Amplicon RNA Massive Parallel Sequencing</article-title>. <source>Lung Cancer</source> <volume>84</volume> (<issue>3</issue>), <fpage>215</fpage>&#x2013;<lpage>221</lpage>. <pub-id pub-id-type="doi">10.1016/j.lungcan.2014.03.002</pub-id> </citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pekar-Zlotin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hirsch</surname>
<given-names>F. R.</given-names>
</name>
<name>
<surname>Soussan-Gutman</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ilouze</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Dvir</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Boyle</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Fluorescence <italic>In Situ</italic> Hybridization, Immunohistochemistry, and Next-Generation Sequencing for Detection of EML4-ALK Rearrangement in Lung Cancer</article-title>. <source>Oncologist</source> <volume>20</volume> (<issue>3</issue>), <fpage>316</fpage>&#x2013;<lpage>322</lpage>. <pub-id pub-id-type="doi">10.1634/theoncologist.2014-0389</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Robinson</surname>
<given-names>J.&#x20;T.</given-names>
</name>
<name>
<surname>Thorvaldsd&#xf3;ttir</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Winckler</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Guttman</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lander</surname>
<given-names>E. S.</given-names>
</name>
<name>
<surname>Getz</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Integrative Genomics Viewer</article-title>. <source>Nat. Biotechnol.</source> <volume>29</volume> (<issue>1</issue>), <fpage>24</fpage>&#x2013;<lpage>26</lpage>. <pub-id pub-id-type="doi">10.1038/nbt.1754</pub-id> </citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rutkowska</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Stoczy&#x144;ska-Fidelus</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Janik</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>W&#x142;odarczyk</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Rieske</surname>
<given-names>P</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>EGFRvIII: An Oncogene with Ambiguous Role</article-title>. <source>J.&#x20;Oncol.</source> <volume>2019</volume>, <fpage>1092587</fpage>. <pub-id pub-id-type="doi">10.1155/2019/1092587</pub-id> </citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shaw</surname>
<given-names>A. T.</given-names>
</name>
<name>
<surname>Solomon</surname>
<given-names>B. J.</given-names>
</name>
<name>
<surname>Chiari</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Riely</surname>
<given-names>G. J.</given-names>
</name>
<name>
<surname>Besse</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Soo</surname>
<given-names>R. A.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Lorlatinib in Advanced ROS1-Positive Non-small-cell Lung Cancer: a Multicentre, Open-Label, Single-Arm, Phase 1&#x2013;2 Trial</article-title>. <source>Lancet Oncol.</source> <volume>20</volume> (<issue>12</issue>)&#x2013;<lpage>701</lpage>. </citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sze</surname>
<given-names>S. H.</given-names>
</name>
<name>
<surname>Pimsler</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>Tomberlin</surname>
<given-names>J.&#x20;K.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>C. D.</given-names>
</name>
<name>
<surname>Tarone</surname>
<given-names>A. M.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>A Scalable and Memory-Efficient Algorithm for De Novo Transcriptome Assembly of Non-model Organisms</article-title>. <source>BMC Genomics</source> <volume>18</volume> (<issue>4</issue>), <fpage>387</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-017-3735-1</pub-id> </citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thilly</surname>
<given-names>R. S. Ca. W. G.</given-names>
</name>
</person-group> (<year>1993</year>). <article-title>Specificity, Efficiency, and Fidelity of PCR</article-title>. <source>Genome Res.</source> <volume>3</volume>, <fpage>S18</fpage>&#x2013;<lpage>S29</lpage>. </citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tofallis</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>A Better Measure of Relative Prediction Accuracy for Model Selection and Model Estimation</article-title>. <source>J.&#x20;Oper. Res. Soc.</source> <volume>66</volume> (<issue>8</issue>), <fpage>1352</fpage>&#x2013;<lpage>1362</lpage>. <pub-id pub-id-type="doi">10.1057/jors.2014.103</pub-id> </citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Voss</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Full-stack Genomics Pipelining with GATK4 &#x2b; WDL &#x2b; Cromwell</article-title>. <source>F1000Research</source>. <volume>6</volume> (<issue>1381</issue>). </citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vu</surname>
<given-names>T. N.</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Trac</surname>
<given-names>Q. T.</given-names>
</name>
<name>
<surname>Calza</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hwang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Pawitan</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>A Fast Detection of Fusion Genes from Paired-End RNA-Seq Data</article-title>. <source>BMC Genomics</source> <volume>19</volume> (<issue>1</issue>), <fpage>786</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-018-5156-1</pub-id> </citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Gerstein</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Snyder</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>RNA-seq: a Revolutionary Tool for Transcriptomics</article-title>. <source>Nat. Rev. Genet.</source> <volume>10</volume> (<issue>1</issue>), <fpage>57</fpage>&#x2013;<lpage>63</lpage>. <pub-id pub-id-type="doi">10.1038/nrg2484</pub-id> </citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wilcoxon</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>1945</year>). <article-title>Individual Comparisons by Ranking Methods</article-title>. <source>Biometrics Bull.</source> <volume>1</volume> (<issue>6</issue>), <fpage>80</fpage>&#x2013;<lpage>83</lpage>. <pub-id pub-id-type="doi">10.2307/3001968</pub-id> </citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wong</surname>
<given-names>R. K. Y.</given-names>
</name>
<name>
<surname>MacMahon</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Woodside</surname>
<given-names>J.&#x20;V.</given-names>
</name>
<name>
<surname>Simpson</surname>
<given-names>D. A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A Comparison of RNA Extraction and Sequencing Protocols for Detection of Small RNAs in Plasma</article-title>. <source>BMC Genomics</source> <volume>20</volume> (<issue>1</issue>), <fpage>446</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-019-5826-7</pub-id> </citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yates</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Braschi</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Gray</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Seal</surname>
<given-names>R. L.</given-names>
</name>
<name>
<surname>Tweedie</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bruford</surname>
<given-names>E. A.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Genenames.org: the HGNC and VGNC Resources in 2017</article-title>.&#x20;<source>Nucleic Acids Res.</source> <volume>45</volume> (<issue>D1</issue>), <fpage>D619</fpage>&#x2013;<lpage>D625</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkw1033</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>