<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1499456</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2025.1499456</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Validation of a comprehensive long-read sequencing platform for broad clinical genetic diagnosis</article-title>
<alt-title alt-title-type="left-running-head">Sen et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2025.1499456">10.3389/fgene.2025.1499456</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes" equal-contrib="yes">
<name>
<surname>Sen</surname>
<given-names>Siddhartha</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2848664/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Handler</surname>
<given-names>Hillary P.</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2990772/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Victorsen</surname>
<given-names>Alec</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2928170/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Flaten</surname>
<given-names>Zach</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ellison</surname>
<given-names>Aidan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Knutson</surname>
<given-names>Todd P.</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2928158/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Munro</surname>
<given-names>Sarah A.</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Martinez</surname>
<given-names>Ryan J.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Billington</surname>
<given-names>Charles John</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/35802/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Laffin</surname>
<given-names>Jennifer J.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2896142/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Bray</surname>
<given-names>Sarah</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Mroz</surname>
<given-names>Pawel</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1176638/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yohe</surname>
<given-names>Sophia</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1555759/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Nelson</surname>
<given-names>Andrew C.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Bower</surname>
<given-names>Matthew</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn002">
<sup>&#x2021;</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes" equal-contrib="yes">
<name>
<surname>Thyagarajan</surname>
<given-names>Bharat</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<xref ref-type="author-notes" rid="fn002">
<sup>&#x2021;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/26973/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>University of Minnesota Health Sciences</institution>, <institution>University of Minnesota Medical Center</institution>, <addr-line>Minneapolis</addr-line>, <addr-line>MN</addr-line>, <country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Molecular Diagnostics Laboratory</institution>, <institution>Fairview Health</institution>, <institution>University of Minnesota Medical Center</institution>, <addr-line>Minneapolis</addr-line>, <addr-line>MN</addr-line>, <country>United States</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Minnesota Supercomputing Institute</institution>, <institution>University of Minnesota</institution>, <addr-line>Minneapolis</addr-line>, <addr-line>MN</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/44701/overview">Jared C. Roach</ext-link>, Institute for Systems Biology (ISB), United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/192027/overview">Romina D&#x2019;Aurizio</ext-link>, National Research Council (CNR), Italy</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2886702/overview">Danny Miller</ext-link>, University of Washington, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Siddhartha Sen, <email>sen00037@umn.edu</email>; Bharat Thyagarajan, <email>thya0003@umn.edu</email>
</corresp>
<fn fn-type="equal" id="fn001">
<label>
<sup>&#x2020;</sup>
</label>
<p>These authors have contributed equally to this work and share first authorship</p>
</fn>
<fn fn-type="equal" id="fn002">
<label>
<sup>&#x2021;</sup>
</label>
<p>These authors have contributed equally to this work and share last authorship</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>02</day>
<month>05</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1499456</elocation-id>
<history>
<date date-type="received">
<day>20</day>
<month>09</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>15</day>
<month>04</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Sen, Handler, Victorsen, Flaten, Ellison, Knutson, Munro, Martinez, Billington, Laffin, Bray, Mroz, Yohe, Nelson, Bower and Thyagarajan.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Sen, Handler, Victorsen, Flaten, Ellison, Knutson, Munro, Martinez, Billington, Laffin, Bray, Mroz, Yohe, Nelson, Bower and Thyagarajan</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Though short read high-throughput sequencing, commonly known as Next-Generation Sequencing (NGS), has revolutionized genomics and genetic testing, there is no single genetic test that can accurately detect single nucleotide variants (SNVs), small insertions/deletions (indels), complex structural variants (SVs), repetitive genomic alterations, and variants in genes with highly homologous pseudogenes. The implementation of a unified comprehensive technique that can simultaneously detect a broad spectrum of genetic variation would substantially increase efficiency of the diagnostic process. The current study evaluated the clinical utility of long-read sequencing as a comprehensive genetic test for diagnosis of inherited conditions. Using Oxford Nanopore Technologies long read nanopore sequencing, we successfully developed and validated a clinically deployable integrated bioinformatics pipeline that utilizes a combination of eight publicly available variant callers. A concordance assessment comparing the known variant calls from a well-characterized, benchmarked sample called NA12878 from the National Institute of Standards and Technology (NIST) with the variants detected by our pipeline for this sample, determined that the analytical sensitivity of our pipeline was 98.87% and the analytical specificity exceeded 99.99%. We then evaluated our pipeline&#x2019;s ability to detect 167 clinically relevant variants from 72 clinical samples. This set of variants consisted of 80 SNVs, 26 indels, 32 SVs, and 29 repeat expansions, including 14 variants in genes with highly homologous pseudogenes. The overall detection concordance for these clinically relevant variants was 99.4% (95% CI: 99.7%&#x2013;99.9%). Importantly, in addition to detecting known clinically relevant variants, in four cases, our pipeline yielded valuable additional information in support of clinical diagnoses that could not have been established using short-read NGS alone. Our findings suggest that long-read sequencing is successful in identifying diverse genomic alterations and that our pipeline functions well as the basis for a single diagnostic test for patients with suspected genetic disease.</p>
</abstract>
<kwd-group>
<kwd>long-read sequencing</kwd>
<kwd>Oxford Nanopore Technologies</kwd>
<kwd>clinical genomics</kwd>
<kwd>whole genome sequencing</kwd>
<kwd>Tandem repeat expansions</kwd>
<kwd>complex structural variants</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Human and Medical Genomics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>Massively parallel NGS techniques have revolutionized the molecular diagnosis of rare genetic conditions (<xref ref-type="bibr" rid="B9">Fernandez-Marmiesse et al., 2018</xref>) (PMID: 28721829). Specifically, hybridization-based target enrichment in conjunction with short-read sequencing has emerged as a leading molecular diagnostic technique in clinical genetic laboratories (<xref ref-type="bibr" rid="B32">Singh, 2022</xref>) (PMID: 35885445). However, the length of short reads creates several limitations that include mapping ambiguity of highly repetitive and/or &#x2018;GC&#x2019; rich genome regions as well as limited the ability to accurately sequence large complex SVs (<xref ref-type="bibr" rid="B38">Zavodna et al., 2014</xref>) (PMID: 25436869). Patients with rare disorders caused by complex variants typically undergo multiple discrete rounds of genetic testing. This can lead to delays in diagnosis and a significant financial burden for patients. Long-read sequencing methodology capable of sequencing DNA fragments that are tens of thousands of nucleotides in length is an attractive technology that can address the limitations of short-read technology and was named technology of the year in 2022 (<xref ref-type="bibr" rid="B18">Marx, 2023</xref>) (PMID: 36635542). Two companies, Oxford Nanopore Technologies and Pacific Biosystems (PacBio), have pioneered the commercialization of long-read technologies. When first introduced, nanopore long-read sequencing showed higher error rates, excluding it as a candidate technology capable of replacing short read-based sequencing for clinical diagnostics (<xref ref-type="bibr" rid="B17">Lu et al., 2016</xref>) (PMID: 27646134). However, with recent advances in chemistry, flow cell technology (R10), and base-calling algorithms, modal read accuracy has greatly improved and SNV F1 scores are over 98% (<xref ref-type="bibr" rid="B21">Ni et al., 2023</xref>; <xref ref-type="bibr" rid="B16">Liu et al., 2021</xref>; <xref ref-type="bibr" rid="B33">Srivathsan et al., 2024</xref>) (PMID: 37025654, 33612390, 38041646).</p>
<p>While there are several reports demonstrating the advantages of long-read sequencing technologies for detection of CNVs, SVs, and variants in genomic regions that have historically been difficult to sequence using short-read technologies, adoption of long-read sequencing into clinical practice has been limited (<xref ref-type="bibr" rid="B38">Zavodna et al., 2014</xref>) (PMID: 25436869). In a pilot study, researchers successfully used a nanopore-based workflow to perform ultra-rapid whole genome sequencing on critically ill patients, resulting in diagnoses of rare genetic diseases in approximately 8&#xa0;hours (<xref ref-type="bibr" rid="B10">Gorzynski et al., 2022</xref>) (PMID: 35020984). In another recent study, a pathogenic SV was identified in a patient with multiple neoplasia and cardiac myxomata using long-read sequencing, in whom previous targeted short-read sequencing was negative, thereby resulting in a clinical diagnosis (<xref ref-type="bibr" rid="B19">Merker et al., 2018</xref>) (PMID: 28640241). Another pilot study demonstrated the usefulness of ONT-based targeted long-read sequencing for the molecular testing and diagnosis of short tandem repeat (STR) expansion disorders (<xref ref-type="bibr" rid="B34">Stevanovski et al., 2022</xref>) (PMID: 35245110). Thus, ONT has now become a high-throughput, high-fidelity long-read sequencing technology which can be used for rapid diagnosis in a clinical setting. While the high cost of long-read sequencing has been a barrier to clinical implementation, the costs of these technologies (both ONT and PacBio) have been decreasing (<xref ref-type="bibr" rid="B13">Hook and Timp, 2023</xref>) (PMID: 37161088). Another major hurdle to the wide-spread adoption of long-read sequencing into clinical testing is the lack of integrated workflows allowing comprehensive detection of different types of genetic variants, which is essential for establishing a robust clinical assay for inherited disorders. There are currently numerous bioinformatics tools that are specifically employed to identify particular types of genetic variants (e.g., SNVs, CNVs, repeat expansions, etc.) using Nanopore long reads.</p>
<p>We previously developed and successfully integrated a short-read sequencing and a copy number variation (CNV) detection pipeline into a broad-based NGS platform for clinical testing to meet the genetic testing needs in the Molecular Diagnostics Laboratory (MDL) at the University of Minnesota (UMN) (<xref ref-type="bibr" rid="B23">Onsongo et al., 2016</xref>; <xref ref-type="bibr" rid="B12">Hartman et al., 2019</xref>) (PMID: 27597741, 30891420). A current unmet need shared by MDL and other institutions is the implementation of a single comprehensive genetic test that can accurately detect SNVs, small indels, complex SVs, repetitive genomic alterations, and variants in genes with highly homologous pseudogenes. Neurology is one clinical area where this need is evident, particularly with respect to the diagnosis of hereditary cerebellar ataxias. Patients with ataxia often incur multiple rounds of genetic testing and significant financial burden before receiving a diagnosis, if they receive a clear diagnosis at all. Given the relatively low diagnostic rates and long diagnostic odysseys experienced by this patient population, a recent review proposed that the theoretical best approach to genetic diagnosis of hereditary cerebellar ataxia would be to implement a long-read sequencing platform that could supersede a sequential testing approach altogether in all cases where a monogenic cause is not highly suspected (<xref ref-type="bibr" rid="B30">Rudaks et al., 2024</xref>) (PMID: 38760634). The goal of this study is to address this unmet need for patients with diverse genetic conditions caused by a wide range of genetic mutations. Here, we describe the development of a comprehensive genetic testing platform that leverages the advantages of long-read sequencing and uses several variant callers for a complete analysis of the full spectrum of disease-causing genomic alterations for clinical diagnostics.</p>
</sec>
<sec sec-type="methods" id="s2">
<title>Methods</title>
<sec id="s2-1">
<title>Concordance studies</title>
<p>A benchmarked sample, NA12878/HG001 (v4.2.1), was purchased from NIST (<ext-link ext-link-type="uri" xlink:href="https://www.nist.gov/programs-projects/genome-bottle">https://www.nist.gov/programs-projects/genome-bottle</ext-link>). The NA12878 sample represents a single female genome that has been extensively characterized using multiple genomic analysis platforms (<xref ref-type="bibr" rid="B40">Zook et al., 2016</xref>) (PMID: 27271295). Using the custom workflow described below, NA12878 was prepared and sequenced on an Oxford Nanopore Technologies PromethION-24 at the UMN Advanced Research and Diagnostics Laboratory (ARDL). A concordance analysis was performed to determine the ability of our ONT sequencing pipeline to detect the well-characterized SNV and indel variants present in the NA12878 sample. This concordance analysis was restricted to exonic variants (n &#x3d; 26,584) in annotated human genes associated with clinical phenotypes. This restriction was performed by intersecting the NA12878 VCF file with a BED file (DOI: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.5281/zenodo.14532101">10.5281/zenodo.14532101</ext-link>) containing exonic coordinates for 5631 clinically relevant genes (<xref ref-type="sec" rid="s11">Supplementary Table S1</xref>) using BEDTools (<xref ref-type="bibr" rid="B27">Quinlan and Hall, 2010</xref>) (PMID: 20110278). This list of 5631 clinically relevant genes is regularly updated and it is curated using data from publicly available databases such as OMIM as well as new publications in the relevant medical literature identified via periodic targeted PubMed searches. Concordance analysis was performed by comparing VCF files and identifying exact matches at the chromosome (CHROM), position (POS), reference (REF), and alternate (ALT) columns using R (v4.0.4).</p>
</sec>
<sec id="s2-2">
<title>Clinical sample selection and storage</title>
<p>Seventy-two clinical samples previously analyzed at the UMN MDL were selected for nanopore sequencing (<xref ref-type="sec" rid="s11">Supplementary Table S2</xref>). These samples were chosen as a representation of diverse genetic variation. Each sample had at least one clinically relevant abnormal repeat expansion, SNV, indel, SV, or variant in a gene with a highly homologous pseudogene. The available specimens were either extracted DNA or buffy coat stored at &#x2212;80&#xb0;C.</p>
</sec>
<sec id="s2-3">
<title>Sample preparation</title>
<p>For 56 samples, DNA was purified from buffy coats on an Autogen Flexstar; 16 samples had been previously extracted [Qiagen DNeasy Blood &#x26; Tissue Kit (Cat No. 69506)]. Extracted DNA was concentrated using an Eppendorf Vacufuge plus at room temperature. In most cases, 4&#xa0;&#xb5;g of DNA was diluted into 150&#xa0;&#xb5;L water and sheared by centrifugation in Covaris g-TUBEs for 30&#xa0;s at 1,250&#xa0;g. Sheared DNA was characterized using an Invitrogen Qubit using the 1x dsDNA BR assay, and on an Agilent Tapestation. Ideally, samples had approximately 80% of the sheared fragments between 8&#xa0;kb and 48.5&#xa0;kb in length. Though Tapestation fragment size did not directly correlate with N50, the Tapestation fragment size was used as a quality check to estimate optimal levels of DNA shearing prior to sequencing. Samples were prepared for sequencing using the Oxford Nanopore Ligation Sequencing kit V14 using 3&#xa0;&#xb5;g of sheared DNA.</p>
</sec>
<sec id="s2-4">
<title>Whole genome sequencing</title>
<p>Samples were sequenced on a PromethION-24 and run on a single flow cell (R10.4.1 with the E8.2 motor protein) for approximately 5&#xa0;days, with daily washing and reloading. For each sample, one library was built and split into four to five &#x223c;300&#xa0;ng aliquots. One flow cell was used to sequence four library aliquots. Each aliquot was sequenced overnight. The following day, the flow cell was washed for at least an hour before a new aliquot was added. Each flow cell was run for a total of 100&#xa0;h. No adaptive sequencing was used. Several versions of MinKNOW were used depending on when the samples were sequenced. The versions used were 22.12.5, 23.04.6, 23.07.8, 23.07.12, 23.11.4, and 23.11.7.</p>
</sec>
<sec id="s2-5">
<title>Bioinformatics analysis</title>
<p>Data was processed at the Minnesota Supercomputing Institute. Default parameters were used for all software unless otherwise noted. POD5 files were processed using Dorado(v0.3.4&#x2b;5f5cd02&#x2b;cu118) with a minimum q-score of 9. The Dorado basecalling model used was <email>dna_r10.4.1_e8.2_400bps_sup@v4.2.0</email>. Methylation analysis was not performed in this study. Unaligned BAMs were aligned to the hg19 genome using Nanopore&#x2019;s epi2me wf-alignment workflow(v0.3.3). Sequencing depth was calculated using Mosdepth(v0.3.3) (<xref ref-type="bibr" rid="B25">Pedersen and Quinlan, 2018</xref>) (PMID: 29096012) and aligned N50s were calculated using Cramino(v0.9.9) (<xref ref-type="bibr" rid="B7">De Coster and Rademakers, 2023</xref>) (PMID: 37171891). Read quality was assessed with NanoPlot(v1.44.0) (<xref ref-type="bibr" rid="B7">De Coster and Rademakers, 2023</xref>) (PMID: 37171891).</p>
</sec>
<sec id="s2-6">
<title>Variant calling pipeline</title>
<p>SNVs were identified with Clair3 as the default genotyper using the epi2me wf-human-variation workflow (v1.2.0) (<xref ref-type="bibr" rid="B39">Zheng et al., 2022</xref>) (PMID: 38177392). Insertions/deletions/inversions (INS/DEL/INV) were identified using NanoVar(v1.5.1) (<xref ref-type="bibr" rid="B37">Tham et al., 2020</xref>) (PMID: 32127024) and DeBreak (v1.0.2) (<xref ref-type="bibr" rid="B4">Chen Y. et al., 2023</xref>) (PMID: 36650186). SVs identified by DeBreak were limited to a minimum of 100 bp in length while NanoVar was used to identify SVs &#x3e; 50&#xa0;bp. INS/DEL &#x3c; 50&#xa0;bp were identified using the Clair3 genotyper. CNVs on the sex chromosomes were identified using CNVpytor(1.3.1) (<xref ref-type="bibr" rid="B35">Suvakov et al., 2021</xref>) (PMID: 34817058) with 1&#xa0;kbp bins, which were also filtered by: Q0 &#x3c; 0.5, p_N &#x3c; 0.5, and p-values &#x3c; 0.001; choosing these parameters allowed for the highest quality calls for review. Autosomal CNVs were called by QDNAseq (1&#xa0;kbp bin size), which is included in the epi2me workflow (v1.10.1).</p>
<p>A structured filtering approach was developed to accurately identify clinically relevant SVs (<xref ref-type="fig" rid="F1">Figure 1</xref>). In the first filtering step, each of the caller&#x2019;s outputs were restricted to a subset of calls related to its variant calling strengths. The next filtering step involved narrowing the list of calls to variant types that could mechanistically cause Mendelian disease. For example, an inversion impacting a subset of coding exons within a gene would advance to the next filtering step, but a complete gene inversion with intergenic breakpoints would be excluded. Another filtering step was then performed to restrict the calls to genes that have a human phenotype. This list of 5,631 genes associated with clinically relevant human phenotypes is regularly updated and it is curated using data from publicly available databases such as OMIM as well as new publications in the relevant medical literature identified via periodic targeted PubMed searches (<xref ref-type="sec" rid="s11">Supplementary Table S1</xref>). A final set of logic rules was applied using a BEDtools/BCFtools intersect (<xref ref-type="bibr" rid="B27">Quinlan and Hall, 2010</xref>; <xref ref-type="bibr" rid="B6">Danecek et al., 2021</xref>) (PMID: 20112078, 33590861) as the last step to accurately identify and detect a pathogenic SV.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Filtering strategy employed in the detection of clinically relevant structural variants. Flow chart outlining the ONT pipeline&#x2019;s SV call filtering strategy. A combination of breakpoint-based callers (NanoVar and DeBreak) as well as read-depth callers (QDNAseq and CNVpytor) was used to identify structural variants (SVs). Before any filters were applied, there was an average of approximately 40,000 hits for small SVs detected by NanoVar and approximately 14,000 by DeBreak. For the large SVs and CNVs, there was an average of 192 hits identified by QDNAseq and 1803 by CNVpytor, respectively. When filtered for autosomes and sex chromosomes, the average numbers of large SVs/CNVs were 56 for autosomes and 162 for sex chromosomes, respectively. The next filtering step involved restricting the SV calls to genes that have a human phenotype, which again diminished the total number of SVs. The last filtering step involved application of a set of logic rules resulting in a further reduction in the number of SVs. The SVs that remained at the end of all the filtering steps were subjected to a final review and classified.</p>
</caption>
<graphic xlink:href="fgene-16-1499456-g001.tif"/>
</fig>
<p>Samples were processed with Paraphase (v3.1.1) (<xref ref-type="bibr" rid="B3">Chen X. et al., 2023</xref>) (PMID: 36669496) for assessment of variants in genes like <italic>PMS2</italic> and <italic>STRC</italic>, that are known to have homologous pseudogenes. Tandem repeat expansions were called using Tandem Genotypes (v1.9.1) (<xref ref-type="bibr" rid="B20">Mitsuhashi et al., 2019</xref>) (PMID: 30890163) and Sniffles2 (v2.0.7) (<xref ref-type="bibr" rid="B31">Sedlazeck et al., 2018</xref>) (PMID: 29713083). For Tandem Genotypes, reads surrounding loci of interest were realigned using LAST (v1256) (<xref ref-type="bibr" rid="B15">Kielbasa et al., 2011</xref>) (PMID: 21209072). A schematic representation of the pipeline is shown in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Custom nanopore variant calling pipeline. The boxes depict file names and file types. The connecting lines between boxes are labeled with the pipeline components that require and generate respective input and output file types. The pipeline outputs multiple unique variant call files (VCFs) that will be combined for clinical analysis.</p>
</caption>
<graphic xlink:href="fgene-16-1499456-g002.tif"/>
</fig>
<p>All clinical analyses and reporting infrastructure in our laboratory currently use hg19 as human reference. In order to maintain consistency and ensure that results can be accurately compared to our current clinical pipelines, sequencing data generated by our ONT-based pipeline was also analyzed using hg19 as the reference genome.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<sec id="s3-1">
<title>NA12878 concordance</title>
<p>Since SNVs and small indels are the most common clinically relevant genomic alterations, we compared how the ONT-based pipeline performed when compared to our current short-read based clinical pipeline. Our ONT-based long-read pipeline was able to correctly identify 26,210 of the 26,584 gold-standard NA12878 variants across exons of known annotated coding genes, which corresponds to an analytical sensitivity of 98.87% and an analytical specificity exceeding 99.99%. In comparison, when our current clinically validated short-read pipeline was compared with the NA12878 sample, it showed a comparable analytical sensitivity of 99.59% and a specificity exceeding 99.99% on the same set of variants (<xref ref-type="table" rid="T1">Table 1</xref>).</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Clinically significant variant concordance by variant type.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Variant type</th>
<th align="left">Concordance</th>
<th align="left">CI (95%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">SNV</td>
<td align="left">1.0000</td>
<td align="left">0.95&#x2013;1.0</td>
</tr>
<tr>
<td align="left">Indel</td>
<td align="left">0.9615</td>
<td align="left">0.811&#x2013;0.9932</td>
</tr>
<tr>
<td align="left">SV</td>
<td align="left">1.0000</td>
<td align="left">0.8928&#x2013;1.0</td>
</tr>
<tr>
<td align="left">Repeat</td>
<td align="left">1.0000</td>
<td align="left">0.883&#x2013;1.0</td>
</tr>
<tr>
<td align="left">Overall</td>
<td align="left">0.9940</td>
<td align="left">0.9669&#x2013;0.9989</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>To assess and compare the analytical sensitivity and specificity for detection of SNVs and indels, the set of 26,584 gold-standard NA12878 variants was analyzed separately based on variant type. For the 25,514 SNVs in this dataset, the analytical sensitivity was 98.91% and the analytical specificity exceeded 99.99%. For the 1,070 indel calls in this dataset, the analytical sensitivity is 97.81% and the analytical specificity exceeds 99.99% (<xref ref-type="table" rid="T1">Table 1</xref>).</p>
<p>While the National Institute of Standards and Technology (NIST) does have available samples with well-characterized SVs, these samples were not used to assess the analytical detection abilities of our pipeline. The available samples from healthy individuals do not have enough CNV or SV calls in coding regions of clinically relevant genes to accurately evaluate the analytical sensitivity and specificity of our pipeline.</p>
</sec>
<sec id="s3-2">
<title>Evaluation of variant detection tools</title>
<p>Prior to testing our pipeline&#x2019;s ability to detect a diverse spectrum of clinically relevant genetic variation, we assessed the functionality of several variant detection tools on a pilot set of clinical samples. Clair3 accurately identified all SNVs and small indels in the pilot dataset. Therefore, Clair3 was included in our pipeline as the default genotyper for small genetic alterations.</p>
<p>With respect to detecting SVs, we systematically evaluated the functionality of several available breakpoint callers (<xref ref-type="sec" rid="s11">Supplementary Table S3</xref>). In our initial assessment of SV breakpoint callers, we evaluated eight clinical samples with known complex SVs, including a combination of deletions, duplications, inversions, and/or translocations. Across the eight samples with complex SVs, there were 29 known breakpoints. Of the tools evaluated, NanoVar was the most sensitive tool for breakpoint identification, with a detection rate of 93.1%. While DeBreak&#x2019;s SV detection rate (86.2%) was lower than NanoVar&#x2019;s, DeBreak was superior in characterizing the type of rearrangement occurring at a particular breakpoint rather than assigning it a generic classification such as &#x201c;breakpoint not determined.&#x201d; In contrast, SVIM and Sniffles2 displayed significantly inferior SV detection performance, with SV breakpoint detection rates of 58.6% and 6.9%, respectively. Therefore, based on the performance of the callers tested within our clinical sequencing environment (non-targeted capture with a target minimum genome-wide coverage of 30x), DeBreak was selected as the pipeline&#x2019;s primary SV caller (for SVs &#x3e;100 bases), with Nanovar implemented specifically to identify SV calls between 50-100 bases in size.</p>
<p>For detection of primarily larger SVs and CNVs that are not mediated by detectible breakpoints, we assessed two read-depth based callers, QDNAseq and CNVpytor. We found that QDNAseq performed best for autosomes, while CNVpytor performed best for sex chromosomes.</p>
<p>As clinically significant repeat expansion variants are a unique subset of SV that are notoriously difficult to accurately characterize by short-read NGS, we also assessed the functionality of a tool specifically designed to detect and quantify repeat expansion content. The Tandem Genotypes tool performed well for all loci assessed except for the pentanucleotide repeat expansions in <italic>RFC1</italic> that cause cerebellar ataxia, neuropathy, vestibular areflexia syndrome (CANVAS). In search of a bioinformatics tool capable of identifying clinically relevant variants specifically in <italic>RFC1</italic>, we begin by reevaluating the previously discarded SV callers. We found that Sniffles2 worked particularly well for characterizing <italic>RFC1</italic> repeat expansions. Therefore, this tool was added to the pipeline specifically for this purpose.</p>
<p>While several components of the pipeline such as Clair3, CNVpytor, and NanoVar were able to identify variants in some genes with highly homologous pseudogenes, they were not successful in certain genes such as <italic>PMS2 or STRC</italic>. As such, particular components of the Paraphase package were assessed to determine the functionality of this tool in parsing out haplotypes for genes like <italic>PMS2 and STRC</italic>. Since Paraphrase successfully distinguished between variants in highly homologous regions of genes such as <italic>PMS2</italic> from pseudogenes such as <italic>PMS2CL</italic>, this tool was included in our pipeline with the intention of employing it only for particular loci such as <italic>PMS2 and STRC</italic> where other pipeline components were unsuccessful.</p>
</sec>
<sec id="s3-3">
<title>Spectrum of known genomic alterations in clinical samples</title>
<p>All analyzed clinical samples had a genome-wide coverage of approximately 30X; the median N50 was 10,294&#xa0;bp (Range &#x3d; 970&#x2013;15,855 bases). The sequencing metrics are shown in <xref ref-type="fig" rid="F3">Figure 3</xref>. We assessed the performance of long-read sequencing on a cohort of 72 samples from our clinical laboratory. This sample set contained a total of 167 clinically relevant variants that were assessed in this study (<xref ref-type="sec" rid="s11">Supplementary Table S2</xref>; <xref ref-type="fig" rid="F4">Figure 4</xref>). This variant pool consisted of 80 SNVs, 26 indels, 32 SVs, and 29 repeat expansions, which included 14 variants in genes with highly homologous pseudogenes.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Nanopore sequencing metrics. Violin plots depicting the variability in <bold>(A)</bold> coverage, <bold>(B)</bold> average read quality, <bold>(C)</bold> average read length, and <bold>(D)</bold> N50 for the 72 clinical samples sequenced. For each plot, each dot represents one clinical sample. The dashed black horizontal lines represent the median value for each metric. The dashed gray line on the coverage plot shows the <italic>a priori</italic> target minimum coverage value (30x) for samples in this study.</p>
</caption>
<graphic xlink:href="fgene-16-1499456-g003.tif"/>
</fig>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Variants detected using ONT pipeline from 72 clinical samples. Variants assessed using ONT were grouped into four categories: single nucleotide variants (SNV), indels, structural variants (SV) and repeats. For each variant category, the number of variants accurately detected by our pipeline are shown in blue and the variant number is listed at the bottom of each bar. For each variant category, the number of variants that were not accurately detected by our pipeline are shown in red and the variant number is listed above each bar.</p>
</caption>
<graphic xlink:href="fgene-16-1499456-g004.tif"/>
</fig>
</sec>
<sec id="s3-4">
<title>Detection of clinically relevant SNVs</title>
<p>A total of 80 clinically relevant SNVs were assessed by our ONT-based pipeline, all of which were present at a variant allele frequency (VAF) consistent with germline variation (VAF of &#x223c;50% for heterozygous calls and 100% for homozygous calls). All 80 SNVs were accurately identified by Clair3 (<xref ref-type="fig" rid="F2">Figure 2</xref>), indicating that our pipeline has an SNV detection concordance for germline variants of 100% (95% CI: 95%&#x2013;100%) (<xref ref-type="fig" rid="F4">Figure 4</xref>; <xref ref-type="table" rid="T1">Table 1</xref>).</p>
</sec>
<sec id="s3-5">
<title>Detection of clinically relevant indels</title>
<p>We assessed 26 clinically significant indels using our ONT-based pipeline (<xref ref-type="fig" rid="F4">Figure 4</xref>). The indel sample set consisted of insertions, deletions, and delins variants ranging from one to 36 bases in size. Clair3 accurately detected 25 of the 26 indels, including seven that fell within homopolymer stretches (4&#x2013;8 bases in length) and three in genes with highly homologous pseudogenes.</p>
<p>Several examples of specific indels assessed using our pipeline are shown in <xref ref-type="fig" rid="F5">Figure 5</xref>. <xref ref-type="fig" rid="F5">Figure 5A</xref> depicts a complex indel in <italic>ABCD1</italic> that was detected and accurately resolved by our pipeline. The only indel variant that was not detected by our pipeline was a complex indel in <italic>TRHR</italic> (<xref ref-type="fig" rid="F5">Figure 5B</xref>). This variant, c.1137_1152delinsTTTTGTGGCAGGTGCTTGGCTGCCTGCCACAGGCAA, includes a loss of 16 bases and a gain of 36 bases. This <italic>TRHR</italic> variant was the largest independent indel variant assessed in this study and it was not called by Clair3. The next-largest indel variant assessed was <italic>BRCA1</italic> c.2472_2490delinsGTAT, which resulted in a loss of 19 bases and a gain of four bases and was accurately called by Clair3.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>IGV images of indels used to assess functionality of variant callers for indels of different sizes. <bold>(A)</bold> shows the <italic>ABCD1</italic> c.1635-16_1645delinsCACAGACATGTAGGGC variant, which results in a loss of 26 bases and a gain of 16 bases. This variant was accurately detected by the Clair3 component of the pipeline. The blue bars above the coverage data indicate the variants present in the Clair3 VCF. <bold>(B)</bold> shows the <italic>TRHR</italic> c.1137_1152delinsTTTTGTGGCAGGTGCTTGGCTGCCTGCCACAGGCAA variant, which results in a loss of 16 bases and a gain of 36 bases. This was the largest independent indel variant assessed in this study. This indel was not detected by Clair3 nor SV callers. <bold>(C)</bold> shows a complex recombinant allele at the 3&#x2032; end of <italic>GBA</italic>. The recombinant allele includes several SNVs as well as a 55-base deletion, representing a gene conversion to pseudogene (<italic>GBAP1</italic>) sequence. The black arrow below the <italic>GBA</italic> sequence at the bottom of the image indicates the approximate junction between reference <italic>GBA</italic> sequence and pseudogene sequence. The genomic region to the right (upstream) of the black arrow is <italic>GBA</italic> sequence. The region to the left (downstream) of the black arrow is the region of gene conversion. The two rows of blue boxes above the coverage data indicate the SNVs called by Clair3 (upper row) and the 55-base deletion called by NanoVar.</p>
</caption>
<graphic xlink:href="fgene-16-1499456-g005.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="F5">Figure 5C</xref> shows a complex structural variant in <italic>GBA</italic> that contains a 55-base deletion. This <italic>GBA</italic> deletion is part of a complex recombinant allele that also includes several SNVs. As such, the recombinant allele was considered a SV in this study and was not included as an independent indel variant for data analysis. Notably, while Clair3 did not detect the 55-base <italic>GBA</italic> deletion, it was accurately detected by the breakpoint-based caller, NanoVar.</p>
<p>Since NanoVar accurately detected the 55-base deletion component of the <italic>GBA</italic> recombinant allele and Clair3 accurately detected a 19-base indel in <italic>BRCA1</italic>, but all pipeline components failed to detect the 36-base <italic>TRHR</italic> indel, a potential limitation of the current pipeline may be detection of variants between 20 and 55 bases in size, based on this dataset. Given the missed indel call in <italic>TRHR</italic>, the overall indel detection concordance of our pipeline was 96.15% (95% CI: 81%&#x2013;99%) (<xref ref-type="table" rid="T1">Table 1</xref>).</p>
</sec>
<sec id="s3-6">
<title>Detection of clinically relevant SVs</title>
<p>Thirty-two clinically relevant SVs were assessed by our pipeline. A combination of breakpoint-based callers (NanoVar and DeBreak) and read depth-based callers (QDNAseq and CNVpytor) was required to accurately call and detect SVs, including copy number variants (CNVs). The combined output from the four callers for SVs and CNVs resulted in an average of about 56,000 raw calls per patient. This number does not indicate unique calls, but rather the sum of all calls across the four callers, including low quality calls that are subsequently filtered out. This combined total also does not account for overlapping calls made by multiple variant calling tools. After implementing a structured filtering strategy, the average number of relevant calls per sample requiring further evaluation and manual review was reduced to about 62 (<xref ref-type="fig" rid="F1">Figure 1</xref>). All 32 SVs were successfully identified by our pipeline, indicating an SV detection concordance of 100% (95% CI: 89%&#x2013;100%) (<xref ref-type="fig" rid="F4">Figure 4</xref>; <xref ref-type="table" rid="T1">Table 1</xref>).</p>
</sec>
<sec id="s3-7">
<title>Detection of clinically relevant repeat expansions</title>
<p>The validation cohort included 29 clinically significant repeat expansion variants. Our pipeline accurately detected and resolved all 29 repeat expansions, demonstrating a detection concordance of 100% (95% CI: 88%&#x2013;100%) (<xref ref-type="table" rid="T1">Table 1</xref>). Tandem Genotypes software correctly identified clinically significant repeat expansions in <italic>ATXN1, ATXN3, ATXN7, DMPK, FGF14, FMR1, FXN</italic>, and <italic>HTT</italic>. <xref ref-type="fig" rid="F6">Figure 6</xref> depicts the wild type and expanded pathogenic CAG repeat sequences detected in a case of autosomal dominant Spinocerebellar ataxia type 1. All genes with expanded repeats in protein coding regions were accurately detected and sized. Tandem Genotypes also detected larger non-coding expansions, including <italic>FXN</italic> expansions greater than 3,000 nucleotides in length.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Tandem Genotypes waterfall plot depicting CAG expansion in ATXN1 (SCA1) for one proband. The <italic>x</italic>-axis represents genomic position beginning at the start of the CAG repeat tract in <italic>ATXN1</italic>. Genomic reads are stacked and the y-axis depicts read number. Orange regions represent CAG repeat sequence, blue regions represent CAT interruptions, and grey regions represent genetic sequence that is not CAT nor CAG. Half of the reads (upper region of the graph) show a wild type <italic>ATXN1</italic> allele with two CAT interruptions. The remaining reads (lower region of the graph) show an expanded <italic>ATXN1</italic> allele that does not have protective CAT interruptions and has expanded into the pathogenic range causative of spinocerebellar ataxia type 1 (SCA1).</p>
</caption>
<graphic xlink:href="fgene-16-1499456-g006.tif"/>
</fig>
<p>Included in these 29 pathogenic repeats were six <italic>RFC1</italic> expansions, all of which were accurately detected by Sniffles2 but were missed by Tandem Genotypes. Importantly, manual review also showed that the expansion had converted from wild type sequence AAAAG to pathogenic sequence AAGGG.</p>
</sec>
<sec id="s3-8">
<title>Detection of clinically relevant variants in genes with highly homologous pseudogenes</title>
<p>Among the clinical samples assessed, there were 14 clinically significant variants in genes with highly homologous pseudogenes (<xref ref-type="fig" rid="F7">Figure 7</xref>). These variants included SNVs, SVs, and indels in <italic>STRC, CHFR1, CHFR3, SBDS, GBA, PKD1, and SORD</italic>. Importantly, several of these variants could not be detected by short-read NGS and required specialized supplementary assays. The ten SNV and indel variants in genes with highly homologous pseudogenes were detected accurately by the Clair3 component of the ONT-based pipeline. The four SVs were called by Clair3, CNVpytor, and/or Paraphase (<xref ref-type="sec" rid="s11">Supplementary Table S2</xref>). As such, our pipeline was able to detect all these variants in genes with highly homologous pseudogenes, including those that required supplemental methods previously, resulting in a detection concordance of 100% (95% CI: 78%&#x2013;100%) (<xref ref-type="table" rid="T1">Table 1</xref>). Thus, detection of variants in genes with highly homologous pseudogenes is a notable strength of the ONT-based pipeline when compared to short-read NGS.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Variants in genes with highly homologous pseudogenes. Fourteen of the 167 variants from clinical samples assessed in this study occurred in genes with highly homologous pseudogenes. All 14 of these variants were detected by our pipeline. The variant types are coded by color and the specific genes in which the variants occurred are listed to the right of each bar segment.</p>
</caption>
<graphic xlink:href="fgene-16-1499456-g007.tif"/>
</fig>
<p>Since our laboratory routinely offers sequence analysis of <italic>PMS2</italic> by long-range PCR (LR-PCR), we focused specifically on our pipeline&#x2019;s ability to differentiate between variants in <italic>PMS2</italic> and its pseudogene, <italic>PMS2CL</italic>. Paraphase (<xref ref-type="bibr" rid="B3">Chen X. et al., 2023</xref>) (PMID: 36669496) is a computational tool specifically designed for genotyping a small set of genes with homologous pseudogenes, including <italic>PMS2</italic>. LR-PCR results were used as a validation tool for providing orthogonal confirmation of results when possible. A key advantage of Paraphase is its ability to infer distinct <italic>PMS2</italic> and <italic>PMS2CL</italic> haplotypes. Five of the 72 samples had orthogonal LR-PCR data available for <italic>PMS2</italic> comparison with ONT outputs. Therefore, we elected to perform a concordance assessment specifically for Paraphase calls with a VAF &#x3e; 0.2 and a quality score &#x3e;1. In four of the five samples, all calls that satisfied these criteria were concordant between the nanopore and LR-PCR outputs. Of note, the concordance for other calls, such as those within non-coding homopolymer stretches greater than 10 bases long was lower. Typically, variants within these deep intronic non-coding homopolymer stretches are not expected to be clinically relevant.</p>
<p>Importantly, in one of the five samples, Paraphase identified an intronic block of 18 variants that were not called by LR-PCR. Notably, this is a region of the gene where variants are not analyzed and reported clinically. Paraphase accurately established these calls as being present in <italic>PMS2</italic> despite having substantial overlap with <italic>PMS2CL</italic>. Since nanopore sequencing and Paraphase allow for reads to be &#x201c;laddered&#x201d; together to infer distinct haplotypes, this tool allowed us to conclude that the calls made by Paraphase but not LR-PCR represented a true gene conversion of <italic>PMS2</italic> such that it contains a large but clinically insignificant region of <italic>PMS2CL</italic>.</p>
</sec>
<sec id="s3-9">
<title>Cases resolved using ONT</title>
<p>In four cases where short-read NGS findings were not sufficient for a molecular diagnosis in the proband, our ONT-based pipeline was able to accurately detect and resolve variants leading to a complete clinical diagnosis.</p>
<p>In the first case, our current short-read NGS clinical pipeline identified what appeared to be a single heterozygous <italic>FANCA</italic> deletion (exons 1&#x2013;23), in a 6-year-old male with a clinical presentation consistent with Fanconi anemia. A second pathogenic variant in <italic>FANCA</italic> was not detected. In contrast, our ONT-based sequencing pipeline detected two distinct <italic>FANCA</italic> deletions in trans (exons 1&#x2013;11 and exons 12&#x2013;23), with a 140 base pair overlap (<xref ref-type="fig" rid="F8">Figures 8A,B</xref>), allowing for a complete molecular diagnosis for the proband.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Fanconi anemia case resolved using long read sequencing. <bold>(A)</bold> IGV image showing exons 1&#x2013;26 of <italic>FANCA</italic> in a sample from a patient with a clinical diagnosis of Fanconi anemia. By short read NGS, a single deletion call was made spanning exons 1&#x2013;23. Nanopore sequencing identified two distinct <italic>FANCA</italic> deletions in trans (exons 1&#x2013;11 and exons 12&#x2013;23), with a 140&#xa0;bp overlap (blue bars above coverage data). Both deletions were called accurately by Debreak. The red box depicts the genomic region magnified in <xref ref-type="fig" rid="F7">Figure 7B</xref>. <bold>(B)</bold> IGV image showing the region surrounding the 140&#xa0;bp overlap of both <italic>FANCA</italic> deletions.</p>
</caption>
<graphic xlink:href="fgene-16-1499456-g008.tif"/>
</fig>
<p>The second case is a 62-year-old-male with neurologic symptoms consistent with ataxia. Prior testing via a short tandem repeat panel was negative for this patient, albeit before the discovery of Spinocerebellar ataxia 27B (SCA27B) (<xref ref-type="bibr" rid="B26">Pellerin et al., 2023</xref>; <xref ref-type="bibr" rid="B28">Rafehi et al., 2023</xref>) (PMID: 36516086, 37267898). ONT-based testing detected a pathogenic GAA repeat expansion in <italic>FGF14</italic>, indicating a molecular diagnosis of SCA27B for this proband. Subsequently, this patient received <italic>FGF14</italic> short tandem repeat testing at an external laboratory via his clinical care team and the results were consistent with our finding.</p>
<p>In the third case, panel-based NGS testing of <italic>PKD1, PKD2</italic> and <italic>PKHD1</italic> performed in a 12-year-old female with Caroli&#x2019;s syndrome identified two heterozygous pathogenic variants in the <italic>PKD1</italic> gene. It was unclear whether these variants were present in cis or in trans. ONT-based sequencing clearly showed that these variants were present in cis as the result of a gene conversion event that incorporated pseudogene sequence into the <italic>PKD1</italic> gene (<xref ref-type="fig" rid="F9">Figure 9</xref>). It has been shown that the presence of two <italic>PKD1</italic> variants in trans is associated with more severe disease and poorer prognosis (<xref ref-type="bibr" rid="B2">Ali et al., 2015</xref>) (PMID: 25880449). Therefore, establishing the cis phase of the variants in this proband was vital for informed clinical management and provided a more accurate estimate of disease progression.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Caroli syndrome case resolved using long read sequencing IGV image showing two heterozygous pathogenic <italic>PKD1</italic> variants in cis. Reads are grouped by nucleotide at chr16:2164490 demonstrating that all the reads with one pathogenic variant share the other pathogenic variant.</p>
</caption>
<graphic xlink:href="fgene-16-1499456-g009.tif"/>
</fig>
<p>To further demonstrate the prospective clinical utility of this technology, we analyzed a single case that had not been definitively resolved with prior short-read sequencing. The proband was a one-year-old male with a clinical diagnosis of osteopetrosis. Prior to referral to our institution, the patient had short-read whole genome sequencing performed at an external laboratory. This testing identified a single heterozygous c.346C&#x3e;T (p.Gln116&#x2a;) nonsense variant in the <italic>TCIRG1</italic> gene. A second clinically significant <italic>TCIRG1</italic> variant was not identified by the external testing. The patient was referred to our institution for potential bone marrow transplantation based on a presumed diagnosis of autosomal recessive <italic>TCIRG1</italic>-related osteopetrosis. Upon receiving a new sample from the proband and both parents, targeted testing of the parental samples revealed that the c.346C&#x3e;T (p.Gln116&#x2a;) variant was paternally inherited. Short-read whole genome sequencing performed at MDL provided evidence of a maternally inherited potential structural variant, but the exact nature of the variant could not be resolved. This sample was sequenced with ONT and a novel Alu insertion was identified at chr11:67816810 (hg19). This Alu insertion was located 49 bases from the exon 15/intron 15 splice boundary. Both breakpoints of this novel Alu insertion were confirmed by Sanger sequencing. The finding of a novel Alu insertion in trans with a pathogenic nonsense variant allowed for a highly probable molecular diagnosis in this case (<xref ref-type="fig" rid="F10">Figure 10</xref>). This insertion was classified as a variant of uncertain clinical significance by strict application of the ACMG criteria (<xref ref-type="bibr" rid="B29">Richards et al., 2015</xref>) (PMID: 25741868). This Alu insertion is an example of a variant that is difficult to fit into an interpretive framework that is very specifically designed for sequence variants and small indels. However, in consensus conversations with the clinical providers caring for this patient, we strongly believe that this variant represents a diagnostic finding, which was reflected on the final report. Overall, we chose a conservative formal interpretation and classification given the lack of functional evidence regarding this novel intronic insertion. Of note, because this Alu insertion variant could not be formally classified as clinically significant, this case was not included in the table of clinically significant variants.</p>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>Osteopetrosis case resolved using long read sequencing. IGV image showing a single heterozygous pathogenic nonsense <italic>TCIRG1</italic> variant (red star above reads on left side of image) that was identified previously by an external laboratory. Long read sequencing identified a <italic>TCIRG1-</italic>disrupting Alu insertion (purple triangle above reads on right side of image) in trans, leading to a definitive molecular diagnosis. No spanning reads have both the nonsense variant and the Alu insertion, confirming the trans conformation of these variants.</p>
</caption>
<graphic xlink:href="fgene-16-1499456-g010.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<p>This is one of the largest clinical validations performed using long-read sequencing technology to date. Importantly, we successfully combined eight variant callers and developed a pipeline capable of comprehensively detecting a wide range of genetic variants, which is essential for clinical diagnosis of inherited disorders. We also successfully adapted and optimized short-read NGS callers like CNVpytor for a long-read pipeline and have demonstrated that integrating multiple variant callers is essential for utilizing long-read sequencing in clinical diagnostics to detect various genetic alterations. In addition to detecting different types of variants, we have demonstrated that long read sequencing can identify and resolve several complex variants that were not detected by conventional methods, including short read NGS.</p>
<p>Despite the slightly lower analytical sensitivity for small sequence variants seen with long-read sequencing of GIAB sample NA12878 as compared to short-read sequencing, currently there is no single genotyping platform that can match the overall performance of our ONT-based pipeline in the detection of a wide array of genomic alterations. On closer analysis of the 0.5% lower analytical sensitivity for the NA12878 sample, we found that the missed calls did not appear to cluster in any specific gene or region. The missed calls were mostly due to skewed VAFs that were outside the range typically associated with heterozygous calls (<xref ref-type="sec" rid="s11">Supplementary Figure S1</xref>). Notably, multiple iterations of adjusting Clair3 settings were unable to improve detection of these variants. These skewed VAFs were predominantly in the coding regions of the genes and not impacted by homopolymer regions. We recognize that restricting our analysis to coding exons may have reduced the number of artifacts related to homopolymer elements, as the majority of homopolymer elements reside in non-coding regions of the genome. Many of the indel discrepancies did occur adjacent to homopolymer elements, but for SNVs in coding exons, there was no relation to homopolymer content. Due to homopolymer stretches being overrepresented in non-coding sequence, we would expect the analytical sensitivity and specificity to be slightly reduced if non-coding regions were included.</p>
<p>With respect to our clinical cohort of 72 samples, 100% of the clinically relevant small indels (&#x3c;20 bases) and SNVs were identified by our ONT-based pipeline. Notably, this cohort included a sample with a CFTR intron 9 homopolymer 5T allele, which was used to challenge the pipeline performance and this variant was accurately detected.</p>
<p>There is also an opportunity for continued chemistry and bioinformatics improvement of our pipeline as we move towards larger-scale clinical implementation. While significant advancements have been made in the ONT sequencing chemistry and hardware, we recognize that frequently changing tools can present a challenge for clinical diagnostic laboratories. We acknowledge that long read sequencing chemistries and available software to analyze sequencing data have been rapidly improving and anticipate that it will continue to evolve in the future. While re-analyzing the samples in this project with the latest available versions of the pipeline&#x2019;s bioinformatics components was not feasible within the scope of this project, we plan to continually assess our ONT pipeline components for optimal clinical performance as we do for the laboratory&#x2019;s other clinical genomics test offerings.</p>
<p>We will assess future versions of sequencing chemistries and software as they become available, evaluate their impact on variant calling accuracy and make periodic updates to our clinical pipeline as indicated.</p>
<sec id="s4-1">
<title>SNVs</title>
<p>All 80 SNVs evaluated in clinical samples were accurately detected, thus showing a 100% detection concordance for clinically relevant SNVs (<xref ref-type="fig" rid="F4">Figure 4</xref>; <xref ref-type="table" rid="T1">Table 1</xref>). To stress test the pipeline and assess the ability of our platform to detect variants at non-germline VAFs, we utilized an incidental finding of polycythemia vera in one of the samples studied. The JAK2 c.1849G&#x3e;T (p.Val617Phe) variant, observed at a VAF of 24% in peripheral blood on prior short-read testing, was not detected by the variant callers in our pipeline. Manual review of the BAM file confirmed that the variant was present in sequencing reads for this sample. The inability of our pipeline to detect this mosaic variant is not surprising, given that the pipeline was optimized for detecting germline mutations. Therefore, reliable detection of constitutional mosaic findings may be limited with the current ONT-based pipeline.</p>
</sec>
<sec id="s4-2">
<title>Indels and SVs</title>
<p>In several cases with complex SVs, although a diagnostic finding had been reported based on short-read NGS or other techniques, a full characterization of these complex variants could not be completed using existing clinical methods. In these cases, our ONT-based pipeline helped us to clarify the exact genes involved in a particular SV and better understand the ways these complex SVs altered cellular biology and contributed to proband phenotypes. These SVs were detected using a combination of Clair3, breakpoint-based callers, and read-depth-based callers. We were able to develop an effective SV call filtering strategy that eliminated a majority of the superfluous calls, allowing us to focus on reviewing those with the highest likelihood of being clinically relevant, thus allowing for a high detection concordance of clinically relevant SVs (<xref ref-type="fig" rid="F1">Figure 1</xref>). This filtering strategy is critical for establishing a workflow that can be practically implemented in clinical laboratories. A combination of both read-depth and breakpoint-based callers was used; Clair3 is our default genotyper, while the primary SV callers are NanoVar and DeBreak. DeBreak could accurately detect SVs between 100&#xa0;kb and 361&#xa0;kb, but large deletions and duplications (&#x3e;750&#xa0;kb) mediated by repetitive elements and those extending to the chromosome centromere or telomere could not be routinely detected by this caller. NanoVar calls were only used to bridge the gap between the maximum for Clair3 (50 bases) and the minimum for DeBreak (100 bases). In contrast, the general purpose SV caller, Sniffles2, showed poor overall performance in detection of complex SVs (<xref ref-type="sec" rid="s11">Supplementary Figure S3</xref>) and was therefore not included as a primary SV caller in our pipeline. It is unclear as to exactly why Sniffles2 did not perform as well in detecting SV breakpoints. This could be due to the use of complex multi-component SVs intended to challenge the limitations of the SV-calling tools as our pilot dataset. Our data therefore suggests that while Sniffles2 may work well for detecting simple deletions and duplications, it showed less than optimal performance for detection of the complex SVs included in our dataset.</p>
<p>As the first step in our SV filtering strategy, we restricted each of the caller&#x2019;s outputs to a subset of calls related to its variant calling strengths. Our second filtering step entailed identification of variants that were most likely to have an effect on Mendelian disease; another filtering step was then performed to restrict the calls to genes that have a human phenotype (<xref ref-type="sec" rid="s11">Supplementary Table S1</xref>). A final set of logic rules was then applied for accurate detection of a pathogenic SV. This list is not patient- or sample-specific; therefore, this strategy can be adapted for any clinical scenario. We recognize that there are unique complex structural variants, so we maintain all raw outputs for possible manual review in cases with high levels of clinical suspicion.</p>
<p>A 55-base deletion, which was part of the complex SV in <italic>GBA</italic> was not detected by Clair3; but was detected by the breakpoint-based caller NanoVar (<xref ref-type="fig" rid="F5">Figure 5C</xref>). However, the 36-base <italic>TRHR</italic> indel (<xref ref-type="fig" rid="F5">Figure 5B</xref>) was missed both by Clair3 and NanoVar. Manual review of the BAM file showed that this <italic>TRHR</italic> indel was present, indicating that the miss was not due to an error in sequencing. Unfortunately, there were no other SVs in the clinical dataset between 20 and 55 base pairs in length that could be assessed to further narrow the size window of variants this pipeline has difficulty detecting. Thus, reduced capacity to detect indels between 20 and 55 bp may be a potential limitation of our current bioinformatics pipeline that will need to be addressed in future optimization of this clinical workflow.</p>
</sec>
<sec id="s4-3">
<title>Repeat expansions</title>
<p>For analysis of repeat expansions, one specific challenge is that the mere presence of an expanded repeat does not always correlate with pathogenicity. In addition to size of the repeat expansion, nucleotide content is often critical for predicting disease penetrance, severity, age of onset in affected individuals as well as meiotic instability. Using CANVAS as an example we demonstrated that our pipeline was not only able to detect clinically relevant expanded repeats in <italic>RFC1</italic>, but also provided nuanced information about the content. Two key components are required to confer molecular pathogenicity in CANVAS: the repeat content must change from an AAAAG repeat sequence to one of several pathogenic pentanucleotide repeat sequences and the repeat length must expand from a nonpathogenic size (typically 11&#x2013;200 repeats) to greater than 400 repeats (<xref ref-type="bibr" rid="B8">Delforge et al., 2024</xref>) (PMID: 38627134). Indeed, manual review of the BAM files from samples with known pathogenic <italic>RFC1</italic> expansions showed that the reads containing the pathogenic repeats had altered from wild type pentanucleotide sequence AAAAG, to pathogenic pentanucleotide repeat sequence, AAGGG, indicating that the genotyping pipeline was able to accurately measure the size as well as content of this region for clinical diagnostic purposes. Importantly, we demonstrate that Tandem Genotypes was able to identify all the tested repeat expansions in our cohort but was unable to accurately detect all the pathogenic <italic>RFC1</italic> repeats tested. Expanded <italic>RFC1</italic> repeats, implicated in CANVAS, were accurately detected only by Sniffles2. Interestingly, This demonstrates the need to incorporate multiple different callers into a clinical pipeline for detection of clinically relevant variants to achieve high confidence in detection of clinically relevant repeat expansions.</p>
</sec>
<sec id="s4-4">
<title>Genes with highly homologous pseudogenes</title>
<p>As indicated in <xref ref-type="fig" rid="F7">Figure 7</xref>, our pipeline was able to detect all variants in genes with highly homologous pseudogenes. The majority of these variants were detected by Clair3, Nanovar, and CNVpytor. The Paraphase tool was used to assist with variant detection in a specific subset of genes requiring a haplotype based approach, such as <italic>PMS2</italic> and <italic>STRC</italic>, which is in alignment with the UMN MDL&#x2019;s currently available test offerings. When the laboratory eventually transitions to using a newer reference genome (discussed below), the available Paraphase loci offerings supported for that reference will be evaluated for potential clinical implementation.</p>
</sec>
</sec>
<sec id="s5">
<title>Conclusion and future directions</title>
<p>Despite previously being considered suboptimal for clinical applications due to relatively low base calling accuracy, recent improvements in the ONT technology along with decreasing costs and shorter turnaround time have made it an attractive tool for several genomics applications. ONT sequencing is being increasingly used in clinical microbiology laboratories, particularly in the study of infectious diseases, detection of drug resistance, identification of rare and unknown pathogens, as well as real-time genomic surveillance (<xref ref-type="bibr" rid="B18">Marx, 2023</xref>; <xref ref-type="bibr" rid="B24">Park et al., 2021</xref>) (PMID: 36635542, 34211026). However, its use as a diagnostic clinical test for human genetic disorders has been limited. While it has been utilized to complement the findings of short-read NGS and provide additional orthogonal confirmation (<xref ref-type="bibr" rid="B14">Kaplun et al., 2023</xref>) (PMID: 37152986), long-read sequencing has not been used as a standalone method for comprehensive clinical genetic testing. We have developed a comprehensive workflow that is optimized for detection of several different types of genomic variants. Hence, this study is the first of its kind wherein we use ONT-based long-read sequencing as a single comprehensive testing platform for WGS-based clinical testing.</p>
<p>A key highlight of our custom pipeline was the use of eight different variant callers; while certain SV calls were made by more than one caller, all components were essential to achieve high concordance with other reference methods for detection of clinically relevant variants. However, the use of eight callers resulted in multiple output formats, in this case VCFs generated by various callers. Future work will focus on restructuring and formatting the individual outputs for merging into a single VCF that can be used for variant interpretation. This will enable a streamlined workflow capable of identifying a wide spectrum of genetic variation that can be scaled to analyze the vast majority of clinical samples sent for molecular diagnostics. Our clinical workflow will also incorporate manual review of specific complex loci (such as <italic>RFC1</italic>) to optimize its clinical utility.</p>
<p>Our laboratory is also considering the use of newer human reference builds, GRCh38 or Telomere-to-Telomere (T2T), which provide a more thorough and accurate characterization of the human genome along with improved representation of structural variants and complex genomic regions (<xref ref-type="bibr" rid="B11">Guo et al., 2017</xref>; <xref ref-type="bibr" rid="B22">Nurk et al., 2022</xref>; <xref ref-type="bibr" rid="B1">Aganezov et al., 2022</xref>) (PMID: 28131802, 35357919, 35357935). At present, all clinical analyses and reporting infrastructure in our laboratory use GRCh37/hg19 as human reference. As one of the main goals of this project is to make this pipeline available for immediate clinical use, GRCh37/hg19 will remain our human reference sequence to maintain internal consistency and to ensure that our results can be accurately compared to our current clinical pipelines. However, we plan to take the necessary steps to transition this pipeline and our other current clinical pipelines to GRCh38 or T2T in the near future.</p>
<p>With respect to economic efficiency, the current approximate cost of ONT sequencing and producing the raw sequencing data is between $800 and $1,200. This does not include the costs for bioinformatic analysis, professional interpretation, and data storage. This analysis is more cost effective than our current short-read whole genome sequencing platform because it requires fewer orthogonal assays such as repeat expansion analysis to rule out differential diagnoses than our short-read platform. Furthermore, when the assay is clinically implemented, we anticipate that the associated larger batch sizes will result in lower costs per sample.</p>
<p>When it comes to the diagnosis of certain rare diseases, the genetic assessment process can be complex and intricate. A classic example would be hereditary cerebellar ataxia, which can have many putative causal genes, including a constantly growing list of repeat expansion disorders. The diagnostic yield in these patients, even after the advent of short-read NGS, is low, and there continues to be a diagnostic gap (<xref ref-type="bibr" rid="B5">Cortese et al., 2019</xref>; <xref ref-type="bibr" rid="B36">Tenorio et al., 2024</xref>) (PMID: 30926972, 37950147). To improve the diagnostic yield in such rare diseases, it will be essential to develop a one-stop comprehensive genetic test or an assay that can be easily adapted to the identification of new repeat expansion disorders. Our work has shown that long-read sequencing can be leveraged to address this unmet need.</p>
<p>We demonstrate the feasibility of a single cost-effective long-read sequencing platform for the diagnosis of rare genetic disorders caused by a broad spectrum of genetic variation. Offering this comprehensive testing clinically will improve turnaround time and enhance patient care by limiting the number of separate tests required for diagnosis and would allow for future review of the raw data for additional diagnostic testing, should new clinical indications arise. Our validation study has shown that ONT sequencing can efficiently detect a wide range of reportable mutations, thus ensuring rapid turnaround for whole genome sequencing-based clinical genetic testing.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The bioinformatics code is available in a publicly accessible repository. This data can be found here: DOI: <ext-link ext-link-type="uri" xlink:href="http://10.5281/zenodo.14532101">10.5281/zenodo.14532101</ext-link>, <ext-link ext-link-type="uri" xlink:href="https://zenodo.org/records/14805778">https://zenodo.org/records/14805778</ext-link>. All other sequencing data is available in the main and supplementary sections of the manuscript.</p>
</sec>
<sec sec-type="ethics-statement" id="s7">
<title>Ethics statement</title>
<p>The studies involving humans were approved by University of Minnesota Institutional Review Board. The studies were conducted in accordance with the local legislation and institutional requirements. The participants provided their written informed consent to participate in this study. Written informed consent was obtained from the individual(s) for the publication of any potentially identifiable images or data included in this article.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>SS: Conceptualization, Investigation, Supervision, Writing &#x2013; original draft, Writing &#x2013; review and editing. HH: Conceptualization, Data curation, Formal Analysis, Investigation, Supervision, Visualization, Writing &#x2013; original draft, Writing &#x2013; review and editing. AV: Data curation, Methodology, Software, Writing &#x2013; review and editing. ZF: Data curation, Methodology, Software, Writing &#x2013; review and editing. AE: Data curation, Methodology, Software, Writing &#x2013; review and editing. TK: Methodology, Software, Writing &#x2013; review and editing. SM: Methodology, Software, Writing &#x2013; review and editing. RM: Writing &#x2013; review and editing. CB: Writing &#x2013; review and editing. JL: Writing &#x2013; review and editing. SB: Writing &#x2013; review and editing. PM: Writing &#x2013; review and editing. SY: Writing &#x2013; review and editing. AN: Writing &#x2013; review and editing. MB: Conceptualization, Data curation, Formal Analysis, Investigation, Methodology, Software, Supervision, Validation, Writing &#x2013; original draft, Writing &#x2013; review and editing. BT: Conceptualization, Funding acquisition, Investigation, Project administration, Resources, Supervision, Writing &#x2013; original draft, Writing &#x2013; review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. University of Minnesota Department of Laboratory Medicine and Pathology.</p>
</sec>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2025.1499456/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2025.1499456/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material>
<label>SUPPLEMENTARY FIGURE S1</label>
<caption>
<p>Representative &#x2018;missed&#x2019; ONT SNV calls. IGV image showing two SNVs in coding regions for GIAB sample NA12878. The upper set of read data for each variant is from ONT sequencing and the lower set of read data is from short-read sequencing. These SNVs were not accurately called by our ONT pipeline, but they were accurately detected by our short-read pipeline. Both SNVs shown are representative of the skewed VAFs that caused variants called on the short-read pipeline to be missed by the ONT pipeline. These skewed VAFs occurred predominantly in coding regions and were not impacted by homopolymer stretches.</p>
</caption>
</supplementary-material>
<supplementary-material xlink:href="Table2.xlsx" id="SM1" mimetype="application/xlsx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table3.xlsx" id="SM2" mimetype="application/xlsx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image1.tif" id="SM3" mimetype="application/tif" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table1.xlsx" id="SM4" mimetype="application/xlsx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Aganezov</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Soto</surname>
<given-names>D. C.</given-names>
</name>
<name>
<surname>Kirsche</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zarate</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Avdeyev</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>A complete reference genome improves analysis of human genetic variation</article-title>. <source>Science</source> <volume>376</volume>, <fpage>eabl3533</fpage>. <pub-id pub-id-type="doi">10.1126/science.abl3533</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ali</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Hussain</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Naim</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zayed</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Al-Mulla</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Kehinde</surname>
<given-names>E. O.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>A novel PKD1 variant demonstrates a disease-modifying role in trans with a truncating PKD1 mutation in patients with autosomal dominant polycystic kidney disease</article-title>. <source>BMC Nephrol.</source> <volume>16</volume>, <fpage>26</fpage>. <pub-id pub-id-type="doi">10.1186/s12882-015-0015-7</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Harting</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Farrow</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Thiffault</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Kasperaviciute</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Genomics England Research</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Comprehensive SMN1 and SMN2 profiling for spinal muscular atrophy analysis using long-read PacBio HiFi sequencing</article-title>. <source>Am. J. Hum. Genet.</source> <volume>110</volume>, <fpage>240</fpage>&#x2013;<lpage>250</lpage>. <pub-id pub-id-type="doi">10.1016/j.ajhg.2023.01.001</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>A. Y.</given-names>
</name>
<name>
<surname>Barkley</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Deciphering the exact breakpoints of structural variations using long sequencing reads with DeBreak</article-title>. <source>Nat. Commun.</source> <volume>14</volume>, <fpage>283</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-023-35996-1</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cortese</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Simone</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Sullivan</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Vandrovcova</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Tariq</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yau</surname>
<given-names>W. Y.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Biallelic expansion of an intronic repeat in RFC1 is a common cause of late-onset ataxia</article-title>. <source>Nat. Genet.</source> <volume>51</volume>, <fpage>649</fpage>&#x2013;<lpage>658</lpage>. <pub-id pub-id-type="doi">10.1038/s41588-019-0372-4</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Danecek</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Bonfield</surname>
<given-names>J. K.</given-names>
</name>
<name>
<surname>Liddle</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Marshall</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ohan</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Pollard</surname>
<given-names>M. O.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Twelve years of SAMtools and BCFtools gigascience 10</article-title>. <source>Gigascience</source> <volume>10</volume>, <fpage>giab008</fpage>. <pub-id pub-id-type="doi">10.1093/gigascience/giab008</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>De Coster</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Rademakers</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>NanoPack2: population-scale evaluation of long-read sequencing data</article-title>. <source>Bioinformatics</source> <volume>39</volume>, <fpage>btad311</fpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btad311</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Delforge</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Tard</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Davion</surname>
<given-names>J. B.</given-names>
</name>
<name>
<surname>Dujardin</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Wissocq</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dhaenens</surname>
<given-names>C. M.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>RFC1: motifs and phenotypes</article-title>. <source>Rev. Neurol. (Paris)</source> <volume>180</volume>, <fpage>393</fpage>&#x2013;<lpage>409</lpage>. <pub-id pub-id-type="doi">10.1016/j.neurol.2024.03.006</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fernandez-Marmiesse</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gouveia</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Couce</surname>
<given-names>M. L.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>NGS technologies as a turning point in rare disease research, diagnosis and treatment</article-title>. <source>Curr. Med. Chem.</source> <volume>25</volume>, <fpage>404</fpage>&#x2013;<lpage>432</lpage>. <pub-id pub-id-type="doi">10.2174/0929867324666170718101946</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gorzynski</surname>
<given-names>J. E.</given-names>
</name>
<name>
<surname>Goenka</surname>
<given-names>S. D.</given-names>
</name>
<name>
<surname>Shafin</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Jensen</surname>
<given-names>T. D.</given-names>
</name>
<name>
<surname>Fisk</surname>
<given-names>D. G.</given-names>
</name>
<name>
<surname>Grove</surname>
<given-names>M. E.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Ultrarapid nanopore genome sequencing in a critical care setting</article-title>. <source>N. Engl. J. Med.</source> <volume>386</volume>, <fpage>700</fpage>&#x2013;<lpage>702</lpage>. <pub-id pub-id-type="doi">10.1056/NEJMc2112090</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Samuels</surname>
<given-names>D. C.</given-names>
</name>
<name>
<surname>Shyr</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Improvements and impacts of GRCh38 human reference on high throughput sequencing data analysis</article-title>. <source>Genomics</source> <volume>109</volume>, <fpage>83</fpage>&#x2013;<lpage>90</lpage>. <pub-id pub-id-type="doi">10.1016/j.ygeno.2017.01.005</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hartman</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Beckman</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Silverstein</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Yohe</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Schomaker</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Henzler</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Next generation sequencing for clinical diagnostics: five year experience of an academic laboratory</article-title>. <source>Mol. Genet. Metab. Rep.</source> <volume>19</volume>, <fpage>100464</fpage>. <pub-id pub-id-type="doi">10.1016/j.ymgmr.2019.100464</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hook</surname>
<given-names>P. W.</given-names>
</name>
<name>
<surname>Timp</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Beyond assembly: the increasing flexibility of single-molecule sequencing technology</article-title>. <source>Nat. Rev. Genet.</source> <volume>24</volume>, <fpage>627</fpage>&#x2013;<lpage>641</lpage>. <pub-id pub-id-type="doi">10.1038/s41576-023-00600-1</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kaplun</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Krautz-Peterson</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Neerman</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Stanley</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Hussey</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Folwick</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>ONT long-read WGS for variant discovery and orthogonal confirmation of short read WGS derived genetic variants in clinical genetic testing</article-title>. <source>Front. Genet.</source> <volume>14</volume>, <fpage>1145285</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2023.1145285</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kielbasa</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Wan</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Sato</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Horton</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Frith</surname>
<given-names>M. C.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Adaptive seeds tame genomic sequence comparison</article-title>. <source>Genome Res.</source> <volume>21</volume>, <fpage>487</fpage>&#x2013;<lpage>493</lpage>. <pub-id pub-id-type="doi">10.1101/gr.113985.110</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Duffy</surname>
<given-names>B. F.</given-names>
</name>
<name>
<surname>Hoisington-Lopez</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Crosby</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Porche-Sorbet</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>High-resolution HLA typing by long reads from the R10.3 Oxford nanopore flow cells</article-title>. <source>Hum. Immunol.</source> <volume>82</volume>, <fpage>288</fpage>&#x2013;<lpage>295</lpage>. <pub-id pub-id-type="doi">10.1016/j.humimm.2021.02.005</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Giordano</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Ning</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Oxford nanopore MinION sequencing and genome assembly</article-title>. <source>Genomics Proteomics Bioinformatics</source> <volume>14</volume>, <fpage>265</fpage>&#x2013;<lpage>279</lpage>. <pub-id pub-id-type="doi">10.1016/j.gpb.2016.05.004</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Marx</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Method of the year: long-read sequencing</article-title>. <source>Nat. Methods</source> <volume>20</volume>, <fpage>6</fpage>&#x2013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1038/s41592-022-01730-w</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Merker</surname>
<given-names>J. D.</given-names>
</name>
<name>
<surname>Wenger</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Sneddon</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Grove</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zappala</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Fresard</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Long-read genome sequencing identifies causal structural variation in a Mendelian disease</article-title>. <source>Genet. Med.</source> <volume>20</volume>, <fpage>159</fpage>&#x2013;<lpage>163</lpage>. <pub-id pub-id-type="doi">10.1038/gim.2017.86</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mitsuhashi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Frith</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Mizuguchi</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Miyatake</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Toyota</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Adachi</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Tandem-genotypes: robust detection of tandem repeat expansions from long DNA reads</article-title>. <source>Genome Biol.</source> <volume>20</volume>, <fpage>58</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-019-1667-6</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ni</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Simeneh</surname>
<given-names>Z. M.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Benchmarking of Nanopore R10.4 and R9.4.1 flow cells in single-cell whole-genome amplification and whole-genome shotgun sequencing</article-title>. <source>Comput. Struct. Biotechnol. J.</source> <volume>21</volume>, <fpage>2352</fpage>&#x2013;<lpage>2364</lpage>. <pub-id pub-id-type="doi">10.1016/j.csbj.2023.03.038</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nurk</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Koren</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rhie</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Rautiainen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bzikadze</surname>
<given-names>A. V.</given-names>
</name>
<name>
<surname>Mikheenko</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>The complete sequence of a human genome</article-title>. <source>Science</source> <volume>376</volume>, <fpage>44</fpage>&#x2013;<lpage>53</lpage>. <pub-id pub-id-type="doi">10.1126/science.abj6987</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Onsongo</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Baughn</surname>
<given-names>L. B.</given-names>
</name>
<name>
<surname>Bower</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Henzler</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Schomaker</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Silverstein</surname>
<given-names>K. A.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>CNV-RF is a random forest-based copy number variation detection method using next-generation sequencing</article-title>. <source>J. Mol. Diagn</source> <volume>18</volume>, <fpage>872</fpage>&#x2013;<lpage>881</lpage>. <pub-id pub-id-type="doi">10.1016/j.jmoldx.2016.07.001</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Park</surname>
<given-names>S. Y.</given-names>
</name>
<name>
<surname>Faraci</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Ward</surname>
<given-names>P. M.</given-names>
</name>
<name>
<surname>Emerson</surname>
<given-names>J. F.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>H. Y.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>High-precision and cost-efficient sequencing for real-time COVID-19 surveillance</article-title>. <source>Sci. Rep.</source> <volume>11</volume>, <fpage>13669</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-021-93145-4</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pedersen</surname>
<given-names>B. S.</given-names>
</name>
<name>
<surname>Quinlan</surname>
<given-names>A. R.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Mosdepth: quick coverage calculation for genomes and exomes</article-title>. <source>Bioinformatics</source> <volume>34</volume>, <fpage>867</fpage>&#x2013;<lpage>868</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btx699</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pellerin</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Danzi</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Wilke</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Renaud</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Fazal</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Dicaire</surname>
<given-names>M. J.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Deep intronic FGF14 GAA repeat expansion in late-onset cerebellar ataxia</article-title>. <source>N. Engl. J. Med.</source> <volume>388</volume>, <fpage>128</fpage>&#x2013;<lpage>141</lpage>. <pub-id pub-id-type="doi">10.1056/NEJMoa2207406</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Quinlan</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Hall</surname>
<given-names>I. M.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>BEDTools: a flexible suite of utilities for comparing genomic features</article-title>. <source>Bioinformatics</source> <volume>26</volume>, <fpage>841</fpage>&#x2013;<lpage>842</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btq033</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rafehi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Read</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Szmulewicz</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Davies</surname>
<given-names>K. C.</given-names>
</name>
<name>
<surname>Snell</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Fearnley</surname>
<given-names>L. G.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>An intronic GAA repeat expansion in FGF14 causes the autosomal-dominant adult-onset ataxia SCA27B/ATX-FGF14</article-title>. <source>Am. J. Hum. Genet.</source> <volume>110</volume>, <fpage>1018</fpage>. <pub-id pub-id-type="doi">10.1016/j.ajhg.2023.05.005</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Richards</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Aziz</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Bale</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bick</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Das</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gastier-Foster</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Standards and guidelines for the interpretation of sequence variants: a joint consensus recommendation of the American college of medical genetics and genomics and the association for molecular Pathology</article-title>. <source>Genet. Med.</source> <volume>17</volume>, <fpage>405</fpage>&#x2013;<lpage>424</lpage>. <pub-id pub-id-type="doi">10.1038/gim.2015.30</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rudaks</surname>
<given-names>L. I.</given-names>
</name>
<name>
<surname>Yeow</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ng</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Deveson</surname>
<given-names>I. W.</given-names>
</name>
<name>
<surname>Kennerson</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>K. R.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>An update on the adult-onset hereditary cerebellar ataxias: novel genetic causes and new diagnostic approaches</article-title>. <source>Cerebellum</source> <volume>23</volume>, <fpage>2152</fpage>&#x2013;<lpage>2168</lpage>. <pub-id pub-id-type="doi">10.1007/s12311-024-01703-z</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sedlazeck</surname>
<given-names>F. J.</given-names>
</name>
<name>
<surname>Rescheneder</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Smolka</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Nattestad</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>von Haeseler</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Accurate detection of complex structural variations using single-molecule sequencing</article-title>. <source>Nat. Methods</source> <volume>15</volume>, <fpage>461</fpage>&#x2013;<lpage>468</lpage>. <pub-id pub-id-type="doi">10.1038/s41592-018-0001-7</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Singh</surname>
<given-names>R. R.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Target enrichment approaches for next-generation sequencing applications in oncology</article-title>. <source>Diagnostics (Basel)</source> <volume>12</volume>, <fpage>1539</fpage>. <pub-id pub-id-type="doi">10.3390/diagnostics12071539</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Srivathsan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Suarez</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Emerson</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Meier</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>ONTbarcoder 2.0: rapid species discovery and identification with real&#x2010;time barcoding facilitated by Oxford Nanopore R10.4</article-title>. <source>Cladistics</source> <volume>40</volume>, <fpage>192</fpage>&#x2013;<lpage>203</lpage>. <pub-id pub-id-type="doi">10.1111/cla.12566</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stevanovski</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Chintalaphani</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Gamaarachchi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ferguson</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Pineda</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Scriba</surname>
<given-names>C. K.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Comprehensive genetic diagnosis of tandem repeat expansion disorders with programmable targeted nanopore sequencing</article-title>. <source>Sci. Adv.</source> <volume>8</volume>, <fpage>eabm5386</fpage>. <pub-id pub-id-type="doi">10.1126/sciadv.abm5386</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Suvakov</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Panda</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Diesh</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Holmes</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Abyzov</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>CNVpytor: a tool for copy number variation detection and analysis from read depth and allele imbalance in whole-genome sequencing</article-title>. <source>Gigascience</source> <volume>10</volume>, <fpage>giab074</fpage>. <pub-id pub-id-type="doi">10.1093/gigascience/giab074</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tenorio</surname>
<given-names>R. B.</given-names>
</name>
<name>
<surname>Camargo</surname>
<given-names>C. H. F.</given-names>
</name>
<name>
<surname>Donis</surname>
<given-names>K. C.</given-names>
</name>
<name>
<surname>Almeida</surname>
<given-names>C. C. B.</given-names>
</name>
<name>
<surname>Teive</surname>
<given-names>H. A. G.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Diagnostic yield of NGS tests for hereditary ataxia: a systematic review</article-title>. <source>Cerebellum</source> <volume>23</volume>, <fpage>1552</fpage>&#x2013;<lpage>1565</lpage>. <pub-id pub-id-type="doi">10.1007/s12311-023-01629-y</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tham</surname>
<given-names>C. Y.</given-names>
</name>
<name>
<surname>Tirado-Magallanes</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Goh</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fullwood</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Koh</surname>
<given-names>B. T. H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>NanoVar: accurate characterization of patients&#x2019; genomic structural variants using low-depth nanopore sequencing</article-title>. <source>Genome Biol.</source> <volume>21</volume>, <fpage>56</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-020-01968-7</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zavodna</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bagshaw</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Brauning</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gemmell</surname>
<given-names>N. J.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>The accuracy, feasibility and challenges of sequencing short tandem repeats using next-generation sequencing platforms</article-title>. <source>PLoS One</source> <volume>9</volume>, <fpage>e113862</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0113862</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Leung</surname>
<given-names>A. W.</given-names>
</name>
<name>
<surname>Lam</surname>
<given-names>T. W.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Symphonizing pileup and full-alignment for deep learning-based long-read variant calling</article-title>. <source>Nat. Comput. Sci.</source> <volume>2</volume>, <fpage>797</fpage>&#x2013;<lpage>803</lpage>. <pub-id pub-id-type="doi">10.1038/s43588-022-00387-x</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zook</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Catoe</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>McDaniel</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Vang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Spies</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Sidow</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Extensive sequencing of seven human genomes to characterize benchmark reference materials</article-title>. <source>Sci. Data</source> <volume>3</volume>, <fpage>160025</fpage>. <pub-id pub-id-type="doi">10.1038/sdata.2016.25</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>