<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1213907</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2023.1213907</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Enhancing genomic mutation data storage optimization based on the compression of asymmetry of sparsity</article-title>
<alt-title alt-title-type="left-running-head">Ding et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2023.1213907">10.3389/fgene.2023.1213907</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Ding</surname>
<given-names>Youde</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2133169/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liao</surname>
<given-names>Yuan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>He</surname>
<given-names>Ji</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ma</surname>
<given-names>Jianfeng</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wei</surname>
<given-names>Xu</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Xuemei</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Zhang</surname>
<given-names>Guiying</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Wang</surname>
<given-names>Jing</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>The Sixth Affiliated Hospital of Guangzhou Medical University</institution>, <institution>Qingyuan People&#x2019;s Hospital</institution>, <addr-line>Qingyuan</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>School of Biomedical Engineering</institution>, Guangzhou Medical University, <addr-line>Guangzhou</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1090079/overview">Zhenhua Yu</ext-link>, Ningxia University, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1530424/overview">Xianling Dong</ext-link>, Chengde Medical University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1297181/overview">Zhenhui Dai</ext-link>, Guangzhou University of Chinese Medicine, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Guiying Zhang, <email>guiyingzh@126.com</email>; Jing Wang, <email>eewangjing@163.com</email>
</corresp>
<fn fn-type="equal" id="fn1">
<label>
<sup>&#x2020;</sup>
</label>
<p>These authors have contributed equally to this work</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>01</day>
<month>06</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>14</volume>
<elocation-id>1213907</elocation-id>
<history>
<date date-type="received">
<day>28</day>
<month>04</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>24</day>
<month>05</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Ding, Liao, He, Ma, Wei, Liu, Zhang and Wang.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Ding, Liao, He, Ma, Wei, Liu, Zhang and Wang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>
<bold>Background:</bold> With the rapid development of high-throughput sequencing technology and the explosive growth of genomic data, storing, transmitting and processing massive amounts of data has become a new challenge. How to achieve fast lossless compression and decompression according to the characteristics of the data to speed up data transmission and processing requires research on relevant compression algorithms.</p>
<p>
<bold>Methods:</bold> In this paper, a compression algorithm for sparse asymmetric gene mutations (CA_SAGM) based on the characteristics of sparse genomic mutation data was proposed. The data was first sorted on a row-first basis so that neighboring non-zero elements were as close as possible to each other. The data were then renumbered using the reverse Cuthill-Mckee sorting technique. Finally the data were compressed into sparse row format (CSR) and stored. We had analyzed and compared the results of the CA_SAGM, coordinate format (COO) and compressed sparse column format (CSC) algorithms for sparse asymmetric genomic data. Nine types of single-nucleotide variation (SNV) data and six types of copy number variation (CNV) data from the TCGA database were used as the subjects of this study. Compression and decompression time, compression and decompression rate, compression memory and compression ratio were used as evaluation metrics. The correlation between each metric and the basic characteristics of the original data was further investigated.</p>
<p>
<bold>Results:</bold> The experimental results showed that the COO method had the shortest compression time, the fastest compression rate and the largest compression ratio, and had the best compression performance. CSC compression performance was the worst, and CA_SAGM compression performance was between the two. When decompressing the data, CA_SAGM performed the best, with the shortest decompression time and the fastest decompression rate. COO decompression performance was the worst. With increasing sparsity, the COO, CSC and CA_SAGM algorithms all exhibited longer compression and decompression times, lower compression and decompression rates, larger compression memory and lower compression ratios. When the sparsity was large, the compression memory and compression ratio of the three algorithms showed no difference characteristics, but the rest of the indexes were still different.</p>
<p>
<bold>Conclusion:</bold> CA_SAGM was an efficient compression algorithm that combines compression and decompression performance for sparse genomic mutation data.</p>
</abstract>
<kwd-group>
<kwd>genomic</kwd>
<kwd>sparse</kwd>
<kwd>compression</kwd>
<kwd>single-nucleotide variation</kwd>
<kwd>copy number variation</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Genomics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Genes are one of the basic units of life and are of irreplaceable importance in the fields of understanding life phenomena, exploring the laws of biological evolution, and preventing and controlling human diseases (<xref ref-type="bibr" rid="B40">Tu et al., 2006</xref>; <xref ref-type="bibr" rid="B31">Oh et al., 2012</xref>). Gene sequences are the carriers of biological genetic information, and the biological properties of all organisms are related to genes (<xref ref-type="bibr" rid="B28">Mota and Franke, 2020</xref>). Due to the enormous usefulness of genetic data and the reduced cost of sequencing, many countries and organizations have initiated various genetic engineering projects, such as the Personal Genome Project (<xref ref-type="bibr" rid="B1">Ball et al., 2012</xref>) and the Bio Genome Project (<xref ref-type="bibr" rid="B22">Lewin et al., 2018</xref>). The rapid growth of genetic data can provide a significant boost to the life sciences. A rich gene pool can be very beneficial to the study of certain types of diseases, providing a new breakthrough to promote precision medicine and help solve medical problems (<xref ref-type="bibr" rid="B14">Janssen et al., 2011</xref>; <xref ref-type="bibr" rid="B5">Chen et al., 2020</xref>; <xref ref-type="bibr" rid="B12">Garand et al., 2020</xref>).</p>
<p>However, the growth of genetic data has now greatly outpaced the growth of storage and transmission bandwidth, posing significant storage and transmission challenges (<xref ref-type="bibr" rid="B44">Xi et al., 2023a</xref>). The Human Genome Project (<xref ref-type="bibr" rid="B4">Cavalli-Sforza, 2005</xref>; <xref ref-type="bibr" rid="B3">Boeke et al., 2016</xref>) and the 1000 Genomes Project (<xref ref-type="bibr" rid="B2">Belsare et al., 2019</xref>; <xref ref-type="bibr" rid="B9">Fairley et al., 2020</xref>), for example, generate huge amounts of data, tens of terabytes or even more. Thus, issues related to genetic data compression have become a hot topic and focus of research in recent years. Genomic mutation data contain a large amount of genetic variation information that can be used to resolve the functional and phenotypic effects of genetic variants, which is of great value for human evolutionary genetic and medical research. Comparative databases (such as dbSNP and ClinVar) allow sequencing and differential analysis of genes in individuals or populations of species. Genetic information such as single-nucleotide variation (SNV), insertion deletion (InDel), structural variation (SV) and copy number variation (CNV) can be used to develop molecular markers and create databases of genetic polymorphisms. Cross-species genome alignment methods provide genomic context for the identification of annotated gene regions for variation across species (<xref ref-type="bibr" rid="B35">Samaha et al., 2021</xref>). In recent years, many researchers have developed a variety of rapid detection methods or tools for CNV (<xref ref-type="bibr" rid="B13">Huang et al., 2021</xref>; <xref ref-type="bibr" rid="B20">Lavrichenko et al., 2021</xref>; <xref ref-type="bibr" rid="B16">Kim et al., 2022</xref>) and SNV (<xref ref-type="bibr" rid="B41">van der Borght et al., 2015</xref>; <xref ref-type="bibr" rid="B36">Schnepp et al., 2019</xref>; <xref ref-type="bibr" rid="B23">Li et al., 2022</xref>). However, variant genomic mutation data are often sparse data formats that are difficult to apply with traditional compression methods.</p>
<p>Traditional compression algorithms generally reduce the storage space of data by encoding it, such as Huffman coding (<xref ref-type="bibr" rid="B27">Moffat, 2019</xref>), Lempel-Ziv-Welch coding (<xref ref-type="bibr" rid="B10">Fira and Goras, 2008</xref>; <xref ref-type="bibr" rid="B29">Naqvi et al., 2011</xref>), etc. These algorithms are designed based on the assumption that there is a large amount of repetitive information in the data. But for sparse data, there is less redundancy in the information present in the data, making it difficult to compress effectively. The operations in turn waste a lot of time performing invalid operations with zero elements. As a result, traditional algorithms such as gzip, bzip2, lzo, snappy, etc. Are memory wasting and inefficient. As a result, compressed storage methods for sparse genes, a special form of data, have received increasing attention from researchers (<xref ref-type="bibr" rid="B37">Shekaramiz et al., 2019</xref>; <xref ref-type="bibr" rid="B51">Yao et al., 2019</xref>; <xref ref-type="bibr" rid="B24">Li et al., 2021</xref>; <xref ref-type="bibr" rid="B42">Wang et al., 2022</xref>). Although there are some sparse compression methods available, such as coordinate format (COO) and compressed sparse column format (CSC) compression (<xref ref-type="bibr" rid="B32">Park et al., 2020</xref>), they suffer from different drawbacks. Some are difficult to operate and cannot perform matrix operations, while others have problems such as slow inner product operations and slow row/column slicing operations, so none are particularly desirable either.</p>
<p>In this paper, based on the sparse asymmetry of variant genomic data, we propose a method for lossless compression of genomic mutation data called CA_SAGM. Preprocessing steps such as prioritization and reverse Cuthill-Mckee (RCM) sorting are performed on the data to greatly reduce the bandwidth of the matrix, so that the scattered non-zero elements all converge towards the diagonal. The data is then compressed sparse row format (CSR) (<xref ref-type="bibr" rid="B17">Koza et al., 2014</xref>; <xref ref-type="bibr" rid="B7">Chen et al., 2018</xref>; <xref ref-type="bibr" rid="B49">Xing et al., 2022</xref>) and stored. This method can theoretically optimize the efficiency and quality of the rearranged data, saving processing time and memory requirements. This study shows that CA_SAGM exhibits higher compression performance and best decompression performance for sparse genomic data compared to COO and CSC. From a combination of several evaluation metrics such as compression and decompression time, compression and decompression rate, compression size and compression ratio, the CA_SAGM method performs the best and outperforms the rest of the methods. It is confirmed that the CA_SAGM method has fast and efficient compression and decompression performance for sparse genomic data, has good applicability and can be further extended to other similar data.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<p>Both SNV and CNV are common formats for genomic mutation data storage. SNV is a single nucleotide mutation resulting in a deletion, insertion or substitution in a normal human gene. Large-scale tumor sequencing studies have shown that most cancers are caused by SNV (<xref ref-type="bibr" rid="B25">Macintyre et al., 2016</xref>). DNA copy number variation is a structural form of genomic variation (<xref ref-type="bibr" rid="B26">Medvedev et al., 2009</xref>; <xref ref-type="bibr" rid="B38">Stankiewicz and Lupski, 2010</xref>). Many studies have shown that CNVs are associated with complex diseases such as autism, schizophrenia, Alzheimer&#x2019;s disease, and cancer. In recent years, there have been a large number of studies on SNVs and CNVs (<xref ref-type="bibr" rid="B15">Jugas et al., 2021</xref>; <xref ref-type="bibr" rid="B33">Prashant et al., 2021</xref>; <xref ref-type="bibr" rid="B19">Ladeira et al., 2022</xref>; <xref ref-type="bibr" rid="B21">Lee et al., 2022</xref>; <xref ref-type="bibr" rid="B23">Li et al., 2022</xref>; <xref ref-type="bibr" rid="B52">Zheng, 2022</xref>).</p>
<sec id="s2-1">
<title>2.1 Materials</title>
<p>In this paper, SNV data for nine different diseases and CNV data for six different diseases were selected from the TCGA database (<xref ref-type="bibr" rid="B39">The ICGC/TCGA Pan-Cancer Analysis of Whole Genomes Consortium, 2020</xref>), all data are level3. SNVs or mutations are less common than other variants and mutations and cannot be observed in the diversity of the genome (<xref ref-type="bibr" rid="B34">Press et al., 2019</xref>). It is a single-nucleotide variation without any frequency restriction and may arise in somatic cells. The number and type of SNVs and other characteristics can reflect the genetic diversity, evolutionary history and other information of a species. SNVs also play an important role in the occurrence and development of human diseases (<xref ref-type="bibr" rid="B48">Xi et al., 2020a</xref>; <xref ref-type="bibr" rid="B47">Xi et al., 2023b</xref>). For example, some SNVs may cause gene mutations and affect protein structure and function, leading to the development of diseases; SNV-based research also helps to find susceptibility genes for diseases and develop corresponding drug targets, etc. SNV data are from brain tumor, acute myeloid leukemia, thyroid cancer, prostate cancer, ovarian cancer, breast cancer, bladder cancer, renal clear cell carcinoma and colorectal cancer.</p>
<p>CNV, or copy number variation, is caused by rearrangements in the genome. It generally refers to an increase or decrease in the copy number of a large segment of the genome. It is mainly manifested as deletions and duplications at the sub-microscopic level. CNV is an important genetic basis for individual differences and is widely distributed in the human genome (<xref ref-type="bibr" rid="B45">Xi and Li, 2016</xref>; <xref ref-type="bibr" rid="B46">Xi et al., 2020b</xref>). The CNV data are more complex than the SNV data, with larger data sets, higher numbers of non-zeros and higher densities. CNV data were obtained from acute spinal leukemia, thyroid cancer, prostate cancer, bladder cancer, renal clear cell carcinoma and colorectal cancer. The basic characteristics of SNV and CNV data were analyzed in detail, including Data set size (n), non-zero number (n), sparsity (%), rows (n), rows/columns (%), file size (K), L1-norm, L2-norm and Rank. They are shown in <xref ref-type="table" rid="T1">Tables 1</xref>, <xref ref-type="table" rid="T2">2</xref> respectively.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Raw SNV data benchmark results.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">SNV data</th>
<th align="left">Data set size(n)</th>
<th align="left">Non-zero number(n)</th>
<th align="left">Sparsity (%)</th>
<th align="left">Rows (n)</th>
<th align="left">Rows/columns (%)</th>
<th align="left">File size (K)</th>
<th align="left">L1-norm</th>
<th align="left">L2-norm</th>
<th align="left">Rank</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Brain tumor</td>
<td align="center">1006707</td>
<td align="center">395</td>
<td align="center">0.039</td>
<td align="center">83</td>
<td align="center">0.684</td>
<td align="center">2</td>
<td align="center">38</td>
<td align="center">10.167</td>
<td align="center">66</td>
</tr>
<tr>
<td align="left">Acute myelogenous leukemia</td>
<td align="center">2377284</td>
<td align="center">1683</td>
<td align="center">0.071</td>
<td align="center">196</td>
<td align="center">1.616</td>
<td align="center">7</td>
<td align="center">56</td>
<td align="center">11.417</td>
<td align="center">187</td>
</tr>
<tr>
<td align="left">Thyroid carcinoma</td>
<td align="center">4863729</td>
<td align="center">4780</td>
<td align="center">0.098</td>
<td align="center">401</td>
<td align="center">3.306</td>
<td align="center">17</td>
<td align="center">241</td>
<td align="center">16.173</td>
<td align="center">400</td>
</tr>
<tr>
<td align="left">Prostate cancer</td>
<td align="center">4038957</td>
<td align="center">9004</td>
<td align="center">0.223</td>
<td align="center">333</td>
<td align="center">2.745</td>
<td align="center">27</td>
<td align="center">52</td>
<td align="center">29.835</td>
<td align="center">332</td>
</tr>
<tr>
<td align="left">Ovarian cancer</td>
<td align="center">3832764</td>
<td align="center">12873</td>
<td align="center">0.336</td>
<td align="center">316</td>
<td align="center">2.605</td>
<td align="center">35</td>
<td align="center">312</td>
<td align="center">22.599</td>
<td align="center">316</td>
</tr>
<tr>
<td align="left">Breast cancer</td>
<td align="center">6202131</td>
<td align="center">20287</td>
<td align="center">0.327</td>
<td align="center">507</td>
<td align="center">4.145</td>
<td align="center">52</td>
<td align="center">208</td>
<td align="center">24.876</td>
<td align="center">507</td>
</tr>
<tr>
<td align="left">Bladder cancer</td>
<td align="center">1576770</td>
<td align="center">25368</td>
<td align="center">1.609</td>
<td align="center">130</td>
<td align="center">1.072</td>
<td align="center">53</td>
<td align="center">167</td>
<td align="center">52.160</td>
<td align="center">130</td>
</tr>
<tr>
<td align="left">Clear cell carcinoma of kidney</td>
<td align="center">5142696</td>
<td align="center">24023</td>
<td align="center">0.470</td>
<td align="center">424</td>
<td align="center">3.496</td>
<td align="center">60</td>
<td align="center">274</td>
<td align="center">30.251</td>
<td align="center">424</td>
</tr>
<tr>
<td align="left">Colorectal cancer</td>
<td align="center">2716896</td>
<td align="center">48038</td>
<td align="center">1.768</td>
<td align="center">224</td>
<td align="center">1.847</td>
<td align="center">91</td>
<td align="center">415</td>
<td align="center">294.987</td>
<td align="center">224</td>
</tr>
<tr>
<td align="left">mean</td>
<td align="center">3528659.33</td>
<td align="center">16272.33</td>
<td align="center">0.55</td>
<td align="center">290.44</td>
<td align="center">2.39</td>
<td align="center">38.22</td>
<td align="center">195.89</td>
<td align="center">54.72</td>
<td align="center">287.33</td>
</tr>
<tr>
<td align="left">sd</td>
<td align="center">1733587.32</td>
<td align="center">15121.90</td>
<td align="center">0.62</td>
<td align="center">142.10</td>
<td align="center">1.10</td>
<td align="center">28.60</td>
<td align="center">130.29</td>
<td align="center">90.98</td>
<td align="center">145.89</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Where n represents the number of data, % represents the percentage and k represents kilobytes.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Raw CNV data benchmark results.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">CNV data</th>
<th align="left">Data set size(n)</th>
<th align="left">Non-zero number(n)</th>
<th align="left">Non-negative ratio(%)</th>
<th align="left">Sparsity (%)</th>
<th align="left">Rows (n)</th>
<th align="left">Rows/columns (%)</th>
<th align="left">File size (K)</th>
<th align="left">L1-norm</th>
<th align="left">L2-norm</th>
<th align="left">Rank</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Acute myelogenous leukemia</td>
<td align="center">2316639</td>
<td align="center">81104</td>
<td align="center">48.12</td>
<td align="center">3.50</td>
<td align="center">191</td>
<td align="center">1.574738</td>
<td align="center">92</td>
<td align="center">34</td>
<td align="center">143.673</td>
<td align="center">157</td>
</tr>
<tr>
<td align="left">Thyroid carcinoma</td>
<td align="center">6015984</td>
<td align="center">189060</td>
<td align="center">58.81</td>
<td align="center">3.14</td>
<td align="center">496</td>
<td align="center">4.089373</td>
<td align="center">175</td>
<td align="center">92</td>
<td align="center">224.882</td>
<td align="center">279</td>
</tr>
<tr>
<td align="left">Prostate cancer</td>
<td align="center">4038957</td>
<td align="center">556304</td>
<td align="center">38.74</td>
<td align="center">13.77</td>
<td align="center">333</td>
<td align="center">2.745486</td>
<td align="center">812</td>
<td align="center">259</td>
<td align="center">372.086</td>
<td align="center">325</td>
</tr>
<tr>
<td align="left">Colorectal cancer</td>
<td align="center">3117153</td>
<td align="center">753833</td>
<td align="center">54.11</td>
<td align="center">24.18</td>
<td align="center">257</td>
<td align="center">2.118889</td>
<td align="center">910</td>
<td align="center">224</td>
<td align="center">492.879</td>
<td align="center">253</td>
</tr>
<tr>
<td align="left">Bladder cancer</td>
<td align="center">1552512</td>
<td align="center">780530</td>
<td align="center">53.59</td>
<td align="center">50.28</td>
<td align="center">128</td>
<td align="center">1.055322</td>
<td align="center">800</td>
<td align="center">144</td>
<td align="center">435.185</td>
<td align="center">127</td>
</tr>
<tr>
<td align="left">Clear cell carcinoma of kidney</td>
<td align="center">5288244</td>
<td align="center">1196243</td>
<td align="center">50.32</td>
<td align="center">22.62</td>
<td align="center">436</td>
<td align="center">3.59469</td>
<td align="center">1489</td>
<td align="center">420</td>
<td align="center">653.243</td>
<td align="center">433</td>
</tr>
<tr>
<td align="left">mean</td>
<td align="center">3721581.50</td>
<td align="center">592845.67</td>
<td align="center">50.62</td>
<td align="center">19.58</td>
<td align="center">306.83</td>
<td align="center">2.530</td>
<td align="center">713.00</td>
<td align="center">195.50</td>
<td align="center">386.99</td>
<td align="center">262.33</td>
</tr>
<tr>
<td align="left">sd</td>
<td align="center">1724152.03</td>
<td align="center">412671.39</td>
<td align="center">6.86</td>
<td align="center">17.52</td>
<td align="center">142.15</td>
<td align="center">1.171994</td>
<td align="center">516.42</td>
<td align="center">137.62</td>
<td align="center">184.46</td>
<td align="center">112.1</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Where n represents the number of data, % represents the percentage and k represents kilobytes.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s2-2">
<title>2.2 Methods</title>
<sec id="s2-2-1">
<title>2.2.1 Compression algorithm</title>
<p>COO and CSC are two common compression methods for sparse data. COO uses a triplet to store information about the non-zero elements of the matrix, storing the row subscripts, column subscripts and values of the non-zero elements respectively. The non-zero elements are found by traversing the rows and columns and storing the corresponding number of rows, columns and values in the corresponding arrays. Let A &#x2208; Rm&#xd7;<italic>n</italic> be a sparse matrix where the number of non-zero elements. Using the COO storage method, A can be stored as three vectors (I, J, V). Where I and J store the coordinates of the rows and columns of the non-zero elements respectively, and V stores the values of the non-zero elements. Examples of mathematical formulas are as follows:<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>00</mml:mn>
</mml:msub>
</mml:mtd>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>02</mml:mn>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>12</mml:mn>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>20</mml:mn>
</mml:msub>
</mml:mtd>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>21</mml:mn>
</mml:msub>
</mml:mtd>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>22</mml:mn>
</mml:msub>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x21d2;</mml:mo>
<mml:mi>I</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>2</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>2</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>2</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>J</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>2</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>2</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>2</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>V</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>00</mml:mn>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>02</mml:mn>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>12</mml:mn>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>20</mml:mn>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>21</mml:mn>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>22</mml:mn>
</mml:msub>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
<p>Data can be converted to other storage formats by COO method quickly and easily, and data can be quickly converted with compressed sparse row format (CSR)/CSC formats and can be repeatedly indexed. However, the COO format is almost impossible to manipulate or matrix-operate except by converting it to other formats.</p>
<p>The CSC is compressed and stored according to the principle of data column precedence. The matrix is determined by the row indexes of non-zero elements, index pointers, and non-zero data. Suppose an m &#xd7; <italic>n</italic> sparse matrix, with A<sub>ij</sub> denoting the elements of row i and column j. CSC can store A as three vectors (in_dices, indptr and value). Where in_dices is the row index of the non-zero elements, indptr is an array of index pointers and value is the non-zero data in the matrix. The steps are as follows:<list list-type="simple">
<list-item>
<p>1. Get the row index of the non-zero element in column i according to indices [indptr[i]: indptr[i&#x2b;1]].</p>
</list-item>
<list-item>
<p>2. Get the number of non-zero elements in column i according to [indptr[i]: indptr[i&#x2b;1]].</p>
</list-item>
<list-item>
<p>3. The column index and row index are obtained and the corresponding data is stored in: value [indptr[i]: indptr[i&#x2b;1]]. The following mathematical formula is an example:</p>
</list-item>
</list>
<disp-formula id="e2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>00</mml:mn>
</mml:msub>
</mml:mtd>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>02</mml:mn>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>12</mml:mn>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>20</mml:mn>
</mml:msub>
</mml:mtd>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>21</mml:mn>
</mml:msub>
</mml:mtd>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>22</mml:mn>
</mml:msub>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x21d2;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>2</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>3</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>6</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>_</mml:mo>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>2</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>2</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>2</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>v</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>00</mml:mn>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>20</mml:mn>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>21</mml:mn>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>02</mml:mn>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>12</mml:mn>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:mn>22</mml:mn>
</mml:msub>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
</p>
<p>The CSC data format performs efficient column slicing, but the inner matrix product and row slicing operations are relatively slow.</p>
</sec>
<sec id="s2-2-2">
<title>2.2.2 CA_SAGM algorithm</title>
<p>CA_SAGM is an optimization algorithm based on compressed sparse row format, which is implemented by optimizing the matrix ordering for the characteristics of variable genomic data. The process is as follows: first, the variant genomic data is sorted by row-major order so that adjacent non-zero elements are also physically stored as close as possible to each other. Then, the reverse Cuthill-McKee sorting algorithm is used to renumber the rows and columns of the data according to the sorting results. Finally, using the new row and column numbering, the sparse matrix is CSR compressed and stored in a file.</p>
<p>Reverse Cuthill-Mckee sorting is an algorithm that can be used to optimize the storage of sparse matrices by rearranging the rows and columns of a sparse matrix so that the matrix has a smaller bandwidth. Bandwidth is understood to be the widest diagonal distance between the non-zero elements of a matrix and has a significant impact on the efficiency of computational operations such as matrix multiplication. The basic idea of the RCM sorting algorithm is to reduce the bandwidth of a matrix by arranging interconnected points as close to each other as possible. The sparse matrix is first transformed into an undirected graph, and then this graph is traversed and pruned as a way to determine the new order of nodes, which in turn leads to the rearranged matrix. Specifically, when a node is processed, the traversal of that branch is stopped if the number of remaining nodes is not sufficient to cause a smaller bandwidth to the already traversed nodes. In addition, the RCM sorting algorithm can also use other heuristic rules such as degree sorting and greedy strategy to further improve the efficiency and quality of matrix reordering. The main ideological steps of the RCM algorithm are as follows:<list list-type="simple">
<list-item>
<p>1. Select a starting point and mark it as a visited node.</p>
</list-item>
<list-item>
<p>2. Sort the nodes adjacent to this starting point in order of traversal distance from closest to farthest.</p>
</list-item>
<list-item>
<p>3. Recursively executes steps 1 and 2 for the sorted neighboring nodes.</p>
</list-item>
<list-item>
<p>4. When all adjacent nodes have been traversed, return to the previous level of nodes and continue until the last level has been traversed.</p>
</list-item>
<list-item>
<p>5. For all unvisited nodes, sort the nodes according to the depth-first traversal method, again prioritizing the nodes adjacent to the visited nodes until all nodes have been traversed.</p>
</list-item>
</list>
</p>
<p>Sparse genomic matrix data has a large bandwidth due to the dispersed arrangement of non-zero elements. With the use of reverse Cuthill-Mckee matrix bandwidth compression, the bandwidth of the matrix is greatly reduced, and the scattered non-zero elements all converge towards the diagonal, which greatly improves computational efficiency and reduces memory requirements according to the relationship between computational complexity of lower-upper (LU) decomposition and memory requirements and bandwidth, which is followed by LU decomposition after RCM pre-processing. For most sparse matrix problems, due to the small number of elements being sorted, RCM has proven to be a more efficient algorithm in practice, as neither quick sort nor merge sort is as fast. It performs as fast as traditional execution, but with no reduction in speed for problems with a high number of nodes. The steps of the reverse Cuthill-Mckee algorithm are as follows:<list list-type="simple">
<list-item>
<p>1. Instantiate an empty queue Q for the alignment of the object R.</p>
</list-item>
<list-item>
<p>2. Find the object with the smallest degree whose index has not been added to R. Assume that the object corresponding to row p has been identified as the object with the smallest degree. Add p to R. (The degree of a node is defined as the sum of the non-diagonal elements in the corresponding row.)</p>
</list-item>
<list-item>
<p>3. Add the index to R, and add all neighbors of the corresponding object at the index, in increasing order to Q. Neighbors are nodes with non-zero values between them.</p>
</list-item>
<list-item>
<p>4. Extract the first node in Q, e.g., C. Insert C into R if it has not already been inserted, then add Q&#x2019;s C neighbors to Q in increasing order.</p>
</list-item>
<list-item>
<p>5. If Q is not empty, repeat step4.</p>
</list-item>
<list-item>
<p>6. If Q is empty, but there are objects in the matrix that are not yet included in R, start again from Step2.</p>
</list-item>
<list-item>
<p>7. Until all objects are contained in R terminate the algorithm.</p>
</list-item>
</list>
</p>
</sec>
</sec>
<sec id="s2-3">
<title>2.3 Performance evaluation metrics</title>
<p>A number of metrics were used to evaluate the compression and decompression performance between COO, CSC and CA_SAGM. Compression time (CT, Milliseconds or Seconds), compression rate (CR, Megabytes/Second), compression memory (CM, Kilobytes or Megabytes), compression ratio (CRO), decompression time (DCT, Milliseconds) and decompression rate (DCR, Megabytes/Second) are included. These parameters are calculated in Eqs <xref ref-type="disp-formula" rid="e3">3&#x2013;8</xref>.<disp-formula id="e3">
<mml:math id="m3">
<mml:mrow>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
<disp-formula id="e4">
<mml:math id="m4">
<mml:mrow>
<mml:mi mathvariant="normal">D</mml:mi>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">D</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">c</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="normal">D</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">c</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
<disp-formula id="e5">
<mml:math id="m5">
<mml:mrow>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">z</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mo>/</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi mathvariant="normal">T</mml:mi>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
<disp-formula id="e6">
<mml:math id="m6">
<mml:mrow>
<mml:mi mathvariant="normal">D</mml:mi>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">D</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">c</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">z</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mo>/</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">D</mml:mi>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi mathvariant="normal">T</mml:mi>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
<disp-formula id="e7">
<mml:math id="m7">
<mml:mrow>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">y</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">z</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">f</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">c</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
<disp-formula id="e8">
<mml:math id="m8">
<mml:mrow>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mi mathvariant="normal">O</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="normal">c</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">y</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mo>/</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="normal">c</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">y</mml:mi>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>
</p>
<p>The above metrics allow the compression algorithms to be evaluated in terms of the speed at which the data is compressed/decompressed for work, the amount of data, the memory space occupied and other different aspects. In general, shorter CT and DCT, faster CR and DCR, smaller CM and larger CRO represent better compression and decompression performance. And, we performed a statistical analysis of the experimental results. However, we can also evaluate algorithms based on different data, different usage scenarios and requirements. Different compression algorithms will perform differently in these performance metrics, users will need to choose the right algorithm for their specific scenario and needs.</p>
</sec>
</sec>
<sec id="s3">
<title>3 Experiments and results</title>
<p>In order to objectively compare the performance metrics of the different algorithms, all experiments were conducted in the same environmental configuration. The system configuration used in this study is Windows 10 (Microsoft Corporation, United States), CPU: Inter(R) Core(TM) I5-10500, 3.10 GHz; RAM: 8&#xa0;G. The compression algorithm processing software is MATLAB R2022a (Mathworks. United States). And the statistical analysis software is IBM SPSS Statistics 26 (IBM Corp. United States). No other applications were run during any of the programs to ensure a consistent working environment.</p>
<sec id="s3-1">
<title>3.1 SNV data compression performance</title>
<sec id="s3-1-1">
<title>3.1.1 Comparison of SNV data compression algorithms</title>
<p>The general process of processing SNV data includes data read-in, pre-processing, compression and storage. The original SNV data is read in and tested for basic characteristics, including data set size (n), non-zero number (n), sparsity (%), rows (n), rows/columns (%), file size (K), L1-norm, L2-norm and rank. First, the SNV data runs the COO and CSC programs separately. The sparse data matrix was then preprocessed by row-first sorting and RCM sorting successively. Next, SNV data were run through CA_SAGM compression programs. Compression time, decompression time, compression rate, decompression rate, compression memory and compression ratio are respectively obtained by the three methods. The results are shown in <xref ref-type="fig" rid="F1">Figure 1</xref>. Finally, the compressed data were stored in a suitable location. The experimental results were in mean &#xb1; SD (Standard deviation, SD) format, and were analyzed by comparing the evaluation indexes among different algorithms and using statistical methods.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Compares the compression and decompression metrics of COO, CSC and CA_SAGM for SNV. Where <bold>(A)</bold> stands for compression time, <bold>(B)</bold> for decompression time, <bold>(C)</bold> for compression speed, <bold>(D)</bold> for decompression speed, <bold>(E)</bold> for compressed memory and <bold>(F)</bold> for compression ratio.</p>
</caption>
<graphic xlink:href="fgene-14-1213907-g001.tif"/>
</fig>
<p>As can be seen from the <xref ref-type="fig" rid="F1">Figure 1</xref>, the COO algorithm has the shortest CT (5.91 &#xb1; 2.42 vs. 184.61 &#xb1; 142.89 vs. 12.25 &#xb1; 5.81), the largest CR (4989.43 &#xb1; 2753.14 vs. 238.85 &#xb1; 153.2 vs. 2377.17 &#xb1; 1093.17), the smallest CM (509.53 &#xb1; 472.56 vs. 604.55 &#xb1; 472.59 vs. 511.97 &#xb1; 472.65), CRO was the largest (170.31 &#xb1; 192.38 vs. 70.8 &#xb1; 46.65 vs. 164.48 &#xb1; 181.78), compression performance was the best. However, decompression took the longest to recover the original data (30.76 &#xb1; 23.89 vs. 21.33 &#xb1; 9.42 vs. 7.96 &#xb1; 3.32) and had a smaller decompression rate (1596.72 &#xb1; 1187.87 vs. 1389.44 &#xb1; 629.08 vs. 3467.85 &#xb1; 1246.34). The performance of CSC was the opposite of COO, CT was the longest (184.61 &#xb1; 142.89), CR was the lowest (238.85 &#xb1; 153.2), CM was the largest (604.55 &#xb1; 472.59) and CRO was the smallest (70.8 &#xb1; 46.65). The decompression performance of CSC is between COO and CA_SAGM, with DCT and DCR both performing in the middle. In addition, CA_SAGM has the best decompression performance, with the shortest DCT (7.96 &#xb1; 3.32) and the largest DCR (3467.85 &#xb1; 1246.34). If the overall total time of compression and decompression time, the average rate of compression rate and decompression rate are considered, it is clear that the CA_SAGM algorithm has the shortest total time and the largest average rate.</p>
<p>A paired sample <italic>t</italic>-test was used to assess whether there were differences in the same metrics between any two algorithms. The results show that there is a significant difference (<italic>p</italic> &#x3c; 0.05) between any two algorithms for almost all metrics: compression time (COO to CSC: 0.005; COO to CA_SAGM: 0.001; CA_SAGM to CSC: 0.006), decompression time (COO to CSC: 0.111; COO to CA_SAGM: 0.013; CA_SAGM to CSC: 0.000), compression rate (COO to CSC: 0.001; COO to CA_SAGM: 0.003; CA_SAGM to CSC: 0.000), decompression rate (COO to CSC: 0.493; COO to CA_SAGM: 0.001; CA_SAGM to CSC: 0.000), compression memory (COO to CSC: 0.000; COO to CA_SAGM: 0.000; CA_SAGM to CSC: 0.000), compression ratio (COO to CSC: 0.003; COO to CA_SAGM: 0.000; CA_SAGM to CSC: 0.003). There is little difference between COO and CSC in terms of compression time and decompression speed.</p>
</sec>
<sec id="s3-1-2">
<title>3.1.2 Correlation analysis of SNV data</title>
<p>We used spearman correlation analysis to investigate whether the compression and decompression performance was correlated with the basic characteristics of the original SNV data. <xref ref-type="table" rid="T3">Table 3</xref> shows that the compression time, decompression time, compression rate, decompression rate, compression memory and compression ratio are all correlated with the non-zero number of the original data, sparsity, file size, L1-norm and L2-norm. There was a strong correlation between sparsity and the non-zero number of raw data (<italic>p</italic> &#x3d; 0.983), file size (<italic>p</italic> &#x3d; 0.967), L1-norm (<italic>p</italic> &#x3d; 0.983) and L2-norm (<italic>p</italic> &#x3d; 0.983).</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Spearman correlation analysis between compression and decompression metrics of COO, CSC and CA_SAGM algorithms for SNV data and basic characteristics of the original data.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Index</th>
<th align="left">Data set size(n)</th>
<th align="left">Non-zero number(n)</th>
<th align="left">Sparsity (%)</th>
<th align="left">Rows (n)</th>
<th align="left">Row/column (%)</th>
<th align="left">File size (K)</th>
<th align="left">L1-norm</th>
<th align="left">L2-norm</th>
<th align="left">Rank</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">1_CT</td>
<td align="left">0.167</td>
<td align="left">.983&#x2a;&#x2a;</td>
<td align="left">.967&#x2a;&#x2a;</td>
<td align="left">0.167</td>
<td align="left">0.167</td>
<td align="left">.967&#x2a;&#x2a;</td>
<td align="left">0.617</td>
<td align="left">.933&#x2a;&#x2a;</td>
<td align="left">0.167</td>
</tr>
<tr>
<td align="left">2_CT</td>
<td align="left">0.233</td>
<td align="left">1.000&#x2a;&#x2a;</td>
<td align="left">.983&#x2a;&#x2a;</td>
<td align="left">0.233</td>
<td align="left">0.233</td>
<td align="left">.983&#x2a;&#x2a;</td>
<td align="left">.667&#x2a;</td>
<td align="left">.950&#x2a;&#x2a;</td>
<td align="left">0.233</td>
</tr>
<tr>
<td align="left">3_CT</td>
<td align="left">0.467</td>
<td align="left">.933&#x2a;&#x2a;</td>
<td align="left">.883&#x2a;&#x2a;</td>
<td align="left">0.467</td>
<td align="left">0.467</td>
<td align="left">.950&#x2a;&#x2a;</td>
<td align="left">.700&#x2a;</td>
<td align="left">.850&#x2a;&#x2a;</td>
<td align="left">0.467</td>
</tr>
<tr>
<td align="left">1_DCT</td>
<td align="left">0.333</td>
<td align="left">.983&#x2a;&#x2a;</td>
<td align="left">.967&#x2a;&#x2a;</td>
<td align="left">0.333</td>
<td align="left">0.333</td>
<td align="left">1.000&#x2a;&#x2a;</td>
<td align="left">.717&#x2a;</td>
<td align="left">.933&#x2a;&#x2a;</td>
<td align="left">0.333</td>
</tr>
<tr>
<td align="left">2_DCT</td>
<td align="left">0.45</td>
<td align="left">.950&#x2a;&#x2a;</td>
<td align="left">.917&#x2a;&#x2a;</td>
<td align="left">0.45</td>
<td align="left">0.45</td>
<td align="left">.983&#x2a;&#x2a;</td>
<td align="left">.733&#x2a;</td>
<td align="left">.883&#x2a;&#x2a;</td>
<td align="left">0.45</td>
</tr>
<tr>
<td align="left">3_DCT</td>
<td align="left">.717&#x2a;</td>
<td align="left">.750&#x2a;</td>
<td align="left">.717&#x2a;</td>
<td align="left">.717&#x2a;</td>
<td align="left">.717&#x2a;</td>
<td align="left">.833&#x2a;&#x2a;</td>
<td align="left">.667&#x2a;</td>
<td align="left">.683&#x2a;</td>
<td align="left">.717&#x2a;</td>
</tr>
<tr>
<td align="left">1_CM</td>
<td align="left">0.233</td>
<td align="left">1.000&#x2a;&#x2a;</td>
<td align="left">.983&#x2a;&#x2a;</td>
<td align="left">0.233</td>
<td align="left">0.233</td>
<td align="left">.983&#x2a;&#x2a;</td>
<td align="left">.667&#x2a;</td>
<td align="left">.950&#x2a;&#x2a;</td>
<td align="left">0.233</td>
</tr>
<tr>
<td align="left">2_CM</td>
<td align="left">0.233</td>
<td align="left">1.000&#x2a;&#x2a;</td>
<td align="left">.983&#x2a;&#x2a;</td>
<td align="left">0.233</td>
<td align="left">0.233</td>
<td align="left">.983&#x2a;&#x2a;</td>
<td align="left">.667&#x2a;</td>
<td align="left">.950&#x2a;&#x2a;</td>
<td align="left">0.233</td>
</tr>
<tr>
<td align="left">3_CM</td>
<td align="left">0.233</td>
<td align="left">1.000&#x2a;&#x2a;</td>
<td align="left">.983&#x2a;&#x2a;</td>
<td align="left">0.233</td>
<td align="left">0.233</td>
<td align="left">.983&#x2a;&#x2a;</td>
<td align="left">.667&#x2a;</td>
<td align="left">.950&#x2a;&#x2a;</td>
<td align="left">0.233</td>
</tr>
<tr>
<td align="left">1_CR</td>
<td align="left">.833&#x2a;&#x2a;</td>
<td align="left">&#x2212;0.3</td>
<td align="left">&#x2212;0.35</td>
<td align="left">.833&#x2a;&#x2a;</td>
<td align="left">.833&#x2a;&#x2a;</td>
<td align="left">&#x2212;0.217</td>
<td align="left">0.033</td>
<td align="left">&#x2212;0.283</td>
<td align="left">.833&#x2a;&#x2a;</td>
</tr>
<tr>
<td align="left">2_CR</td>
<td align="left">&#x2212;0.033</td>
<td align="left">&#x2212;.933&#x2a;&#x2a;</td>
<td align="left">&#x2212;.950&#x2a;&#x2a;</td>
<td align="left">&#x2212;0.033</td>
<td align="left">&#x2212;0.033</td>
<td align="left">&#x2212;.917&#x2a;&#x2a;</td>
<td align="left">&#x2212;0.6</td>
<td align="left">&#x2212;.883&#x2a;&#x2a;</td>
<td align="left">&#x2212;0.033</td>
</tr>
<tr>
<td align="left">3_CR</td>
<td align="left">0.45</td>
<td align="left">&#x2212;0.55</td>
<td align="left">&#x2212;0.517</td>
<td align="left">0.45</td>
<td align="left">0.45</td>
<td align="left">&#x2212;0.517</td>
<td align="left">&#x2212;0.133</td>
<td align="left">&#x2212;0.483</td>
<td align="left">0.45</td>
</tr>
<tr>
<td align="left">1_DCR</td>
<td align="left">&#x2212;0.167</td>
<td align="left">&#x2212;.983&#x2a;&#x2a;</td>
<td align="left">&#x2212;1.000&#x2a;&#x2a;</td>
<td align="left">&#x2212;0.167</td>
<td align="left">&#x2212;0.167</td>
<td align="left">&#x2212;.967&#x2a;&#x2a;</td>
<td align="left">&#x2212;.717&#x2a;</td>
<td align="left">&#x2212;.933&#x2a;&#x2a;</td>
<td align="left">&#x2212;0.167</td>
</tr>
<tr>
<td align="left">2_DCR</td>
<td align="left">0.333</td>
<td align="left">&#x2212;.733&#x2a;</td>
<td align="left">&#x2212;.783&#x2a;</td>
<td align="left">0.333</td>
<td align="left">0.333</td>
<td align="left">&#x2212;.700&#x2a;</td>
<td align="left">&#x2212;0.333</td>
<td align="left">&#x2212;.717&#x2a;</td>
<td align="left">0.333</td>
</tr>
<tr>
<td align="left">3_DCR</td>
<td align="left">0.467</td>
<td align="left">&#x2212;0.6</td>
<td align="left">&#x2212;0.617</td>
<td align="left">0.467</td>
<td align="left">0.467</td>
<td align="left">&#x2212;0.55</td>
<td align="left">&#x2212;0.167</td>
<td align="left">&#x2212;0.55</td>
<td align="left">0.467</td>
</tr>
<tr>
<td align="left">1_CRO</td>
<td align="left">0.167</td>
<td align="left">.983&#x2a;&#x2a;</td>
<td align="left">1.000&#x2a;&#x2a;</td>
<td align="left">0.167</td>
<td align="left">0.167</td>
<td align="left">.967&#x2a;&#x2a;</td>
<td align="left">.717&#x2a;</td>
<td align="left">.933&#x2a;&#x2a;</td>
<td align="left">0.167</td>
</tr>
<tr>
<td align="left">2_CRO</td>
<td align="left">&#x2212;0.117</td>
<td align="left">.867&#x2a;&#x2a;</td>
<td align="left">.883&#x2a;&#x2a;</td>
<td align="left">&#x2212;0.117</td>
<td align="left">&#x2212;0.117</td>
<td align="left">.850&#x2a;&#x2a;</td>
<td align="left">0.533</td>
<td align="left">.783&#x2a;</td>
<td align="left">&#x2212;0.117</td>
</tr>
<tr>
<td align="left">3_CRO</td>
<td align="left">0.167</td>
<td align="left">.983&#x2a;&#x2a;</td>
<td align="left">1.000&#x2a;&#x2a;</td>
<td align="left">0.167</td>
<td align="left">0.167</td>
<td align="left">.967&#x2a;&#x2a;</td>
<td align="left">.717&#x2a;</td>
<td align="left">.933&#x2a;&#x2a;</td>
<td align="left">0.167</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Where 1_ represents the COO compression algorithm, 2_ represents the CSC compression algorithm and 3_ represents the CA_SAGM compression algorithm. &#x2a;&#x2a; At level 0.01, the correlation was significant.&#x2a; At level 0.05, the correlation was significant.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>As sparsity is easy to calculate and obtain, we further analyzed the effect of sparsity on the SNV data, as shown in <xref ref-type="fig" rid="F2">Figure 2</xref>. As can be seen from the figure, CSC compression performance performs the worst, with the longest CT, the smallest CR and the smallest CRO. Both COO and CA_SAGM show better compression characteristics, with shorter CT and larger CR. In terms of decompression, COO performs the worst, with the longest DCT and smallest DCR. CA_SAGM performs the best, with the shortest DCT and largest DCR, CSC performs in the middle. The difference between the compression &#x26; decompression performance of COO, CSC and CA_SAGM is small when the sparsity is close to 0. As the data sparsity increases (but the sparsity is still small, &#x3c;2%), the compression &#x26; decompression time tends to become larger, the compression and decompression rate tends to decrease, and the compression ratio also tends to decrease. The difference in compression and decompression times between algorithms increases with sparsity.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Curves of compression and decompression metrics vs. sparsity variation for COO, CSC and CA_SAGM for SNV. Where <bold>(A)</bold> stands for compression time, <bold>(B)</bold> for decompression time, <bold>(C)</bold> for compression speed, <bold>(D)</bold> for decompression speed, <bold>(E)</bold> for compressed memory and <bold>(F)</bold> for compression ratio.</p>
</caption>
<graphic xlink:href="fgene-14-1213907-g002.tif"/>
</fig>
</sec>
</sec>
<sec id="s3-2">
<title>3.2 CNV data compression performance</title>
<sec id="s3-2-1">
<title>3.2.1 Comparison of CNV data compression algorithms</title>
<p>CNV data are more complex than SNV data, with larger datasets, a larger number of non-zeros and greater sparsity. Thus, we further investigated and analyzed the experimental results of the CNV data. Similarly, the process of processing CNV data includes steps such as data read-in, pre-processing, compression and storage. The raw CNV data is read in and tested for basic characteristics, including data set size (n), non-zero number (n), sparsity (%), rows (n), rows/columns (%), file size (K), L1-norm, L2-norm and rank. First, the CNV data runs the COO and CSC programs separately. The sparse data matrix was then preprocessed by row-first sorting and RCM sorting successively. Next, SNV data were run through CA_SAGM compression programs. Compression time, decompression time, compression rate, decompression rate, compression memory and compression ratio are respectively obtained by the three methods. The results are shown in <xref ref-type="fig" rid="F3">Figure 3</xref>. Finally, the compressed data were stored in a suitable location. The experimental results were in mean &#xb1; SD, and were analyzed by comparing the evaluation indexes among different algorithms and using statistical methods.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Compares the compression and decompression metrics of COO, CSC and CA_SAGM for CNV. Where <bold>(A)</bold> stands for compression time, <bold>(B)</bold> for decompression time, <bold>(C)</bold> for compression speed, <bold>(D)</bold> for decompression speed, <bold>(E)</bold> for compressed memory and <bold>(F)</bold> for compression ratio.</p>
</caption>
<graphic xlink:href="fgene-14-1213907-g003.tif"/>
</fig>
<p>From the <xref ref-type="fig" rid="F3">Figure 3</xref>, we can see that in terms of compression performance, COO performs the best with the shortest CT (0.11 &#xb1; 0.06 vs. 4.51 &#xb1; 3.71 vs. 0.24 &#xb1; 0.21) and the largest CR (357.02 &#xb1; 337.97 vs. 12.72 &#xb1; 12.72 vs. 238.27 &#xb1; 240.35). CSC has the worst compression performance with the longest CT and the smallest CR. CA_SAGM had the middle compression performance. However, the CM (16.11 &#xb1; 12.45 vs. 16.11 &#xb1; 12.45 vs. 16.11 &#xb1; 12.45) and CRO (0.62 &#xb1; 0.41 vs. 0.62 &#xb1; 0.41 vs. 0.62 &#xb1; 0.41) were the same after compression by the three methods, which may be associated with a larger sparsity (19.58% &#xb1; 17.52%). In terms of decompression, COO had the worst performance, with the longest DCT (0.86 &#xb1; 0.59 vs. 0.09 &#xb1; 0.04 vs. 0.07 &#xb1; 0.04) and the smallest DCR (65.71 &#xb1; 67.19 vs. 375.74 &#xb1; 252.88 vs. 639.42 &#xb1; 553.6). CA_SAGM had the best decompression performance, with the shortest DCT and the smallest DCR. CSC decompression performance in the middle.</p>
<p>Similarly, a paired sample <italic>t</italic>-test was used to assess whether there were differences between any two algorithms for the same metrics. The results showed that almost all metrics were significantly different between any two algorithms (<italic>p</italic> &#x3c; 0.05), with the exception of compression memory and compression ratio (<italic>p</italic> &#x3e; 0.05). The detailed analysis results are as follows: Compression time (COO to CSC: 0.032; COO to CA_SAGM: 0.087; CA_SAGM to CSC: 0.031), decompression time (COO to CSC: 0.018; COO to CA_SAGM: 0.016; CA_SAGM to CSC: 0.000), compression rate (COO to CSC: 0.05; COO to CA_SAGM: 0.357; CA_SAGM to CSC: 0.06), decompression rate (COO to CSC: 0.011; COO to CA_SAGM: 0.034; CA_SAGM to CSC: 0.115), compression memory (COO to CSC: 0.018; COO to CA_SAGM: 0.002; CA_SAGM to CSC: 0.000), compression ratio (COO to CSC: 0.006; COO to CA_SAGM: 0.000; CA_SAGM to CSC: 0.007).</p>
</sec>
<sec id="s3-2-2">
<title>3.2.2 Correlation analysis of CNV data</title>
<p>Spearman correlation analysis was used to investigate whether the compression and decompression performance was correlated with the basic characteristics of the CNV raw data (see <xref ref-type="table" rid="T4">Table 4</xref>). The results show that CT, DCT, CR, DCR, CM and CRO all have large correlation coefficients with the non-zero number, sparsity and L2-norm of the original data. In addition, CT, DCT and CM are strongly correlated with data file size and L1-norm. Also, there was a strong correlation between sparsity, non-zero number (<italic>p</italic> &#x3d; 0.771) and L2-norm (<italic>p</italic> &#x3d; 0.714). There was also a strong correlation between file size and L1-norm (<italic>p</italic> &#x3d; 0.943).</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Spearman correlation analysis between compression and decompression metrics of COO, CSC and CA_SAGM algorithms for CNV data and basic characteristics of the original data.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Index</th>
<th align="left">Data set size(n)</th>
<th align="left">Non-zero number(n)</th>
<th align="left">Sparsity (%)</th>
<th align="left">Rows (n)</th>
<th align="left">Row/column (%)</th>
<th align="left">Non-negative ratio (%)</th>
<th align="left">File size (K)</th>
<th align="left">L1-norm</th>
<th align="left">L2-norm</th>
<th align="left">Rank</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">1_CT</td>
<td align="left">&#x2212;0.257</td>
<td align="left">.943&#x2a;&#x2a;</td>
<td align="left">.829&#x2a;</td>
<td align="left">&#x2212;0.257</td>
<td align="left">&#x2212;0.257</td>
<td align="left">&#x2212;0.143</td>
<td align="left">0.771&#x2a;</td>
<td align="left">0.714&#x2a;</td>
<td align="left">.886&#x2a;</td>
<td align="left">0.143</td>
</tr>
<tr>
<td align="left">2_CT</td>
<td align="left">&#x2212;0.029</td>
<td align="left">1.000&#x2a;&#x2a;</td>
<td align="left">0.771&#x2a;</td>
<td align="left">&#x2212;0.029</td>
<td align="left">&#x2212;0.029</td>
<td align="left">0.086</td>
<td align="left">.829&#x2a;</td>
<td align="left">0.771&#x2a;</td>
<td align="left">.943&#x2a;&#x2a;</td>
<td align="left">0.257</td>
</tr>
<tr>
<td align="left">3_CT</td>
<td align="left">0.257</td>
<td align="left">.829&#x2a;</td>
<td align="left">0.543</td>
<td align="left">0.257</td>
<td align="left">0.257</td>
<td align="left">&#x2212;0.029</td>
<td align="left">1.000&#x2a;&#x2a;</td>
<td align="left">.943&#x2a;&#x2a;</td>
<td align="left">.943&#x2a;&#x2a;</td>
<td align="left">0.6</td>
</tr>
<tr>
<td align="left">1_DCT</td>
<td align="left">&#x2212;0.029</td>
<td align="left">1.000&#x2a;&#x2a;</td>
<td align="left">0.771&#x2a;</td>
<td align="left">&#x2212;0.029</td>
<td align="left">&#x2212;0.029</td>
<td align="left">0.086</td>
<td align="left">.829&#x2a;</td>
<td align="left">0.771&#x2a;</td>
<td align="left">.943&#x2a;&#x2a;</td>
<td align="left">0.257</td>
</tr>
<tr>
<td align="left">2_DCT</td>
<td align="left">&#x2212;0.029</td>
<td align="left">1.000&#x2a;&#x2a;</td>
<td align="left">0.771&#x2a;</td>
<td align="left">&#x2212;0.029</td>
<td align="left">&#x2212;0.029</td>
<td align="left">0.086</td>
<td align="left">.829&#x2a;</td>
<td align="left">0.771&#x2a;</td>
<td align="left">.943&#x2a;&#x2a;</td>
<td align="left">0.257</td>
</tr>
<tr>
<td align="left">3_DCT</td>
<td align="left">&#x2212;0.029</td>
<td align="left">1.000&#x2a;&#x2a;</td>
<td align="left">0.771&#x2a;</td>
<td align="left">&#x2212;0.029</td>
<td align="left">&#x2212;0.029</td>
<td align="left">0.086</td>
<td align="left">.829&#x2a;</td>
<td align="left">0.771&#x2a;</td>
<td align="left">.943&#x2a;&#x2a;</td>
<td align="left">0.257</td>
</tr>
<tr>
<td align="left">1_CM</td>
<td align="left">&#x2212;0.029</td>
<td align="left">1.000&#x2a;&#x2a;</td>
<td align="left">0.771&#x2a;</td>
<td align="left">&#x2212;0.029</td>
<td align="left">&#x2212;0.029</td>
<td align="left">0.086</td>
<td align="left">.829&#x2a;</td>
<td align="left">0.771&#x2a;</td>
<td align="left">.943&#x2a;&#x2a;</td>
<td align="left">0.257</td>
</tr>
<tr>
<td align="left">2_CM</td>
<td align="left">&#x2212;0.029</td>
<td align="left">1.000&#x2a;&#x2a;</td>
<td align="left">0.771&#x2a;</td>
<td align="left">&#x2212;0.029</td>
<td align="left">&#x2212;0.029</td>
<td align="left">0.086</td>
<td align="left">.829&#x2a;</td>
<td align="left">0.771&#x2a;</td>
<td align="left">.943&#x2a;&#x2a;</td>
<td align="left">0.257</td>
</tr>
<tr>
<td align="left">3_CM</td>
<td align="left">&#x2212;0.029</td>
<td align="left">1.000&#x2a;&#x2a;</td>
<td align="left">0.771&#x2a;</td>
<td align="left">&#x2212;0.029</td>
<td align="left">&#x2212;0.029</td>
<td align="left">0.086</td>
<td align="left">.829&#x2a;</td>
<td align="left">0.771&#x2a;</td>
<td align="left">.943&#x2a;&#x2a;</td>
<td align="left">0.257</td>
</tr>
<tr>
<td align="left">1_CR</td>
<td align="left">0.6</td>
<td align="left">&#x2212;0.771&#x2a;</td>
<td align="left">&#x2212;.886&#x2a;</td>
<td align="left">0.6</td>
<td align="left">0.6</td>
<td align="left">0.029</td>
<td align="left">&#x2212;0.429</td>
<td align="left">&#x2212;0.314</td>
<td align="left">&#x2212;0.657</td>
<td align="left">0.314</td>
</tr>
<tr>
<td align="left">2_CR</td>
<td align="left">0.486</td>
<td align="left">&#x2212;.886&#x2a;</td>
<td align="left">&#x2212;.943&#x2a;&#x2a;</td>
<td align="left">0.486</td>
<td align="left">0.486</td>
<td align="left">0.086</td>
<td align="left">&#x2212;0.6</td>
<td align="left">&#x2212;0.543</td>
<td align="left">&#x2212;0.771&#x2a;</td>
<td align="left">0.143</td>
</tr>
<tr>
<td align="left">3_CR</td>
<td align="left">0.257</td>
<td align="left">&#x2212;.943&#x2a;&#x2a;</td>
<td align="left">&#x2212;.886&#x2a;</td>
<td align="left">0.257</td>
<td align="left">0.257</td>
<td align="left">&#x2212;0.143</td>
<td align="left">&#x2212;0.657</td>
<td align="left">&#x2212;0.6</td>
<td align="left">&#x2212;.829&#x2a;</td>
<td align="left">0.029</td>
</tr>
<tr>
<td align="left">1_DCR</td>
<td align="left">0.6</td>
<td align="left">&#x2212;0.771&#x2a;</td>
<td align="left">&#x2212;1.000&#x2a;&#x2a;</td>
<td align="left">0.6</td>
<td align="left">0.6</td>
<td align="left">&#x2212;0.029</td>
<td align="left">&#x2212;0.543</td>
<td align="left">&#x2212;0.429</td>
<td align="left">&#x2212;0.714&#x2a;</td>
<td align="left">0.314</td>
</tr>
<tr>
<td align="left">2_DCR</td>
<td align="left">0.6</td>
<td align="left">&#x2212;0.771&#x2a;</td>
<td align="left">&#x2212;1.000&#x2a;&#x2a;</td>
<td align="left">0.6</td>
<td align="left">0.6</td>
<td align="left">&#x2212;0.029</td>
<td align="left">&#x2212;0.543</td>
<td align="left">&#x2212;0.429</td>
<td align="left">&#x2212;0.714&#x2a;</td>
<td align="left">0.314</td>
</tr>
<tr>
<td align="left">3_DCR</td>
<td align="left">0.6</td>
<td align="left">&#x2212;0.771&#x2a;</td>
<td align="left">&#x2212;1.000&#x2a;&#x2a;</td>
<td align="left">0.6</td>
<td align="left">0.6</td>
<td align="left">&#x2212;0.029</td>
<td align="left">&#x2212;0.543</td>
<td align="left">&#x2212;0.429</td>
<td align="left">&#x2212;0.714&#x2a;</td>
<td align="left">0.314</td>
</tr>
<tr>
<td align="left">1_CRO</td>
<td align="left">&#x2212;0.6</td>
<td align="left">0.771&#x2a;</td>
<td align="left">1.000&#x2a;&#x2a;</td>
<td align="left">&#x2212;0.6</td>
<td align="left">&#x2212;0.6</td>
<td align="left">0.029</td>
<td align="left">0.543</td>
<td align="left">0.429</td>
<td align="left">0.714&#x2a;</td>
<td align="left">&#x2212;0.314</td>
</tr>
<tr>
<td align="left">2_CRO</td>
<td align="left">&#x2212;0.6</td>
<td align="left">0.771&#x2a;</td>
<td align="left">1.000&#x2a;&#x2a;</td>
<td align="left">&#x2212;0.6</td>
<td align="left">&#x2212;0.6</td>
<td align="left">0.029</td>
<td align="left">0.543</td>
<td align="left">0.429</td>
<td align="left">0.714&#x2a;</td>
<td align="left">&#x2212;0.314</td>
</tr>
<tr>
<td align="left">3_CRO</td>
<td align="left">&#x2212;0.6</td>
<td align="left">0.771&#x2a;</td>
<td align="left">1.000&#x2a;&#x2a;</td>
<td align="left">&#x2212;0.6</td>
<td align="left">&#x2212;0.6</td>
<td align="left">0.029</td>
<td align="left">0.543</td>
<td align="left">0.429</td>
<td align="left">0.714&#x2a;</td>
<td align="left">&#x2212;0.314</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Where 1_ represents the COO compression algorithm, 2_ represents the CSC compression algorithm and 3_ represents the CA_SAGM compression algorithm. &#x2a;&#x2a; At level 0.01, the correlation was significant.&#x2a; At level 0.05, the correlation was significant.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Similarly, we have further analyzed the effect of the variation of CNV data sparsity on the experimental results, as shown in <xref ref-type="fig" rid="F4">Figure 4</xref>. It can also be seen from the figure that in terms of compression performance, CSC has the worst compression characteristics, with the longest CT and the smallest CR. While both COO and CA_SAGM show better compression characteristics, with shorter CT and larger CR, with less difference between them. In terms of decompression, COO has the worst performance, with the longest DCT and the smallest DCR. CA_SAGM shows the best decompression characteristics, with the shortest DCT and the largest DCR. CSC decompression characteristics are between COO and CA_SAGM. When the sparsity is relatively small, the difference in compression and decompression performance between COO, CSC and CA_SAGM is small. The difference in compression and decompression time between CSC, COO and CA_SAGM increases as the sparsity increases. However, the difference between CR and DCR decreases with increasing sparsity.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Curves of compression and decompression metrics vs. sparsity variation for COO, CSC and CA_SAGM for CNV. Where <bold>(A)</bold> stands for compression time, <bold>(B)</bold> for decompression time, <bold>(C)</bold> for compression speed, <bold>(D)</bold> for decompression speed, <bold>(E)</bold> for compressed memory and <bold>(F)</bold> for compression ratio.</p>
</caption>
<graphic xlink:href="fgene-14-1213907-g004.tif"/>
</fig>
</sec>
</sec>
</sec>
<sec id="s4">
<title>4 Discussion and conclusion</title>
<p>In this paper, we propose a sparse asymmetric gene mutation compression algorithm CA_SAGM. The compression and decompression performance of COO, CSC and CA_SAGM is compared and analyzed using SNV and CNV data as the study objects. The results show that CA_SAGM can meet the high performance requirements of compression and decompression, achieve fast and lossless compression and decompression. In addition, it was found that the compression and decompression performance has a strong correlation with sparse. As the sparsity increases, all algorithms show longer compression and decompression times, lower compression and decompression rates, increased compression memory and lower compression ratios.</p>
<p>In our current study, CA_SAGM proved to have high compression and decompression performance for sparse genomic mutation data. CA_SAGM is a CSR compression algorithm for row-first sorting and reverse Cuthill-McKee sorting optimization. CA_SAGM has its own unique advantages over other compression algorithms. In combination with the reverse Cuthill-McKee sorting and optimization algorithm phase, the scattered non-zero elements of the data can be brought together on the diagonal and the bandwidth of the matrix is reduced considerably. Computational complexity versus memory and bandwidth based on the results of low-high (LU) decomposition. RCM pre-processing followed by LU decomposition can significantly reduce processing time, improve computational efficiency and reduce memory requirements. CA_SAGM has significant advantages in terms of compression and decompression time, as well as compression and decompression speed. CA_SAGM also has a very significant compression ratio advantage when the sparsity is low.</p>
<p>It should be noted that the results of this paper also have some limitations. Firstly, the SNV and CNV data from the experiments are limited and the sources of test data need to be expanded. Secondly, the data were only obtained from TCGA and the rest of the databases (e.g., GEO) were not studied. Recently, dedicated and integrated tools, genetic data compression algorithms, software and methods for compression in combination with machine learning (<xref ref-type="bibr" rid="B43">Wang et al., 2019</xref>; <xref ref-type="bibr" rid="B18">Kryukov et al., 2020</xref>; <xref ref-type="bibr" rid="B6">Chen et al., 2022</xref>; <xref ref-type="bibr" rid="B30">Niu et al., 2022</xref>; <xref ref-type="bibr" rid="B50">Yao et al., 2022</xref>) have received increasing attention and application by researchers, making it possible to process huge amounts of genetic data. For example, Cui Huanyu et al. proposed a new method of matrix compression based on CSR and COO: PBC algorithm for the problem that SPMV (sparse matrix vector multiplication) computation leads to computational redundancy, storage redundancy, load imbalance and low GPU utilization (<xref ref-type="bibr" rid="B8">Cui et al., 2022</xref>). The method considers load balancing conditions during the SPMV calculation. The blocks are divided according to a row-major order strategy, ensuring that the standard deviation between each block is minimized to satisfy the maximum similarity in the number of non-zero elements between each block. The result exhibits both speed-up ratio and compression performance. For lossless compression, researchers such as Jiabing Fu recommended LCQS; a lossless compression tool specialized for quality scores (<xref ref-type="bibr" rid="B11">Fu et al., 2020</xref>). The further development of specialized and integrated tools, software and evaluation methods, combined with artificial intelligence algorithms for the analysis and processing of genetic data are also the main directions and elements of our next research work. In summary, CA_SAGM has been shown to reduce data transfer time and storage space, and improve the utilization of network and storage resources. Promoting the use of this method will make the researcher&#x2019;s work more effective and convenient.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/Supplementary Material, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s6">
<title>Author contributions</title>
<p>Conception and design: GZ and JW; Data analysis and interpretation: YD, YL, JH, JM, XW, and XL. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec id="s7">
<title>Funding</title>
<p>This work was supported by the Medical Scientific Research Foundation of Guangdong Province, China (No. B2022347).</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ball</surname>
<given-names>M. P.</given-names>
</name>
<name>
<surname>Thakuria</surname>
<given-names>J. V.</given-names>
</name>
<name>
<surname>Zaranek</surname>
<given-names>A. W.</given-names>
</name>
<name>
<surname>Clegg</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Rosenbaum</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>A public resource facilitating clinical use of genomes</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A.</source> <volume>109</volume> (<issue>30</issue>), <fpage>11920</fpage>&#x2013;<lpage>11927</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1201904109</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Belsare</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Levy-Sakin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mostovoy</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Durinck</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chaudhuri</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Evaluating the quality of the 1000 genomes project data</article-title>. <source>Bmc Genomics</source> <volume>20</volume> (<issue>1</issue>), <fpage>620</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-019-5957-x</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Boeke</surname>
<given-names>J. D.</given-names>
</name>
<name>
<surname>Church</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Hessel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kelley</surname>
<given-names>N. J.</given-names>
</name>
<name>
<surname>Arkin</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>GENOME ENGINEERING. The genome project-write</article-title>. <source>Science</source> <volume>353</volume> (<issue>6295</issue>), <fpage>126</fpage>&#x2013;<lpage>127</lpage>. <pub-id pub-id-type="doi">10.1126/science.aaf6850</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cavalli-Sforza</surname>
<given-names>L. L.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>The human genome diversity project: Past, present and future</article-title>. <source>Nat. Rev. Genet.</source> <volume>6</volume> (<issue>4</issue>), <fpage>333</fpage>&#x2013;<lpage>340</lpage>. <pub-id pub-id-type="doi">10.1038/nrg1596</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Prognostic implications of programmed death ligand 1 expression in resected lung adenocarcinoma: A systematic review and meta-analysis</article-title>. <source>Eur. J. Cardio-Thoracic Surg.</source> <volume>58</volume> (<issue>5</issue>), <fpage>888</fpage>&#x2013;<lpage>898</lpage>. <pub-id pub-id-type="doi">10.1093/ejcts/ezaa172</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Cmic: An efficient quality score compressor with random access functionality</article-title>. <source>BMC Bioinforma.</source> <volume>23</volume> (<issue>1</issue>), <fpage>294</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-022-04837-1</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Chi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gong</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>An efficient SIMD compression format for sparse matrix-vector multiplication</article-title>. <source>Concurrency Computation-Practice Exp.</source> <volume>30</volume> (<issue>23</issue>), <fpage>e4800</fpage>. <pub-id pub-id-type="doi">10.1002/cpe.4800</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cui</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>An effective SPMV based on block strategy and hybrid compression on GPU</article-title>. <source>J. Supercomput.</source> <volume>78</volume> (<issue>5</issue>), <fpage>6318</fpage>&#x2013;<lpage>6339</lpage>. <pub-id pub-id-type="doi">10.1007/s11227-021-04123-6</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fairley</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lowy-Gallego</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Perry</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Flicek</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>The International Genome Sample Resource (IGSR) collection of open human genomic variation resources</article-title>. <source>Nucleic Acids Res.</source> <volume>48</volume> (<issue>D1</issue>), <fpage>D941</fpage>&#x2013;<lpage>D947</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkz836</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fira</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Goras</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>An ECG signals compression method and its validation using NNs</article-title>. <source>Ieee Trans. Biomed. Eng.</source> <volume>55</volume> (<issue>4</issue>), <fpage>1319</fpage>&#x2013;<lpage>1326</lpage>. <pub-id pub-id-type="doi">10.1109/TBME.2008.918465</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ke</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Lcqs: An efficient lossless compression tool of quality scores with random access functionality</article-title>. <source>BMC Bioinforma.</source> <volume>21</volume> (<issue>1</issue>), <fpage>109</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-020-3428-7</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Garand</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>S. S. Y.</given-names>
</name>
<name>
<surname>Al Khodor</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A literature-based approach for curating gene signatures in multifaceted diseases</article-title>. <source>J. Transl. Med.</source> <volume>18</volume> (<issue>1</issue>), <fpage>279</fpage>. <pub-id pub-id-type="doi">10.1186/s12967-020-02408-7</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Sang</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>CNV-MEANN: A neural network and mind evolutionary algorithm-based detection of copy number variations from next-generation sequencing data</article-title>. <source>Front. Genet.</source> <volume>12</volume>, <fpage>700874</fpage>&#x2013;<lpage>708021</lpage>. <pub-id pub-id-type="doi">10.3389/fgene.2021.700874</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Janssen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ramaswami</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Davis</surname>
<given-names>E. E.</given-names>
</name>
<name>
<surname>Hurd</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Airik</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Kasanuki</surname>
<given-names>J. M.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Mutation analysis in Bardet-Biedl syndrome by DNA pooling and massively parallel resequencing in 105 individuals</article-title>. <source>Hum. Genet.</source> <volume>129</volume> (<issue>1</issue>), <fpage>79</fpage>&#x2013;<lpage>90</lpage>. <pub-id pub-id-type="doi">10.1007/s00439-010-0902-8</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jugas</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Sedlar</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Vitek</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Nykrynova</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Barton</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Bezdicek</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>CNproScan: Hybrid CNV detection for bacterial genomes</article-title>. <source>Genomics</source> <volume>113</volume> (<issue>5</issue>), <fpage>3103</fpage>&#x2013;<lpage>3111</lpage>. <pub-id pub-id-type="doi">10.1016/j.ygeno.2021.06.040</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yun</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Cho</surname>
<given-names>S. I.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>J. S.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Consistent count region-copy number variation (CCR-CNV): An expandable and robust tool for clinical diagnosis of copy number variation at the exon level using next-generation sequencing data</article-title>. <source>Genet. Med.</source> <volume>24</volume> (<issue>3</issue>), <fpage>663</fpage>&#x2013;<lpage>672</lpage>. <pub-id pub-id-type="doi">10.1016/j.gim.2021.10.025</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Koza</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Matyka</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Szkoda</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Miros&#x142;aw</surname>
<given-names>&#x141;.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Compressed multirow storage format for sparse matrices on graphics processing units</article-title>. <source>Siam J. Sci. Comput.</source> <volume>36</volume> (<issue>2</issue>), <fpage>C219</fpage>&#x2013;<lpage>C239</lpage>. <pub-id pub-id-type="doi">10.1137/120900216</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kryukov</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Ueda</surname>
<given-names>M. T.</given-names>
</name>
<name>
<surname>Nakagawa</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Imanishi</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Sequence Compression Benchmark (SCB) database-A comprehensive evaluation of reference-free compressors for FASTA-formatted sequences</article-title>. <source>Gigascience</source> <volume>9</volume> (<issue>7</issue>), <fpage>giaa072</fpage>. <pub-id pub-id-type="doi">10.1093/gigascience/giaa072</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ladeira</surname>
<given-names>G. C.</given-names>
</name>
<name>
<surname>Pilonetto</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Fernandes</surname>
<given-names>A. C.</given-names>
</name>
<name>
<surname>B&#xf3;scollo</surname>
<given-names>P. P.</given-names>
</name>
<name>
<surname>Dauria</surname>
<given-names>B. D.</given-names>
</name>
<name>
<surname>Titto</surname>
<given-names>C. G.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>CNV detection and their association with growth, efficiency and carcass traits in Santa Ines sheep</article-title>. <source>J. Animal Breed. Genet.</source> <volume>139</volume> (<issue>4</issue>), <fpage>476</fpage>&#x2013;<lpage>487</lpage>. <pub-id pub-id-type="doi">10.1111/jbg.12671</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lavrichenko</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Johansson</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jonassen</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Comprehensive characterization of copy number variation (CNV) called from array, long- and short-read data</article-title>. <source>BMC Genomics</source> <volume>22</volume> (<issue>1</issue>), <fpage>826</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-021-08082-3</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname>
<given-names>W.-P.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Cerveria</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Ryan</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>A whole-genome sequencing-based algorithm for copy number detection at clinical grade level</article-title>. <source>Genomics, proteomics Bioinforma.</source> <volume>20</volume>. <fpage>1197</fpage>. <pub-id pub-id-type="doi">10.1016/j.gpb.2021.06.003</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lewin</surname>
<given-names>H. A.</given-names>
</name>
<name>
<surname>Robinson</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Kress</surname>
<given-names>W. J.</given-names>
</name>
<name>
<surname>Baker</surname>
<given-names>W. J.</given-names>
</name>
<name>
<surname>Coddington</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Crandall</surname>
<given-names>K. A.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Earth BioGenome project: Sequencing life for the future of life</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A.</source> <volume>115</volume> (<issue>17</issue>), <fpage>4325</fpage>&#x2013;<lpage>4333</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1720115115</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Cancer classification based on multiple dimensions: SNV patterns</article-title>. <source>Comput. Biol. Med.</source> <volume>151</volume>, <fpage>106270</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2022.106270</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Tanigawa</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Narasimhan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Hastie</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Tibshirani</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Fast numerical optimization for genome sequencing data in population biobanks</article-title>. <source>Bioinformatics</source> <volume>37</volume> (<issue>22</issue>), <fpage>4148</fpage>&#x2013;<lpage>4155</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btab452</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Macintyre</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Ylstra</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Brenton</surname>
<given-names>J. D.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Sequencing structural variants in cancer for precision therapeutics</article-title>. <source>Trends Genet.</source> <volume>32</volume> (<issue>9</issue>), <fpage>530</fpage>&#x2013;<lpage>542</lpage>. <pub-id pub-id-type="doi">10.1016/j.tig.2016.07.002</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Medvedev</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Stanciu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Brudno</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Computational methods for discovering structural variation with next-generation sequencing</article-title>. <source>Nat. Methods</source> <volume>6</volume> (<issue>11</issue>), <fpage>S13</fpage>&#x2013;<lpage>S20</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.1374</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moffat</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Huffman coding</article-title>. <source>Acm Comput. Surv.</source> <volume>52</volume> (<issue>4</issue>), <fpage>1</fpage>&#x2013;<lpage>35</lpage>. <pub-id pub-id-type="doi">10.1145/3342555</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mota</surname>
<given-names>N. R.</given-names>
</name>
<name>
<surname>Franke</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>30-year journey from the start of the human genome project to clinical application of genomics in psychiatry: Are we there yet?</article-title> <source>Lancet Psychiatry</source> <volume>7</volume> (<issue>1</issue>), <fpage>7</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1016/S2215-0366(19)30477-8</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Naqvi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Naqvi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Riaz</surname>
<given-names>R. R.</given-names>
</name>
<name>
<surname>Siddiqi</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Optimized RTL design and implementation of LZW algorithm for high bandwidth applications</article-title>. <source>Przeglad Elektrotechniczny</source> <volume>87</volume> (<issue>4</issue>), <fpage>279</fpage>&#x2013;<lpage>285</lpage>.</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Niu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>ACO:lossless quality score compression based on adaptive coding order</article-title>. <source>BMC Bioinforma.</source> <volume>23</volume> (<issue>1</issue>), <fpage>219</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-022-04712-z</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Oh</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kwon</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Weir</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Ha</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>A novel method to identify high order gene-gene interactions in genome-wide association studies: Gene-based MDR</article-title>. <source>Bmc Bioinforma.</source> <volume>13</volume>, <fpage>S5</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-13-S9-S5</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Park</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yi</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Ahn</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Kung</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>J. J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Balancing computation loads and optimizing input vector loading in LSTM accelerators</article-title>. <source>Ieee Trans. Computer-Aided Des. Integr. Circuits Syst.</source> <volume>39</volume> (<issue>9</issue>), <fpage>1889</fpage>&#x2013;<lpage>1901</lpage>. <pub-id pub-id-type="doi">10.1109/tcad.2019.2926482</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Prashant</surname>
<given-names>N. M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Dillard</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Ibeawuchi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Alsaeedy</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Improved SNV discovery in barcode-stratified scRNA-seq alignments</article-title>. <source>Genes</source> <volume>12</volume> (<issue>10</issue>), <fpage>1558</fpage>. <pub-id pub-id-type="doi">10.3390/genes12101558</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Press</surname>
<given-names>M. O.</given-names>
</name>
<name>
<surname>Hall</surname>
<given-names>A. N.</given-names>
</name>
<name>
<surname>Morton</surname>
<given-names>E. A.</given-names>
</name>
<name>
<surname>Queitsch</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Substitutions are boring: Some arguments about parallel mutations and high mutation rates</article-title>. <source>Trends Genet.</source> <volume>35</volume> (<issue>4</issue>), <fpage>253</fpage>&#x2013;<lpage>264</lpage>. <pub-id pub-id-type="doi">10.1016/j.tig.2019.01.002</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Samaha</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Wade</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Mazrier</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Grueber</surname>
<given-names>C. E.</given-names>
</name>
<name>
<surname>Haase</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Exploiting genomic synteny in felidae: Cross-species genome alignments and SNV discovery can aid conservation management</article-title>. <source>Bmc Genomics</source> <volume>22</volume> (<issue>1</issue>), <fpage>601</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-021-07899-2</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schnepp</surname>
<given-names>P. M.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Keller</surname>
<given-names>E. T.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>SNV identification from single-cell RNA sequencing data</article-title>. <source>Hum. Mol. Genet.</source> <volume>28</volume> (<issue>21</issue>), <fpage>3569</fpage>&#x2013;<lpage>3583</lpage>. <pub-id pub-id-type="doi">10.1093/hmg/ddz207</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shekaramiz</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Moon</surname>
<given-names>T. K.</given-names>
</name>
<name>
<surname>Gunther</surname>
<given-names>J. H.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Bayesian compressive sensing of sparse signals with unknown clustering patterns</article-title>. <source>Entropy</source> <volume>21</volume> (<issue>3</issue>), <fpage>247</fpage>. <pub-id pub-id-type="doi">10.3390/e21030247</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stankiewicz</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Lupski</surname>
<given-names>J. R.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Structural variation in the human genome and its role in disease</article-title>. <source>Annu. Rev. Med.</source> <volume>61</volume>, <fpage>437</fpage>&#x2013;<lpage>455</lpage>. <pub-id pub-id-type="doi">10.1146/annurev-med-100708-204735</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<collab>The ICGC/TCGA Pan-Cancer Analysis of Whole Genomes Consortium</collab> (<year>2020</year>). <article-title>Pan-cancer analysis of whole genomes</article-title>. <source>Nature</source> <volume>578</volume> (<issue>7793</issue>), <fpage>82</fpage>.</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tu</surname>
<given-names>Z. D.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Further understanding human disease genes by comparing with housekeeping genes and other genes</article-title>. <source>Bmc Genomics</source> <volume>7</volume>, <fpage>31</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2164-7-31</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>van der Borght</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Thys</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Wetzels</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Clement</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Verbist</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Reumers</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>QQ-SNV: Single nucleotide variant detection at low frequency by comparing the quality quantiles</article-title>. <source>Bmc Bioinforma.</source> <volume>16</volume>, <fpage>379</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-015-0812-9</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Sparse tensor-based multiscale representation for point cloud geometry compression</article-title>. <source>IEEE Trans. pattern analysis Mach. Intell.</source> <volume>2022</volume>, <fpage>1</fpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2022.3225816</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Human mitochondrial genome compression using machine learning techniques</article-title>. <source>Hum. Genomics</source> <volume>13</volume> (<issue>1</issue>), <fpage>49</fpage>. <pub-id pub-id-type="doi">10.1186/s40246-019-0225-3</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Integrating multi-type aberrations from DNA and RNA through dynamic mapping gene space for subtype-specific breast cancer driver discovery</article-title>. <source>Peerj</source> <volume>11</volume>, <fpage>e14843</fpage>. <pub-id pub-id-type="doi">10.7717/peerj.14843</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Discovering recurrent copy number aberrations in complex patterns via non-negative sparse singular value decomposition</article-title>. <source>Ieee-Acm Trans. Comput. Biol. Bioinforma.</source> <volume>13</volume> (<issue>4</issue>), <fpage>656</fpage>&#x2013;<lpage>668</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2015.2474404</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>HetRCNA: A novel method to identify recurrent copy number alternations from heterogeneous tumor samples based on matrix decomposition framework</article-title>. <source>Ieee-Acm Trans. Comput. Biol. Bioinforma.</source> <volume>17</volume> (<issue>2</issue>), <fpage>422</fpage>&#x2013;<lpage>434</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2018.2846599</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>An omics-to-omics joint knowledge association subtensor model for radiogenomics cross-modal modules from genomics and ultrasonic images of breast cancers</article-title>. <source>Comput. Biol. Med.</source> <volume>155</volume>, <fpage>106672</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2023.106672</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Inferring subgroup-specific driver genes from heterogeneous cancer samples via subspace learning with subgroup indication</article-title>. <source>Bioinformatics</source> <volume>36</volume> (<issue>6</issue>), <fpage>1855</fpage>&#x2013;<lpage>1863</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz793</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xing</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>An efficient sparse stiffness matrix vector multiplication using compressed sparse row storage format on AMD GPU</article-title>. <source>Concurrency Computation-Practice Exp.</source> <volume>34</volume> (<issue>23</issue>). <pub-id pub-id-type="doi">10.1002/cpe.7186</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ji</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>SparkGC: Spark based genome compression for large collections of genomes</article-title>. <source>BMC Bioinforma.</source> <volume>23</volume> (<issue>1</issue>), <fpage>297</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-022-04825-5</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yao</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Ecogems: Efficient compression and retrieve of SNP data of 2058 rice accessions with integer sparse matrices</article-title>. <source>Bioinformatics</source> <volume>35</volume> (<issue>20</issue>), <fpage>4181</fpage>&#x2013;<lpage>4183</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz186</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>DETexT: An SNV detection enhancement for low read depth by integrating mutational signatures into TextCNN</article-title>. <source>Front. Genet.</source> <volume>13</volume>, <fpage>943972</fpage>&#x2013;<lpage>948021</lpage>. <comment>(Print))</comment>. <pub-id pub-id-type="doi">10.3389/fgene.2022.943972</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>