<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="review-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Med.</journal-id>
<journal-title>Frontiers in Medicine</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Med.</abbrev-journal-title>
<issn pub-type="epub">2296-858X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmed.2025.1503229</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Medicine</subject>
<subj-group>
<subject>Review</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>DNA sequence analysis landscape: a comprehensive review of DNA sequence analysis task types, databases, datasets, word embedding methods, and language models</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Asim</surname> <given-names>Muhammad Nabeel</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/910239/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Ibrahim</surname> <given-names>Muhammad Ali</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2074296/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Zaib</surname> <given-names>Arooj</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Dengel</surname> <given-names>Andreas</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>German Research Center for Artificial Intelligence GmbH</institution>, <addr-line>Kaiserslautern</addr-line>, <country>Germany</country></aff>
<aff id="aff2"><sup>2</sup><institution>Intelligentx GmbH (intelligentx.com)</institution>, <addr-line>Kaiserslautern</addr-line>, <country>Germany</country></aff>
<aff id="aff3"><sup>3</sup><institution>Department of Computer Science, Technical University of Kaiserslautern</institution>, <addr-line>Kaiserslautern</addr-line>, <country>Germany</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Alice Chen, Consultant, Potomac, MD, United States</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Matt Field, James Cook University, Australia</p>
<p>Ankan Bhattacharya, Hooghly Engineering and Technology College, India</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Muhammad Nabeel Asim <email>Muhammad_Nabeel.Asim&#x00040;dfki.de</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>08</day>
<month>04</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>12</volume>
<elocation-id>1503229</elocation-id>
<history>
<date date-type="received">
<day>28</day>
<month>09</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>10</day>
<month>03</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Asim, Ibrahim, Zaib and Dengel.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Asim, Ibrahim, Zaib and Dengel</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Deoxyribonucleic acid (DNA) serves as fundamental genetic blueprint that governs development, functioning, growth, and reproduction of all living organisms. DNA can be altered through germline and somatic mutations. Germline mutations underlie hereditary conditions, while somatic mutations can be induced by various factors including environmental influences, chemicals, lifestyle choices, and errors in DNA replication and repair mechanisms which can lead to cancer. DNA sequence analysis plays a pivotal role in uncovering the intricate information embedded within an organism&#x00027;s genetic blueprint and understanding the factors that can modify it. This analysis helps in early detection of genetic diseases and the design of targeted therapies. Traditional wet-lab experimental DNA sequence analysis through traditional wet-lab experimental methods is costly, time-consuming, and prone to errors. To accelerate large-scale DNA sequence analysis, researchers are developing AI applications that complement wet-lab experimental methods. These AI approaches can help generate hypotheses, prioritize experiments, and interpret results by identifying patterns in large genomic datasets. Effective integration of AI methods with experimental validation requires scientists to understand both fields. Considering the need of a comprehensive literature that bridges the gap between both fields, contributions of this paper are manifold: It presents diverse range of DNA sequence analysis tasks and AI methodologies. It equips AI researchers with essential biological knowledge of 44 distinct DNA sequence analysis tasks and aligns these tasks with 3 distinct AI-paradigms, namely, classification, regression, and clustering. It streamlines the integration of AI into DNA sequence analysis tasks by consolidating information of 36 diverse biological databases that can be used to develop benchmark datasets for 44 different DNA sequence analysis tasks. To ensure performance comparisons between new and existing AI predictors, it provides insights into 140 benchmark datasets related to 44 distinct DNA sequence analysis tasks. It presents word embeddings and language models applications across 44 distinct DNA sequence analysis tasks. It streamlines the development of new predictors by providing a comprehensive survey of 39 word embeddings and 67 language models based predictive pipeline performance values as well as top performing traditional sequence encoding-based predictors and their performances across 44 DNA sequence analysis tasks.</p></abstract>
<kwd-group>
<kwd>computational biology</kwd>
<kwd>computational genomics</kwd>
<kwd>DNA sequence analysis</kwd>
<kwd>artificial intelligence</kwd>
<kwd>deep learning</kwd>
</kwd-group>
<counts>
<fig-count count="8"/>
<table-count count="12"/>
<equation-count count="14"/>
<ref-count count="352"/>
<page-count count="64"/>
<word-count count="43172"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Precision Medicine</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Deoxyribonucleic acid (DNA) functions as the blueprint of life as it contains essential instructions for the development, operation, growth, and reproduction of all living organisms (<xref ref-type="bibr" rid="B1">1</xref>). Organisms utilize cell division process to grow from fertilized egg to a multicellular adult. Throughout an organism&#x00027;s lifespan, the health of tissues and organs is maintained through a continuous cycle of cell replacement. In this cycle, worn-out or damaged cells are systematically replaced with new, healthy cells. When a cell divides, each new cell requires an exact copy of the DNA to function correctly (<xref ref-type="bibr" rid="B1">1</xref>). DNA replication and repair processes ensure that each daughter cell receives the same genetic information as the parent cell, which is essential for the survival and proper functioning of all living organisms (<xref ref-type="bibr" rid="B2">2</xref>). DNA sequence changes occur through two fundamental mechanisms: germline mutations inherited from parents and somatic mutations acquired during an individual&#x00027;s lifetime (<xref ref-type="bibr" rid="B3">3</xref>). Germline mutations are present in all cells and can be passed to offspring, underlying hereditary conditions. Somatic mutations occur post-conception and can be caused by various factors including internal factors such as cellular metabolites, replication errors, and spontaneous chemical changes and external factors such as ionizing radiation, chemical mutagens, environmental pollutants, and lifestyle factors (<xref ref-type="bibr" rid="B3">3</xref>, <xref ref-type="bibr" rid="B4">4</xref>). Understanding these distinct mutation types is crucial as they require different analytical approaches. Germline mutation analysis typically involves comparing an individual&#x00027;s sequence to population databases, while somatic mutation analysis often requires comparing affected tissue to unaffected tissue from the same individual. Regardless of type, mutations in genetic information can lead to complex diseases and disorders such as cancer (<xref ref-type="bibr" rid="B1">1</xref>). To detect susceptibility, initiation, and progression of such diseases at early stages, scientists perform large-scale DNA sequence analysis (<xref ref-type="bibr" rid="B5">5</xref>). Through DNA sequence analysis, scientists can decode the intricate genetic data by uncovering the origins of genetic mutations and disorders (<xref ref-type="bibr" rid="B6">6</xref>). In addition, this analysis is crucial for the development of targeted therapies and the advancement of personalized medicine (<xref ref-type="bibr" rid="B1">1</xref>).</p>
<p>DNA sequence analysis through traditional wet-lab experiments is expensive and time-consuming (<xref ref-type="bibr" rid="B7">7</xref>, <xref ref-type="bibr" rid="B8">8</xref>). This is because wet-lab experiments require specialized equipment, e.g., PCR machines, and costly reagents (e.g., enzymes and chemicals). Detailed experiments on multiple patient samples may take weeks or even months. Moreover, experimentation requires careful execution and validation to prevent incorrect interpretations of genetic mutations due to errors or inconsistencies. The influx of next-generation sequencing and high-throughput approaches has given rise to huge sequences data. This abundance of genomic information has created both opportunities and challenges for comprehensive analysis. To expedite genomics sequence analysis, researchers are analyzing publicly available sequences data by harnessing the capabilities of Artificial Intelligence (AI) methods. It is important to mention that AI approaches serve to augment rather than replace experimental methods in DNA sequence analysis. For example, in precision medicine, AI models trained on large genomic databases can help to interpret patient-specific data by identifying relevant patterns and potential functional impacts. However, patient-specific experimental data remain essential, particularly for understanding unique aspects of individual cases such as tumor mutations. Thus, AI methods provide a valuable tool for generating hypotheses and guiding experimental design while working in concert with traditional molecular biology approaches.</p>
<p>While DNA sequence analysis encompasses a broad range of computational approaches in bioinformatics, from genome assembly and variant detection to evolutionary analysis and microbiome studies, this review focuses specifically on DNA sequence analysis tasks that involve pattern recognition and prediction, where artificial intelligence approaches can be effectively applied. These tasks include predicting functional elements, identifying regulatory regions, and classifying sequence types applications where AI can learn complex sequence patterns that may not be apparent through traditional computational methods.</p>
<p>Most of the AI-based genomics sequence analysis methods fall under the hood of regression and classification paradigms (<xref ref-type="bibr" rid="B9">9</xref>&#x02013;<xref ref-type="bibr" rid="B11">11</xref>). <xref ref-type="fig" rid="F1">Figure 1</xref> illustrates a unified workflow of AI-based predictive pipelines for genomics sequence analysis tasks. It is evident in the Figure that, overall, AI predictive pipelines can be divided into <bold>4</bold> different stages (<xref ref-type="bibr" rid="B12">12</xref>). First stage emphasizes on the collection and development of quality benchmark datasets using public databases (<xref ref-type="bibr" rid="B13">13</xref>). Second stage focuses on the characterization of raw DNA sequences in terms of statistical vectors using different kinds of sequence encoders (<xref ref-type="bibr" rid="B14">14</xref>&#x02013;<xref ref-type="bibr" rid="B16">16</xref>). This is primarily done to address the inherent dependency of AI predictive pipelines on statistical vectors (<xref ref-type="bibr" rid="B17">17</xref>&#x02013;<xref ref-type="bibr" rid="B19">19</xref>). In entire predictive pipeline, this stage is the most crucial one because highly informative and discriminative statistical vectors help the predictors to learn comprehensive useful patterns for accurate prediction (<xref ref-type="bibr" rid="B14">14</xref>&#x02013;<xref ref-type="bibr" rid="B16">16</xref>). It is widely accepted that with quality statistical vectors, even simple machine learning predictors can produce promising performance. On contrary, with less informative and discriminative statistical representations, even sophisticated deep learning predictors fail to produce decent performance (<xref ref-type="bibr" rid="B17">17</xref>&#x02013;<xref ref-type="bibr" rid="B19">19</xref>).</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Predictive pipeline of DNA sequence analysis tasks.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1503229-g0001.tif"/>
</fig>
<p>There is a marathon of developing powerful sequence encoders for generating highly informative and discriminative statistical vectors of raw sequences. To date, hundreds of sequence encoding methods have been developed (<xref ref-type="bibr" rid="B12">12</xref>) that can be broadly classified into four categories: Physico-chemical properties based methods, statistical methods (<xref ref-type="bibr" rid="B12">12</xref>, <xref ref-type="bibr" rid="B20">20</xref>), neural word embedding methods (<xref ref-type="bibr" rid="B21">21</xref>), and language models (<xref ref-type="bibr" rid="B22">22</xref>). While physico-chemical properties based methods generate statistical vectors of raw sequences using pre-computed physical and chemical values of nucleotides, statistical methods rely on occurrence frequencies of individual or group of nucleotides with DNA sequences (<xref ref-type="bibr" rid="B12">12</xref>). Physico-chemical properties based and statistical methods capture the intrinsic characteristics of biological sequences, such as nucleotide composition and distributional information. However, these methods lack to capture complex relationships of nucleotides such as long range interactions of nucleotides in the sequences (<xref ref-type="bibr" rid="B12">12</xref>, <xref ref-type="bibr" rid="B23">23</xref>). In addition, these methods may not fully capture the semantic and functional similarities between sequences (<xref ref-type="bibr" rid="B12">12</xref>, <xref ref-type="bibr" rid="B23">23</xref>). Neural word embedding methods learn distributed representations of nucleotides in the continuous vector space. These methods capture the syntactic and semantic similarities of nucleotides by mapping them to vectors in a high-dimensional space. This enables the representation of residues with similar contexts to be closer together in the vector space. Neural word embeddings methods efficiently capture semantic and contextual information of nucleotides. However, these methods lack to efficiently handle different contexts of same nucleotides (<xref ref-type="bibr" rid="B21">21</xref>). Language models also learn representation of individual nucleotides or groups of nucleotides (k-mers) in an unsupervised fashion by predicting masked nucleotides based on the context of surrounding nucleotides. Language models based methods capture complex nucleotide relations; however, these methods require large amount of sequence data for training and hyperparameter optimization (<xref ref-type="bibr" rid="B22">22</xref>).</p>
<p>Third stage includes predictors that make best use of statistical vectors produced by second stage to extract informative patterns for creating decision boundaries. Overall, these predictors can be classified into two categories: machine learning and deep learning (<xref ref-type="bibr" rid="B12">12</xref>). Machine learning predictors require less data and computational power for training. However, these predictors lack to capture comprehensive complex relationships of nucleotide (<xref ref-type="bibr" rid="B12">12</xref>), whereas deep learning predictors (<xref ref-type="bibr" rid="B24">24</xref>) are capable to learn highly complex relationships of nucleotide. However, these predictors require a huge amount of training data and computational power (<xref ref-type="bibr" rid="B12">12</xref>). In fourth stage, comprehensive evaluation of predictors using different experimental settings and evaluation measures is performed (<xref ref-type="bibr" rid="B24">24</xref>).</p>
<p>AI researchers have been endeavoring to complement wet-lab-based DNA sequence analysis methods by incorporating more innovative sequence encoders at second stage and predictors at third stage of predictive pipeline. However, there is still ample room for the development of more powerful predictive pipelines. Different fields such as Natural Language Processing (NLP), Energy, and Computer Vision have seen substantial progress in the development of diverse predictive pipelines. Whereas, the DNA sequence analysis field is known for its wide range of tasks, still the progress of AI applications in this area is hindered mainly due to the lack of integration between molecular biologist and AI experts. For instance, the field of NLP has made strides with multi-task learning predictors. However, the DNA sequence analysis field lags behind due to AI experts limited understanding of the diverse range of DNA analysis tasks that could support the development of multi-task learning predictors. Furthermore, the efficacy of AI applications hinges on the availability of benchmark datasets. Although developing datasets in DNA sequence analysis is relatively straightforward due to abundance of public databases which contain raw biological sequences along with associated labels, there is a tendency among researchers to overlook existing benchmark datasets, develop new benchmark datasets, and neglect comprehensive performance comparisons with existing predictors. This oversight often complicates the determination of the most effective predictors for specific tasks. For example, up to date, according to our best of knowledge, approximately 127 predictive models have been developed and published in 59 different conferences and journals for widely studied 44 different DNA sequence analysis tasks. To enhance the performance of predictive models developed for diverse DNA sequence analysis tasks, researchers need to conduct a comprehensive examination of existing literature to find most effective algorithms for different stages of new predictive pipelines. With an aim to expedite progress in the development of fair and robust AI applications for DNA sequence analysis, numerous review articles have emerged. However, these reviews typically focus on isolated tasks rather than providing a holistic overview. Considering the need and significance of a comprehensive study that bridges the gap between AI specialists and biologists, this paper makes manifold contributions:</p>
<list list-type="bullet">
<list-item><p>It bridges the gap between DNA sequence analysis and artificial intelligence fields by presenting a diverse range of DNA analysis tasks and AI methodologies.</p></list-item>
<list-item><p>It empowers AI researchers by equipping them with essential biological knowledge related to 44 distinct DNA sequence analysis tasks. It categorizes 44 different DNA sequence analysis tasks into 8 different categories on the basis of sequence analysis goals. This categorization provides a structured overview to biologists and AI researchers in navigating the complex landscape of genomics studies more efficiently.</p></list-item>
<list-item><p>It streamlines the integration of AI into DNA sequence analysis by consolidating information of 36 diverse biological databases being used to develop benchmark datasets for 44 different DNA sequence analysis tasks.</p></list-item>
<list-item><p>It sheds light on the nature of 44 different DNA sequence analysis tasks and categorizes them into three primary categories: regression, classification, and clustering, and three secondary categories: binary classification, multi-class classification, and multi-label classification. This categorization assists computer scientists in efficient selection of most suitable algorithms for each task category, development of more effective and specialized computational frameworks, and to significantly accelerate advancements in AI-driven genomic research.</p></list-item>
<list-item><p>It provides insights of 140 benchmark datasets related to 44 distinct DNA sequence analysis tasks to ensure performance comparisons between new and existing AI predictors.</p></list-item>
<list-item><p>It presents word embeddings and language models applications for 44 distinct DNA sequence analysis tasks.</p></list-item>
<list-item><p>It streamlines the development of new predictors by providing a comprehensive survey of current top predictors, their performances across 44 DNA sequence analysis tasks, and their public accessibility. This comprehensive overview serves as a valuable resource for researchers developing and validating predictive pipelines in computational genomics.</p></list-item>
</list>
<p>It is important to note that our categorization of 44 DNA sequence analysis tasks emerges from the AI and computational biology literature rather than representing a definitive biological taxonomy. We have organized these tasks into biologically relevant groupings based on their functional and analytical similarities, while recognizing that many tasks span multiple biological domains. This organization aims to bridge the gap between computational methodologies and biological applications, although we acknowledge that future refinements with deeper domain expert input would further enhance this framework.</p>
</sec>
<sec id="s2">
<title>2 Research methodology</title>
<p>This section provides a detailed overview of the research methodology used to identify articles focused on word embeddings and large language models applications in DNA sequence analysis landscape (<xref ref-type="bibr" rid="B10">10</xref>, <xref ref-type="bibr" rid="B11">11</xref>). <xref ref-type="fig" rid="F2">Figure 2</xref> illustrates two stage processes for article identification and selection.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Research methodology.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1503229-g0002.tif"/>
</fig>
<sec>
<title>2.1 Article searching</title>
<p>To identify a wide range of relevant scholarly articles, initial stage involves formulation of quality search queries using different keywords. In <xref ref-type="fig" rid="F2">Figure 2</xref>, article identification module contains keywords cell of three different categories, namely, DNA tasks, word embedding methods, and Language models. To formulate quality search queries, keywords within same category are combined using OR &#x02228; operator, while keywords of different categories are combined using AND &#x02227; operator. For instance, few sample search queries include DNA Replication Origins Identification using BERT language model, DNA Replication Origins Identification using DeepWalk word embedding method, etc. To acquire relevant papers, formulated search queries are executed on academic search engines such as Google Scholar,<xref ref-type="fn" rid="fn0001"><sup>1</sup></xref> ACM Digital Library,<xref ref-type="fn" rid="fn0002"><sup>2</sup></xref> Elsevier,<xref ref-type="fn" rid="fn0003"><sup>3</sup></xref> IEEEXplore,<xref ref-type="fn" rid="fn0004"><sup>4</sup></xref> Wiley Online Library,<xref ref-type="fn" rid="fn0005"><sup>5</sup></xref> Springer,<xref ref-type="fn" rid="fn0006"><sup>6</sup></xref> and ScienceDirect.<xref ref-type="fn" rid="fn0007"><sup>7</sup></xref> In addition, snowballing method is employed to explore sources referenced in extracted papers to identify more research articles. This technique is particularly useful in research contexts where access to resources is limited, such as niche topics or hard-to-reach communities, as it expands the pool of resources for a study. Execution of queries across multiple academic databases acquired approximately 238 research articles which are screened and filtered in second stage.</p>
</sec>
<sec>
<title>2.2 Article screening and filtering</title>
<p>Second stage selects most relevant articles in two steps. In the first step, titles and abstracts of 113 word embeddings and 125 large language models related studies were reviewed. This review analysis identified 80 word embeddings and 104 language models related relevant articles. Second step involves full-text assessment of articles selected in first step, resulting in 39 word embeddings and 67 language models related articles.</p>
<p>Our selection criteria focused on DNA sequence analysis tasks where (1) raw DNA sequence data serve as the primary input, (2) AI methods extract patterns from these sequences, and (3) the analysis predicts specific biological properties or functions. This allowed us to examine AI&#x00027;s impact on genomic sequence interpretation while acknowledging that bioinformatics encompasses many other types of analyses not covered here.</p>
</sec>
</sec>
<sec id="s3">
<title>3 Biological foundations of DNA sequence analysis goals and tasks</title>
<p>With an aim to find molecular basis of diseases initiation and progression, their effective detection at early stages, and development of potent drugs, researchers are trying to understand DNA sequence language by performing a variety of sequence analysis tasks. Every unique DNA sequence analysis task aims to enhance the understanding of one specific aspect of DNA, and a bunch of tasks can enhance the understanding of specific major biological goal. To summarize the biological background of 44 distinct DNA sequence analysis tasks, we have categorized them into 8 major biological goals. <xref ref-type="fig" rid="F3">Figure 3</xref> depicts the biological categorization of 44 unique DNA sequences analysis tasks into 8 different goals, namely, genome structure and stability, gene expression regulation, gene analysis, gene network analysis, DNA modification prediction, DNA functional analysis, environmental and microbial genomics, and disease analysis. This biologically informed organization was developed by analyzing both the computational biology literature and aligning with biological processes in genomics research. While computational researchers often approach these tasks through the lens of AI methodologies, we have endeavored to categorize them according to their biological relevance and function. Our categorization into 8 major biological goals represents an attempt to bridge computational approaches with biological understanding. Although we recognize the inherent complexity and interconnectedness of biological systems which indicates that many tasks could reasonably be classified in multiple categories, thus, this categorization represents one of several possible ways to organize these tasks. This categorization reflects the diverse biological applications where AI-based sequence analysis has made significant contributions. However, we recognize that DNA sequence analysis in bioinformatics extends beyond these pattern recognition tasks to include other critical applications such as genome assembly, variant detection, and population genetics studies. We specifically examine how modern AI approaches are transforming our ability to extract meaningful biological insights from sequence data through pattern-based prediction tasks.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Precise classification of 44 unique DNA sequence analysis tasks in 8 major biological goals.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1503229-g0003.tif"/>
</fig>
<p>In living organisms, DNA is packaged at multiple levels to condense vast genetic information into a well-organized structure within the cell nucleus (<xref ref-type="bibr" rid="B1">1</xref>). At the first level, DNA is wrapped around histone octamers also known as nucleosomes. These nucleosomes further assemble into chromatin, which then folds and condenses into an even more compact structure known as the genome (<xref ref-type="bibr" rid="B1">1</xref>). The exploration of genome structure and stability is pivotal in understanding the biological intricacies and potential therapeutic avenues. Genome structure can affect how genes are accessed and used. Disruptions in this structure, such as missing or misplaced DNA sections, or changes in how tightly DNA is wrapped around histone octamers, or irregularities in nucleosomes positions can lead to genes being turned on or off at the wrong times or in the wrong amounts (<xref ref-type="bibr" rid="B1">1</xref>). This can cause various diseases and biological disorders. DNA is an instruction manual that controls biological functioning within living organisms. If genome gets unstable, the manual gets messed up such as typos and missing sections. It can lead to uncontrolled growth of the cells (cancer) and improper working of the genes (many diseases) (<xref ref-type="bibr" rid="B1">1</xref>). In a nutshell, a stable genome possesses clear, complete instruction manual, essential for keeping biological functions working smooth. To better understand genome structure and stability, it is essential to explore various tasks such as DNA Replication Origins Prediction (<xref ref-type="bibr" rid="B25">25</xref>, <xref ref-type="bibr" rid="B26">26</xref>), Genome Structure Analysis (<xref ref-type="bibr" rid="B27">27</xref>, <xref ref-type="bibr" rid="B28">28</xref>), Nucleosome Position Detection (<xref ref-type="bibr" rid="B29">29</xref>, <xref ref-type="bibr" rid="B30">30</xref>), Chromatin Accessibility Prediction (<xref ref-type="bibr" rid="B31">31</xref>&#x02013;<xref ref-type="bibr" rid="B33">33</xref>), Chromatin Feature Prediction (<xref ref-type="bibr" rid="B31">31</xref>, <xref ref-type="bibr" rid="B34">34</xref>, <xref ref-type="bibr" rid="B35">35</xref>), Long-range Chromatin Interaction Prediction (<xref ref-type="bibr" rid="B36">36</xref>, <xref ref-type="bibr" rid="B37">37</xref>), and YY1-Mediated Chromatin Loops Prediction (<xref ref-type="bibr" rid="B38">38</xref>, <xref ref-type="bibr" rid="B39">39</xref>). These tasks are crucial for comprehending the intricate mechanisms governing genetic information processing and regulation within cells (<xref ref-type="bibr" rid="B40">40</xref>).</p>
<p>DNA replication origin prediction is fundamental as accurate replication of the genome is vital for maintaining genomic stability (<xref ref-type="bibr" rid="B25">25</xref>). The prediction of replication origins involves calculating DNA structural properties to identify sites crucial for initiating DNA replication (<xref ref-type="bibr" rid="B25">25</xref>). Understanding where these sites are located and how they are specified is essential for comprehending DNA replication and ensuring genome integrity (<xref ref-type="bibr" rid="B41">41</xref>). Genome structure analysis plays a pivotal role in deciphering the organization and arrangement of genetic material within the cell (<xref ref-type="bibr" rid="B27">27</xref>). By analyzing the structural features of the genome, researchers can gain insights into the functional and spatial organization of chromosomes, aiding in the identification of genomic elements involved in gene regulation and phenotypic variations (<xref ref-type="bibr" rid="B27">27</xref>, <xref ref-type="bibr" rid="B42">42</xref>). Furthermore, nucleosome position detection is essential for understanding how nucleosomes, the basic units of genome, are arranged along the DNA strand (<xref ref-type="bibr" rid="B29">29</xref>, <xref ref-type="bibr" rid="B43">43</xref>). This information is crucial for elucidating gene regulation mechanisms and chromatin dynamics within the cell (<xref ref-type="bibr" rid="B29">29</xref>, <xref ref-type="bibr" rid="B43">43</xref>). Chromatin accessibility prediction is a key task that involves determining the regions of chromatin that are accessible for transcription factors and other regulatory proteins to bind (<xref ref-type="bibr" rid="B31">31</xref>&#x02013;<xref ref-type="bibr" rid="B33">33</xref>). Prediction of chromatin accessibility across different cellular contexts provides valuable insights into gene regulation and chromatin dynamics (<xref ref-type="bibr" rid="B31">31</xref>&#x02013;<xref ref-type="bibr" rid="B33">33</xref>). Chromatin feature prediction complements accessibility prediction by identifying specific chromatin features and epigenetic markers that influence gene expression and regulatory processes (<xref ref-type="bibr" rid="B31">31</xref>, <xref ref-type="bibr" rid="B34">34</xref>, <xref ref-type="bibr" rid="B35">35</xref>, <xref ref-type="bibr" rid="B44">44</xref>). These features include transcription factor (TF) binding sites, DNase I-hypersensitive sites (DHS), and histone marks (HM). By understanding these features, researchers can unravel the mechanisms underlying chromatin regulation and gene expression (<xref ref-type="bibr" rid="B34">34</xref>). Long-range chromatin interactions make bridges between distant enhancers and promoters. These interactions enable interactions between enhancers and promoters by bringing them closer to each other (<xref ref-type="bibr" rid="B36">36</xref>, <xref ref-type="bibr" rid="B37">37</xref>). YY1-mediated chromatin loop prediction provides comprehensive understanding about gene regulation (<xref ref-type="bibr" rid="B38">38</xref>, <xref ref-type="bibr" rid="B39">39</xref>, <xref ref-type="bibr" rid="B45">45</xref>). YY1 is a protein that makes loop between enhancers and promoters. These loops are essential for gene regulation, and by predicting these loops, we can see which genes can be controlled through YY1 protein (<xref ref-type="bibr" rid="B38">38</xref>, <xref ref-type="bibr" rid="B39">39</xref>, <xref ref-type="bibr" rid="B45">45</xref>). This knowledge is valuable for understanding diseases where gene regulation goes wrong. To sum up, only through multi-dimensional exploration of genome structure and stability, researchers can discriminate healthy cellular processes from malfunctioned processes, find the root causes of diseases, and develop potent therapies.</p>
<p>Another major goal of molecular biologists behind is gene expression regulation. Gene expression regulation provides fundamental insights into how genes are activated or repressed in response to various cellular cues (<xref ref-type="bibr" rid="B46">46</xref>). Specifically, researchers are trying to unravel the intricate mechanisms that control when and up to what extent specific genes are turned on or off in different cells and tissues (<xref ref-type="bibr" rid="B46">46</xref>). This knowledge forms the basis for understanding the functional behavior of genes in different biological contexts and sets the stage for further analyses. Hence, it holds immense promise for scientists and pharmaceutical industries. This helps scientists to detect irregularities in normal gene expression regulation, the way diseases develop at the molecular level, and identify potential drug targets (<xref ref-type="bibr" rid="B46">46</xref>). Furthermore, this understanding can assist pharmaceutical industries to develop improved diagnostic tools, innovative personalized therapies, and targeted interventions, which will ultimately contribute to advancements in personalized healthcare (<xref ref-type="bibr" rid="B46">46</xref>). In addition, it can provide a deeper understanding of biological systems which can lead to breakthroughs in biotechnology (<xref ref-type="bibr" rid="B46">46</xref>). For better understanding of gene expression regulation, researchers are performing nine different DNA sequence analysis tasks, namely, enhancer identification (<xref ref-type="bibr" rid="B47">47</xref>), promoter identification (<xref ref-type="bibr" rid="B48">48</xref>), enhancer-promoter interactions prediction (<xref ref-type="bibr" rid="B49">49</xref>), transcription site prediction (<xref ref-type="bibr" rid="B50">50</xref>), transcription factor binding site prediction (<xref ref-type="bibr" rid="B51">51</xref>), transcription factor binding affinity prediction (<xref ref-type="bibr" rid="B52">52</xref>), protein-DNA binding site prediction (<xref ref-type="bibr" rid="B53">53</xref>), splice site prediction (<xref ref-type="bibr" rid="B53">53</xref>), and translation initiation site prediction (<xref ref-type="bibr" rid="B54">54</xref>). Enhancers (<xref ref-type="bibr" rid="B47">47</xref>, <xref ref-type="bibr" rid="B55">55</xref>&#x02013;<xref ref-type="bibr" rid="B75">75</xref>) and promoters identification (<xref ref-type="bibr" rid="B48">48</xref>, <xref ref-type="bibr" rid="B76">76</xref>&#x02013;<xref ref-type="bibr" rid="B81">81</xref>), along with their interactions (<xref ref-type="bibr" rid="B82">82</xref>&#x02013;<xref ref-type="bibr" rid="B86">86</xref>) prediction are important to decipher a complex control panel for gene expression (<xref ref-type="bibr" rid="B47">47</xref>&#x02013;<xref ref-type="bibr" rid="B49">49</xref>). Enhancers are known as distant switches of genes, while promoters are the landing sites where gene activation starts. Identification of these elements and predicting how they loop together provide a comprehensive understanding of gene regulation, including which genes are activated or repressed, the intensity of their expression, and the specific cell types involved (<xref ref-type="bibr" rid="B87">87</xref>, <xref ref-type="bibr" rid="B88">88</xref>). This knowledge reveals the intricate regulatory code that governs gene expression and offers valuable insights into the mechanisms underlying normal cellular function as well as the dysregulation that may contribute to various diseases.</p>
<p>Furthermore, prediction of different genomic sites including transcription sites (<xref ref-type="bibr" rid="B50">50</xref>), transcription factor binding sites (<xref ref-type="bibr" rid="B89">89</xref>&#x02013;<xref ref-type="bibr" rid="B93">93</xref>), transcription factor binding site affinity (<xref ref-type="bibr" rid="B52">52</xref>), protein-DNA binding site (<xref ref-type="bibr" rid="B53">53</xref>, <xref ref-type="bibr" rid="B94">94</xref>&#x02013;<xref ref-type="bibr" rid="B96">96</xref>), splice site (<xref ref-type="bibr" rid="B93">93</xref>, <xref ref-type="bibr" rid="B97">97</xref>&#x02013;<xref ref-type="bibr" rid="B100">100</xref>), and translation initiation site (<xref ref-type="bibr" rid="B50">50</xref>, <xref ref-type="bibr" rid="B101">101</xref>) provide deep insights into gene expression regulation. A transcription site refers to the specific location on the DNA where the process of transcription takes place. Transcription is the synthesis of RNA from a DNA template, and the transcription site represents the region where the RNA polymerase enzyme binds and initiates the transcription process, whereas transcription factor binding sites are specific DNA sequences where transcription factors (proteins), that regulate gene expression, bind. These binding sites are typically located near the transcription start site and are recognized by transcription factors to control the initiation or repression of transcription. In contrast, transcription factor binding site affinity refers to the strength or affinity with which a transcription factor binds to its specific binding site on DNA. It represents the likelihood of a transcription factor binding to its target site and influencing gene expression. A protein-DNA binding site refers to any region on the DNA where a protein binds. This can include transcription factors, as mentioned earlier, as well as other proteins involved in various cellular processes such as DNA replication, repair, and chromatin remodeling. Splice sites are specific sequences within a gene&#x00027;s DNA that mark the boundaries of introns and exons. During the process of RNA splicing, introns are removed from the pre-mRNA molecule, and exons are joined together to form the mature mRNA. Splice sites are essential for the accurate and precise splicing of RNA. Translation initiation site (TIS) is the specific location on the mRNA molecule where the process of translation begins. TIS prediction seems like a RNA sequence analysis task; however, in molecular biology research, to study gene expression, researchers are synthesizing complementary DNA (cDNA) data from messenger RNA (mRNA) template through a process called reverse transcription. In the context of cDNA data, the translation initiation site (TIS) represents the position where the ribosome, the cellular machinery responsible for protein synthesis, binds to the mRNA to initiate translation. The TIS is typically identified by the presence of specific start codons, such as AUG, which serve as signals for the ribosome to start protein synthesis.</p>
<p>To better understand gene functions and their roles in disease initiation, researchers are exploring various aspects such as gene expression prediction (<xref ref-type="bibr" rid="B102">102</xref>, <xref ref-type="bibr" rid="B103">103</xref>), identification of essential (<xref ref-type="bibr" rid="B104">104</xref>&#x02013;<xref ref-type="bibr" rid="B109">109</xref>) and disease-specific genes (<xref ref-type="bibr" rid="B110">110</xref>), gene function prediction (<xref ref-type="bibr" rid="B111">111</xref>, <xref ref-type="bibr" rid="B112">112</xref>), pseudo-gene function prediction (<xref ref-type="bibr" rid="B111">111</xref>), target gene classification (<xref ref-type="bibr" rid="B113">113</xref>), and candidate gene prioritization (<xref ref-type="bibr" rid="B114">114</xref>). Overall together, these tasks provide a comprehensive platform for disease diagnosis and development of treatment strategies by uncovering disease mechanisms, identifying potential therapeutic targets, and organizing genes into functional categories. Specifically, gene expression prediction provides useful information about the level of gene activity in different cells or tissues (<xref ref-type="bibr" rid="B115">115</xref>). This task is vital for understanding the molecular mechanisms underlying complex diseases such as cancer and identifying potential therapeutic targets. Essential gene identification is another critical task in gene analysis that helps researchers pinpoint genes that are crucial for an organism&#x00027;s survival and development (<xref ref-type="bibr" rid="B116">116</xref>, <xref ref-type="bibr" rid="B117">117</xref>). This task is particularly important in understanding gene function and the genetic basis of various disorders. Gene function prediction elucidates the roles of genes in different pathways and biological processes and provides valuable insights into disease mechanisms and potential therapeutic interventions.</p>
<p>Apart from gene function prediction, pseudo-gene function prediction has gained a lot of attention as a critical task in gene analysis (<xref ref-type="bibr" rid="B111">111</xref>). Pseudogenes were once thought to be useless DNA because they cannot code for proteins due to mutations that happened over time. However, recent studies have shown that pseudogenes actually play important roles in controlling genes, especially in cancer. For instance, the pseudogene PTENP1 helps to regulate the tumor suppressor gene PTEN in various cancer conditions, showing that pseudogenes can have important functions. Pseudogene function prediction offers numerous advantages, including better understanding of gene regulation, disease mechanisms, evolutionary biology, and the potential for new biomarkers and drug targets. In addition, disease gene prediction is a pivotal task in gene analysis that focuses on identifying genes associated with specific diseases or disorders (<xref ref-type="bibr" rid="B118">118</xref>). By pinpointing disease-related genes, researchers can unravel the genetic basis of diseases, discover novel biomarkers for diagnosis and prognosis, and develop targeted therapies. This task is instrumental in precision medicine approaches, where understanding the genetic underpinnings of diseases is crucial for personalized treatment strategies. Target gene classification involves categorizing genes based on their functions, interactions, or regulatory mechanisms (<xref ref-type="bibr" rid="B119">119</xref>). By classifying target genes, researchers can better understand gene networks, signaling pathways, and biological processes. This task is essential for deciphering the complex relationships between genes and their roles in health and disease. Candidate gene prioritization and selection are critical tasks in gene analysis that aim to identify genes with the highest likelihood of being involved in a particular biological process or disease (<xref ref-type="bibr" rid="B120">120</xref>). By prioritizing candidate genes, researchers can focus their efforts on studying genes that are most likely to have significant effects, accelerating the discovery of novel gene functions and disease mechanisms. This task is crucial for efficiently allocating research resources and maximizing the impact of genetic studies. Aforementioned seven DNA sequence analysis tasks are essential for advancing our understanding of genes and their roles in health and disease. By leveraging these tasks, researchers can unravel the complexities of the genome, uncover novel gene functions, and pave the way for innovative diagnostic and therapeutic strategies in various fields of biology and medicine.</p>
<p>Furthermore, gene network analysis is a promising goal that seeks to comprehend the intricate interactions and relationships between genes within a biological system. Two primary tasks within Gene Network Analysis are Gene Taxonomy Classification and Gene Network Reconstruction. Gene Taxonomy Classification (<xref ref-type="bibr" rid="B121">121</xref>&#x02013;<xref ref-type="bibr" rid="B123">123</xref>) involves categorizing genes based on their evolutionary relationships and functional similarities, providing a structured framework for organizing genetic information. Gene Taxonomy Classification plays a crucial role in gene network analysis by offering a foundational structure for understanding the evolutionary history and functional relationships between genes. By classifying genes into taxonomic groups based on shared characteristics and evolutionary relatedness, researchers can infer valuable insights into the origins and evolutionary trajectories of genes within a network (<xref ref-type="bibr" rid="B124">124</xref>). This classification allows for the identification of core genes that have remained conserved throughout evolution, providing a basis for inferring phylogenetic relationships and understanding the fundamental building blocks of gene networks. Moreover, Gene Taxonomy Classification enables researchers to utilize existing knowledge about gene functions and evolutionary relationships to guide Gene Network Reconstruction. By categorizing genes into taxonomic groups, researchers can pinpoint gene clusters with similar functions or evolutionary origins, facilitating the identification of modules within gene networks that exhibit coordinated activity (<xref ref-type="bibr" rid="B125">125</xref>). This classification serves as a roadmap for exploring the functional roles of genes within a network and understanding how these roles have evolved over time. On the other hand, Gene Network Reconstruction (<xref ref-type="bibr" rid="B126">126</xref>&#x02013;<xref ref-type="bibr" rid="B128">128</xref>) involves creating a detailed map of the interactions and regulatory relationships between genes within a cell or an organism. The primary input for gene network reconstruction is gene expression data obtained through high-throughput techniques such as RNA sequencing (RNA-seq) or microarrays. This task is pivotal for understanding how genes work together to control various biological functions and processes (<xref ref-type="bibr" rid="B129">129</xref>). By reconstructing gene networks, researchers can uncover key regulatory hubs involving highly connected genes, clusters of closely interacting genes, pathways, and interactions that steer cellular functions and responses to external stimuli (<xref ref-type="bibr" rid="B130">130</xref>).</p>
<p>DNA modification prediction is also a crucial goal where researchers aim is to decipher how tiny tweaks to the DNA code can lead to big changes in cellular functions (<xref ref-type="bibr" rid="B131">131</xref>&#x02013;<xref ref-type="bibr" rid="B133">133</xref>). In DNA modifications, distinct chemical groups are added to specific locations on the DNA molecule. These additions do not change the actual sequence of nucleotides (A, C, G, T) but can alter the physical properties of DNA sequence. Understanding these modifications, such as 4-Methylcytosine (4mc) (<xref ref-type="bibr" rid="B134">134</xref>&#x02013;<xref ref-type="bibr" rid="B143">143</xref>), Methyladenine (6ma) (<xref ref-type="bibr" rid="B144">144</xref>&#x02013;<xref ref-type="bibr" rid="B151">151</xref>), 5-methylcytosine (5mc) (<xref ref-type="bibr" rid="B152">152</xref>, <xref ref-type="bibr" rid="B153">153</xref>), 5-hydroxymethylcytosine (5hmc) (<xref ref-type="bibr" rid="B154">154</xref>&#x02013;<xref ref-type="bibr" rid="B157">157</xref>), and methylation modifications (<xref ref-type="bibr" rid="B146">146</xref>, <xref ref-type="bibr" rid="B154">154</xref>&#x02013;<xref ref-type="bibr" rid="B159">159</xref>), are essential for advancing our comprehension of epigenetic regulation (<xref ref-type="bibr" rid="B160">160</xref>&#x02013;<xref ref-type="bibr" rid="B162">162</xref>). Specifically, methylation modifications that occur due to the addition of methyl groups to DNA molecules play a pivotal role in regulating gene expression and maintaining genomic integrity. Similarly, methyladenine modifications, such as DNA N6-methyladenine (6mA), occur due to the addition of a methyl group to the adenine base of DNA. DNA 6mA modifications dynamically influence DNA thermal stability, curvature, and transcription factor interactions, impacting gene expression in a heritable manner. Understanding the prediction of 6mA sites is pivotal for both basic and clinical research as it aids in the identification of gene expression patterns and potential epigenetic changes induced by environmental factors. These predictions enhance our ability to study the role of 6mA modifications in diseases and could lead to improved therapeutic strategies, highlighting the relevance of accurate prediction methods in unraveling the complexities of DNA modifications. Moreover, 5-methylcytosine (5mc) modification occurs due to the addition of a methyl group to the cytosine base of DNA, whereas 5-hydroxymethylcytosine (5hmc) modification is an oxidized derivative of 5mc, where an additional hydroxyl group (-OH) is added to the methyl group of 5mc. Prediction of 5-methylcytosine (5mc) and 5-hydroxymethylcytosine (5hmc) modifications is essential for decoding their roles in gene regulation, developmental processes, and disease states. These critical epigenetic modifications are dynamically regulated by enzymes and influence gene expression crucial for neuronal differentiation and cellular proliferation. Abnormal levels of these modifications have been linked to diseases such as cancer. Precise prediction of 5mc and 5hmc sites is useful for the development of targeted therapies and improved prognostic assessments.</p>
<p>Functional genomics is also a critical goal that encompasses multiple sub-tasks including species classification (<xref ref-type="bibr" rid="B44">44</xref>), conserved non-coding element (NCE) classification (<xref ref-type="bibr" rid="B163">163</xref>), functional prioritization of non-coding variants (<xref ref-type="bibr" rid="B34">34</xref>), prediction of context specific functional impact of genetic variants (<xref ref-type="bibr" rid="B36">36</xref>), exon and intron region classification (<xref ref-type="bibr" rid="B164">164</xref>), and recombination spots identification (<xref ref-type="bibr" rid="B165">165</xref>). Each of these tasks plays a vital role in unraveling the complexities of genetic regulation and molecular mechanisms within the genome. In biomedical research, understanding the genetic similarities and differences between humans and other species is crucial for modeling diseases and studying genetic disorders. Majority of the genome is conserved across different species which makes it difficult to distinguish humans and non-human species. Despite very high genetic similarity across species (&#x0003C; 10% sequence divergence), small differences are extremely valuable and they have significant biological implications. Species classification determines the source species of genetic sequences based on such differences and pave way for better modeling diseases and studying genetic disorders (<xref ref-type="bibr" rid="B44">44</xref>). Conserved non-coding element classification is another critical task in functional genomics that focuses on identifying and understanding non-coding regions of the genome that are evolutionarily conserved across different species (<xref ref-type="bibr" rid="B163">163</xref>). It is essential for advancing our understanding of gene regulation, evolutionary biology, and the genetic basis of diseases. By elucidating the functions of these non-coding regions, researchers can gain insights into the intricate regulatory networks that govern gene expression and cellular processes and contribute to the development of targeted therapies.</p>
<p>Functional prioritization of non-coding variants (<xref ref-type="bibr" rid="B34">34</xref>) is another crucial task for making sense of the vast amount of genetic data generated by modern sequencing technologies. By identifying which variants have significant biological impacts, researchers can gain a deeper understanding of the genetic architecture of complex diseases, uncover novel therapeutic targets, and advance the field of precision medicine. This prioritization is essential for translating genomic research into practical health benefits and ultimately improving patient outcomes and advancing our knowledge of human biology (<xref ref-type="bibr" rid="B34">34</xref>). As functional prioritization of non-coding variants task involves identifying which non-coding variants among millions are likely to have functional consequences, it does not account for the specific context in which these variants might exert their effects, whereas prediction of context-specific functional impact of genetic variants aims to provide a detailed understanding of how specific variants influence gene function in different contexts (e.g., specific tissue) (<xref ref-type="bibr" rid="B36">36</xref>). This is particularly important for genetic studies that seek to uncover the mechanisms by which variants contribute to disease phenotypes. Unlike functional prioritization of non-coding variants task which only filters the variants that are most likely to have functional significance. Prediction of context-specific functional impact of genetic variants provides a finer level of detail by predicting the actual effect of a variant on gene expression or other functional outcomes in specific tissues. This granularity is essential for precisely understanding the specific biological mechanisms and for developing targeted therapies (<xref ref-type="bibr" rid="B36">36</xref>).</p>
<p>Exon and intron region classification is crucial for understanding gene structure and function within the genome. Exons are coding regions that are translated into proteins, while introns are non-coding regions that are spliced out during mRNA processing. By classifying exons and introns, researchers can describe gene boundaries, identify functional elements, and elucidate the mechanisms of gene expression regulation (<xref ref-type="bibr" rid="B166">166</xref>). This task is essential for deciphering the genetic code and unraveling the complexities of gene regulation in health and disease. Recombination spots identification is a pivotal task in functional genomics that focuses on mapping regions of the genome where genetic recombination events occur. Genetic recombination is a natural process where DNA segments are exchanged between two chromosomes during cell division. Recombination plays a vital role in generating genetic diversity, ensuring proper chromosome segregation, and driving evolution (<xref ref-type="bibr" rid="B167">167</xref>). By identifying recombination hot spots, researchers can gain insights into the mechanisms underlying genetic diversity and genome evolution, shedding light on the processes that shape genetic variation and adaptation in populations. In conclusion, the tasks related to functional genomics, including species classification, conserved non-coding element classification, functional prioritization of non-coding variant, prediction of context-specific functional impact of genetic variants, exon and intron region classification, and recombination spots identification, are essential for advancing our understanding of genetic regulation, molecular mechanisms, and disease pathogenesis. By delving into these tasks, researchers can unravel the complexities of the genome, decipher the genetic basis of diseases, and pave the way for precision medicine and personalized healthcare interventions tailored to an individual&#x00027;s genetic profile.</p>
<p>Another goal of researchers is to study overlap between two distinct fields namely environmental science and microbial genomics (<xref ref-type="bibr" rid="B27">27</xref>). This interdisciplinary study enables researchers to explore how environmental factors such as pollution, climate change, and agricultural practices affect on function and diversity of microbial communities (<xref ref-type="bibr" rid="B27">27</xref>). A key area of focus in this field is the nitrogen cycle prediction. By examining the genomes of microbes involved in nitrogen fixation, nitrification, and denitrification, scientists can predict how these processes might respond to environmental changes (<xref ref-type="bibr" rid="B168">168</xref>). This prediction provides understanding about potential impacts of environmental shifts on ecosystem health (<xref ref-type="bibr" rid="B169">169</xref>) and nitrogen availability, which are essential for plant growth and overall biogeochemical cycles (<xref ref-type="bibr" rid="B170">170</xref>).</p>
<p>From all eight different biological goals, disease analysis goal has received huge attention in scientific community as it aims to understand, diagnose, and treat various illnesses. Within this field, several tasks play a vital role in enhancing our comprehension of diseases. One such task is Pathogen Signatures Identification (<xref ref-type="bibr" rid="B171">171</xref>), which involves identifying specific markers or characteristics of pathogens that can aid in their detection and classification (<xref ref-type="bibr" rid="B172">172</xref>). By pinpointing these signatures, researchers can develop targeted diagnostic tools and therapies, ultimately improving disease management and control. Mutation Susceptibility Analysis (<xref ref-type="bibr" rid="B173">173</xref>) is another essential task in disease analysis. This task focuses on investigating the genetic variations that make individuals more prone to developing certain diseases (<xref ref-type="bibr" rid="B174">174</xref>). Understanding mutation susceptibility can aid in personalized medicine approaches, where individuals at higher risk can be identified early for preventive interventions or closer monitoring. Phage-Host Interactions Prediction (<xref ref-type="bibr" rid="B175">175</xref>&#x02013;<xref ref-type="bibr" rid="B177">177</xref>) is a task that delves into the relationships between bacteriophages and their host bacteria (<xref ref-type="bibr" rid="B178">178</xref>). By predicting these interactions, researchers can gain insights into how phages influence bacterial populations, which is crucial for developing phage-based therapies to combat bacterial infections and antibiotic resistance. Disease Risks Estimation (<xref ref-type="bibr" rid="B90">90</xref>) is a fundamental aspect of disease analysis that involves assessing the likelihood of an individual developing a particular condition based on various factors such as genetics, lifestyle, and environmental exposures (<xref ref-type="bibr" rid="B179">179</xref>). Accurately estimating disease risks enables healthcare providers to offer targeted interventions and counseling to high-risk individuals, potentially preventing the onset or progression of diseases. Tumor Type Prediction (<xref ref-type="bibr" rid="B180">180</xref>) is a significant task in disease analysis that focuses on identifying the specific type of tumor a patient may have based on various characteristics such as genetic markers, imaging features, and histopathological findings (<xref ref-type="bibr" rid="B181">181</xref>). Predicting tumor types is essential for determining the most effective treatment strategies and prognostic outcomes for patients with cancer. Pathogenicity Potential Assessment (<xref ref-type="bibr" rid="B27">27</xref>) is a critical task that involves evaluating the ability of pathogens to cause disease in a host. By assessing the pathogenicity potential of different microorganisms, researchers can prioritize the development of interventions against the most virulent pathogens, thereby improving disease prevention and control strategies. Phylogenetic Analysis (<xref ref-type="bibr" rid="B21">21</xref>) is a key component of disease analysis that involves studying the evolutionary relationships between different strains of pathogens or tumor cells. Phylogenetic analysis provides insights into the origins, spread, and diversification of diseases, aiding in the development of targeted interventions and understanding disease transmission dynamics.</p>
</sec>
<sec id="s4">
<title>4 A look on DNA sequence analysis tasks from the perspective of computer scientists</title>
<p>While Section 3 presents a biologically motivated categorization of DNA sequence analysis tasks, this section reframes these same tasks from a computational perspective. This dual categorization approach (biological and computational) aims to facilitate interdisciplinary understanding between life scientists and AI researchers. With the influx of biological data and rise of AI, researchers are increasingly applying AI in diverse areas of molecular biology. Development of large scale AI applications requires a good understanding of variety of sequence analysis tasks. However, there exist a huge domain gap between computer scientists and molecular biologists. Molecular biologists know the need, biological importance, and pharmaceutical worth of different sequence analysis tasks. However, they do not know which machine or deep learning models are most appropriate to use to either replace or complement experimental work. Similarly, computer scientists know which Artificial Intelligence predictive pipeline can potentially perform better with specific type of data; however, they do not know the nature of biological sequence analysis tasks. For instance, DNA sequence analysis tasks such as gene function prediction, gene network reconstruction, gene expression prediction, and disease risk estimation can be challenging for computer scientists to grasp. However, a comprehensive literature review that explains the basics of these tasks can bridge this gap. For example, gene function prediction is a multi-label classification tasks, gene expression prediction is a regression task, while gene network reconstruction and disease risk estimation are binary classification tasks. With this foundational understanding, computer scientists can more easily develop predictive pipelines for these binary, multi-label classification, and regression tasks. To empower AI experts, we have presented 44 DNA sequence analysis tasks in computer scientist language in <xref ref-type="fig" rid="F4">Figure 4</xref>. A simple look on <xref ref-type="fig" rid="F4">Figure 4</xref> reveals that nature of DNA sequence analysis tasks can be categorized into three primary types: regression, clustering, and classification where classification can be further divided into three secondary types: binary classification, multi-class classification, and multi-label classification. Let us mathematically formulate the possible natures of DNA sequence analysis tasks.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>DNA sequence analysis task representation for computer scientist perspective.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1503229-g0004.tif"/>
</fig>
<p>In binary classification, researchers aim to predict the outcome of a binary variable (0 or 1). Given a dataset with features <italic>X</italic> &#x02208; &#x0211D;<sup><italic>nxd</italic></sup>, binary labels <italic>y</italic> &#x02208; 0, 1, and training dataset (<italic>x</italic><sub>1</sub>, <italic>y</italic><sub>1</sub>), (<italic>x</italic><sub>2</sub>, <italic>y</italic><sub>2</sub>), &#x02026;, our goal is to learn a decision function <italic>f</italic>:<italic>X</italic> &#x02192; <italic>Y</italic> that maps inputs to binary outputs 0, 1 on the basis of hypothesis function <italic>h</italic>(<italic>x</italic>) learned from the training data.</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mrow><mml:mi>f</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mtable columnalign='left'><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mn>1</mml:mn></mml:mtd><mml:mtd columnalign='left'><mml:mrow><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mi>h</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mo>&#x02A7E;</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mn>0.5</mml:mn></mml:mrow></mml:mtd></mml:mtr><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mn>0</mml:mn></mml:mtd><mml:mtd columnalign='left'><mml:mrow><mml:mi>o</mml:mi><mml:mi>t</mml:mi><mml:mi>h</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mi>w</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>In multi-class classification, researchers aim to predict the outcome from more than two classes. Specifically, given a dataset having sequences <italic>X</italic> &#x02208; &#x0211D;<sup><italic>nxd</italic></sup>, labels <italic>y</italic> &#x02208; 1, 2, &#x02026;, <italic>K</italic> where <italic>K</italic> is the number of classes, and training dataset (<italic>x</italic><sub>1</sub>, <italic>y</italic><sub>1</sub>), (<italic>x</italic><sub>2</sub>, <italic>y</italic><sub>2</sub>), &#x02026;, (<italic>x</italic><sub><italic>n</italic></sub>, <italic>y</italic><sub><italic>n</italic></sub>) where <italic>x</italic><sub><italic>i</italic></sub> &#x02208; <italic>X</italic> and <italic>y</italic><sub><italic>i</italic></sub> &#x02208; <italic>Y</italic>, our goal is to learn a decision function <italic>f</italic>:<italic>X</italic>&#x02192;<italic>Y</italic> that assigns inputs to one of the classes.</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mi>g</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>h</italic><sub><italic>k</italic></sub>(<italic>x</italic>) is the hypothesis function for class <italic>k</italic> learned from the training data. On the other hand, in multi-label classification, each input can be assigned to multiple classes simultaneously. Given a dataset with features <italic>X</italic> &#x02208; &#x0211D;<sup><italic>nxd</italic></sup>, labels <italic>y</italic> &#x02208; 1, 2, &#x02026;, <italic>K</italic> where <italic>K</italic> is the number of classes, and training dataset (<italic>x</italic><sub>1</sub>, <italic>y</italic><sub>1</sub>, <italic>y</italic><sub>2</sub>, ..), (<italic>x</italic><sub>2</sub>, <italic>y</italic><sub>1</sub>, <italic>y</italic><sub>4</sub>, &#x02026;), &#x02026;, (<italic>x</italic><sub><italic>n</italic></sub>, <italic>y</italic>5, <italic>y</italic><sub><italic>n</italic></sub>, &#x02026;.) where <italic>x</italic><sub><italic>i</italic></sub> &#x02208; <italic>X</italic> and <italic>y</italic><sub><italic>i</italic></sub> &#x02208; <italic>Y</italic>, our goal is to learn a decision function <italic>f</italic>:<italic>X</italic> &#x02192; 0, 1<sup><italic>K</italic></sup> that assigns inputs to multiple classes simultaneously using hypothesis function <italic>h</italic><sub><italic>k</italic></sub>(<italic>x</italic>) for class <italic>k</italic> learned from the training data.</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M3"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Furthermore, in regression, researchers goal is to predict a continuous outcome variable. Given a dataset with sequences <italic>X</italic> &#x02208; &#x0211D;<sup><italic>nxd</italic></sup>, labels <italic>y</italic> &#x02208; &#x0211D;, and training dataset (<italic>x</italic><sub>1</sub>, <italic>y</italic><sub>1</sub>), (<italic>x</italic><sub>2</sub>, <italic>y</italic><sub>2</sub>), &#x02026;, (<italic>x</italic><sub><italic>n</italic></sub>, <italic>y</italic><sub><italic>n</italic></sub>) where <italic>x</italic><sub><italic>i</italic></sub> &#x02208; <italic>X</italic> and <italic>y</italic><sub><italic>i</italic></sub> &#x02208; <italic>Y</italic>, our goal is to learn a function <italic>f</italic>:<italic>X</italic> &#x02192; &#x0211D; that predicts continuous outputs using hypothesis function <italic>h</italic>(<italic>x</italic>) learned from the training data.</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>h</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>In clustering, the goal is to group similar data points into same clusters. Given a dataset with data points <italic>X</italic> &#x0003D; <italic>x</italic><sub>1</sub>, <italic>x</italic><sub>2</sub>, &#x02026;, <italic>x</italic><sub><italic>n</italic></sub>, where each <inline-formula><mml:math id="M5"><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>, our goal is to find a partition of the data into clusters <italic>C</italic> &#x0003D; <italic>C</italic><sub>1</sub>, <italic>C</italic><sub>2</sub>, &#x02026;, <italic>C</italic><sub><italic>K</italic></sub>. This is done on the basis of a distance metric <italic>d</italic>(<italic>x</italic>, &#x003BC;<sub><italic>c</italic></sub>) between data point <italic>x</italic> and the centroid &#x003BC;<sub><italic>c</italic></sub> of cluster <italic>c</italic>.</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mtext class="textrm" mathvariant="normal">argmin</mml:mtext></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:mi>d</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec>
<sec id="s5">
<title>5 DNA sequence analysis databases</title>
<p>This section provides a comprehensive overview of various databases employed to develop benchmark datasets for development of AI-based applications for 44 distinct DNA sequence analysis tasks. A total of 45 DNA sequence databases have been identified from 127 existing studies. Among these, 36 databases are publicly accessible, while the remaining 9 databases are either inaccessible or no longer exist. To ease the lives of researchers and practitioners, <xref ref-type="table" rid="T1">Table 1</xref> summarizes accessible databases in terms of their release year, types of inherent genetic data (DNA, RNA, protein), details of species and organisms, statistics of raw sequences, and supported data formats.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Summary of publicly accessible biological databases, their inherent data types, species diversity, and statistics of raw sequences related to different genomic and proteomic data.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Database name</bold></th>
<th valign="top" align="left"><bold>Release date</bold></th>
<th valign="top" align="left"><bold>Type of data (DNA/RNA/ Protein)</bold></th>
<th valign="top" align="left"><bold>Organism name</bold></th>
<th valign="top" align="left"><bold>Species name</bold></th>
<th valign="top" align="left"><bold>Sequences statistics</bold></th>
<th valign="top" align="left"><bold>Data format</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://descartes.brotmanbaty.org/">Descartes</ext-link></td>
<td valign="top" align="left">2020</td>
<td valign="top" align="left">DNA, RNA</td>
<td valign="top" align="left">Animals</td>
<td valign="top" align="left">Mus musculus, Homo sapiens</td>
<td valign="top" align="left">Human Gene Expression During Development: 4M Cells, 121 Tissues, 15 Organs; Human Chromatin Accessibility During Development: 720K Cells, 53 Tissues, 15 Organs; Mouse: &#x0007E;2M Cells, 61 Embryos</td>
<td valign="top" align="left">.RDS</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="http://lin-group.cn/database/ppd/index.php">PPD</ext-link></td>
<td valign="top" align="left">2020</td>
<td valign="top" align="left">DNA, RNA</td>
<td valign="top" align="left">Bacteria, Archaea</td>
<td valign="top" align="left">63 species</td>
<td valign="top" align="left">129,148 Promoter Sequences, 63 Species with 74 strains</td>
<td valign="top" align="left">.csv</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="http://www.enhanceratlas.org/index.php">EnhancerAtlas 2.0</ext-link></td>
<td valign="top" align="left">2019</td>
<td valign="top" align="left">DNA</td>
<td valign="top" align="left">Animals, Bacteria</td>
<td valign="top" align="left">Homo sapiens, Mus musculus, Drosophila melanogaster, Caenorhabditis elegans, Danio rerio, Rattus norvegicus, Gallus gallus, Sus scrofa, Saccharomyces cerevisiae</td>
<td valign="top" align="left">13,494,603 Enhancers, 586 tissue</td>
<td valign="top" align="left">.csv, BED, R Object</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="http://rna.sysu.edu.cn/dreamBase">DREAM Base</ext-link></td>
<td valign="top" align="left">2018</td>
<td valign="top" align="left">DNA, RNA, Protein</td>
<td valign="top" align="left">Animals</td>
<td valign="top" align="left">Homo sapiens</td>
<td valign="top" align="left">scRNA-Seq Data: 93,3704 cells; RNA-Seq Data: &#x0007E;18,196; ChiP-Seq Data: &#x0007E;10,000; RNA Modification Data: &#x0007E;500; ribo-Seq Data: 1,570; DNase-Seq Data: 599; CLIP-Seq Data: 568</td>
<td valign="top" align="left">.excel, .txt</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="http://bioinfor.imu.edu.cn/emexplorer/public/">EmExplorer database</ext-link></td>
<td valign="top" align="left">2018</td>
<td valign="top" align="left">DNA, Protein</td>
<td valign="top" align="left">Animals</td>
<td valign="top" align="left">Bos taurus, Homo sapiens, Mus musculus, Rattus norvegicus, Sus scrofa</td>
<td valign="top" align="left">158,000 items that contain more than 32,000 development-related Genes under 306 related pathways</td>
<td valign="top" align="left">.txt</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="http://cancer.sanger.ac.uk/">COSMIC</ext-link></td>
<td valign="top" align="left">2018</td>
<td valign="top" align="left">DNA</td>
<td valign="top" align="left">Animals</td>
<td valign="top" align="left">Homo sapiens</td>
<td valign="top" align="left">Total Genomic variants = 24,599,940; Genomic non-coding variants = 16,748,366,406; Genomic mutations within Exons = 768; Genomic mutations within Intronic and other intragenic regions = 9,217,664; Samples = 1,531,613; Fusions = 19,428; Gene expression variants = 9,215,470; Differentially Methylated CpGs = 7,930,489</td>
<td valign="top" align="left">.FASTA, .tsv</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://www.disgenet.org/">DisGeNet</ext-link></td>
<td valign="top" align="left">2015</td>
<td valign="top" align="left">DNA</td>
<td valign="top" align="left">Animals</td>
<td valign="top" align="left">Homo sapiens</td>
<td valign="top" align="left">1,134,942 GDAs between 21,671 Genes, 30,170 diseases and traits; 369,554 VDAs between 194,515 variants and 14,155 diseases and traits</td>
<td valign="top" align="left">.txt, RDF, SQL Dump</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://gnomad.broadinstitute.org/">genomAD</ext-link></td>
<td valign="top" align="left">2014</td>
<td valign="top" align="left">DNA</td>
<td valign="top" align="left">Animals</td>
<td valign="top" align="left">Homo sapiens</td>
<td valign="top" align="left">730,947 Exomes, 76,215 whole Genomes</td>
<td valign="top" align="left">VCF, Hail Table</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/clinvar/">ClinVar</ext-link></td>
<td valign="top" align="left">2013</td>
<td valign="top" align="left">DNA, RNA, Protein</td>
<td valign="top" align="left">Animals</td>
<td valign="top" align="left">Homo sapiens</td>
<td valign="top" align="left">Records = 4391341, Total Genes = 92225</td>
<td valign="top" align="left">.xml, VCF, .tsv</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://hocomoco11.autosome.org/">HOCOMOCO Human v11 database</ext-link></td>
<td valign="top" align="left">2013</td>
<td valign="top" align="left">DNA</td>
<td valign="top" align="left">Animals</td>
<td valign="top" align="left">Mus musculus, Homo sapiens</td>
<td valign="top" align="left">1,443 TF binding models including secondary motif subtypes for 949 human TFs and 720 Mouse orthologs</td>
<td valign="top" align="left">PWM, PFM, PCM, Flat text files</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://tubic.org/deori/">DeOri</ext-link></td>
<td valign="top" align="left">2012</td>
<td valign="top" align="left">DNA</td>
<td valign="top" align="left">Animals</td>
<td valign="top" align="left">Homo sapiens, Mus musculus, Arabidopsis thaliana, Kluyveromyces lactis, Schizosaccharomyces pombe, Drosophila melanogaster</td>
<td valign="top" align="left">189,743 entries</td>
<td valign="top" align="left">.FASTA</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://zhanggroup.org/BioLiP/index.cgi">BioLip</ext-link></td>
<td valign="top" align="left">2012</td>
<td valign="top" align="left">DNA, RNA, Protein</td>
<td valign="top" align="left">Animals</td>
<td valign="top" align="left">Homo sapiens</td>
<td valign="top" align="left">873,925 Entries, 448,816 regular ligands, 191,485 mental ligands, 37,492 Peptide ligands, 43,448 DNA ligands, 152,684 RNA ligands, 873,925 binding affinity data, 451,485 Protein receptors</td>
<td valign="top" align="left">.FASTA</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://tubic.org/deori/">DeOri6.0</ext-link></td>
<td valign="top" align="left">2011</td>
<td valign="top" align="left">DNA</td>
<td valign="top" align="left">Animals, Plants, Fungi</td>
<td valign="top" align="left">17 species</td>
<td valign="top" align="left">189,740 eukaryotic replication origins</td>
<td valign="top" align="left">.FASTA</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://www.ebi.ac.uk/gwas/">GWAS</ext-link></td>
<td valign="top" align="left">2008</td>
<td valign="top" align="left">DNA</td>
<td valign="top" align="left">Animals</td>
<td valign="top" align="left">Homo sapiens</td>
<td valign="top" align="left">146,394 TASs</td>
<td valign="top" align="left">.tsv, OWL/RD</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://depmap.org/portal/">Broad DepMap</ext-link></td>
<td valign="top" align="left">2008</td>
<td valign="top" align="left">DNA</td>
<td valign="top" align="left">Animals</td>
<td valign="top" align="left">Homo sapiens</td>
<td valign="top" align="left">2,000 Human cancer cell lines</td>
<td valign="top" align="left">.csv, .txt</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://sites.broadinstitute.org/ccle/">CCLE</ext-link></td>
<td valign="top" align="left">2008</td>
<td valign="top" align="left">DNA, RNA, Protein</td>
<td valign="top" align="left">Animals</td>
<td valign="top" align="left">Homo sapiens</td>
<td valign="top" align="left">1,019 RNA cell lines, 954 microRNA expression profiles, 899 Protein lines, 897 Genome-wide histone modifications, 843 DNA methylation, 329 whole Genome Sequencing, 326 whole exome Sequencing</td>
<td valign="top" align="left">.csv</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://www.gencodegenes.org/">GENCODE</ext-link></td>
<td valign="top" align="left">2006</td>
<td valign="top" align="left">DNA</td>
<td valign="top" align="left">Animals</td>
<td valign="top" align="left">Homo sapiens, Mus musculus</td>
<td valign="top" align="left">Homo sapiens: Total genes = 63,086, Total transcripts = 254,070, Total distinct Translations = 65,650; Mus musculus: Total Genes = 57,132, Total Transcripts = 149,138, Total distinct Translations = 44,819</td>
<td valign="top" align="left">.txt</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/clinvar/">Consensus Coding Sequence Database</ext-link></td>
<td valign="top" align="left">2005</td>
<td valign="top" align="left">DNA</td>
<td valign="top" align="left">Animals</td>
<td valign="top" align="left">Homo sapiens, Mus musculus</td>
<td valign="top" align="left">35,608 CCDS IDs that correspond to 19,107 Genes, with 48,062 Protein Sequences</td>
<td valign="top" align="left">.FASTA</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://www.gsea-msigdb.org/gsea/index.jsp">MSigDB</ext-link></td>
<td valign="top" align="left">2005</td>
<td valign="top" align="left">DNA</td>
<td valign="top" align="left">Animals</td>
<td valign="top" align="left">Homo sapiens, Mus musculus</td>
<td valign="top" align="left">8,380 Gene set</td>
<td valign="top" align="left">.gct, .res, .pcl, .txt, .cls, .gmx, .gmt, .grp, .xml, .chip, .rnk</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://geneontology.org/">Gene Ontology</ext-link></td>
<td valign="top" align="left">2004</td>
<td valign="top" align="left">DNA</td>
<td valign="top" align="left">Animals, Bacteria, Fungi, Plants</td>
<td valign="top" align="left">Escherichia coli, Homo sapiens, Oryza sativa, Saccharomyces cerevisiae, Schizosaccharomyces pombe, Mus musculus, many more</td>
<td valign="top" align="left">Annotated Gene products = 1,536,921; Annotated Species = 5,409; Annotated Species with over 1,000 annotations = 183</td>
<td valign="top" align="left">OBO, OWL, GAF, GPAD, GPI, JSON</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://jaspar.elixir.no/">JASPAR</ext-link></td>
<td valign="top" align="left">2004</td>
<td valign="top" align="left">DNA, RNA, Protein</td>
<td valign="top" align="left">Fungi, Insects, Nematoda, Plants, Urochordata, Vertebrata</td>
<td valign="top" align="left">34 species</td>
<td valign="top" align="left">4279 Profiles</td>
<td valign="top" align="left">.txt</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://tubic.org/deg_bak/">Database of Essential Genes</ext-link></td>
<td valign="top" align="left">2004</td>
<td valign="top" align="left">DNA, RNA, Protein</td>
<td valign="top" align="left">Bacteria, Archaea, Eukaryotes</td>
<td valign="top" align="left">51 species</td>
<td valign="top" align="left">53,885 essential Genes, 786 essential non-coding Sequences</td>
<td valign="top" align="left">.csv, DAT</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://www.encodeproject.org/">ENCODE</ext-link></td>
<td valign="top" align="left">2003</td>
<td valign="top" align="left">DNA, RNA, Protein</td>
<td valign="top" align="left">Animals</td>
<td valign="top" align="left">Homo sapiens, Mus musculus</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">.FASTA, BAM, BigWig, BED, VCF</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://dbtss.hgc.jp/">DataBase of Transcriptional Start Sites</ext-link></td>
<td valign="top" align="left">2002</td>
<td valign="top" align="left">DNA</td>
<td valign="top" align="left">Animals</td>
<td valign="top" align="left">Homo sapiens, Mus musculus</td>
<td valign="top" align="left">491M TSS tag Sequences from a total of 20 tissues and 7 cell cultures</td>
<td valign="top" align="left">.FASTA, .csv, .xlsx, .txt</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://horizondiscovery.com/en/gene-modulation/overexpression/cdna-and-orfs/products/mgc-cdnas">MGC</ext-link></td>
<td valign="top" align="left">2001</td>
<td valign="top" align="left">DNA</td>
<td valign="top" align="left">Animals</td>
<td valign="top" align="left">Mus musculus, Homo sapiens, Rat, Bovine</td>
<td valign="top" align="left">Total MGC Full ORF Clones: Homo sapiens: 29,818; Mus musculus: 27,285; Rat: 6,763; Bovine: 9,104; Non-redundant Genes: Homo sapiens: 17,592; Mus musculus: 17,701; Rat: 6,486; Bovine: 8,724</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/geo/">GEO</ext-link></td>
<td valign="top" align="left">2000</td>
<td valign="top" align="left">DNA, RNA, Protein</td>
<td valign="top" align="left">Animals</td>
<td valign="top" align="left">20 species</td>
<td valign="top" align="left">Samples = 7,209,691</td>
<td valign="top" align="left">SOFT, MINiML, .txt</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/10592221/">Exon-Intron Database</ext-link></td>
<td valign="top" align="left">1999</td>
<td valign="top" align="left">DNA, RNA</td>
<td valign="top" align="left">Animals, Plants</td>
<td valign="top" align="left">Homo sapiens, Mus musculus, Drosophila melanogaster, C. Elegans, S. Pombe</td>
<td valign="top" align="left">42,460 Genes (243,589 Exons)</td>
<td valign="top" align="left">.FASTA</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="http://www.ensembl.org/index.html">Ensembl</ext-link></td>
<td valign="top" align="left">1999</td>
<td valign="top" align="left">DNA, RNA, Protein</td>
<td valign="top" align="left">Animals</td>
<td valign="top" align="left">Homo sapiens, Mus musculus, Danio rerio, Sus scrofa</td>
<td valign="top" align="left">Genomes = 44,048, Ensembl Fungi = 1,014 Genomes, Ensembl Metazoa = 78 Genomes (Invertebrate species), Genomes for vertebrate Species = 236, Ensembl Plants = 67 Genomes, Ensembl Protists = 237 Genomes</td>
<td valign="top" align="left">.FASTA, GTF, GFF, MySQL Dump</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://regulondb.ccg.unam.mx/">RegulonDB</ext-link></td>
<td valign="top" align="left">1998</td>
<td valign="top" align="left">DNA, RNA, Protein</td>
<td valign="top" align="left">Bacteria</td>
<td valign="top" align="left">Escherichia coli</td>
<td valign="top" align="left">4,748 Genes, 2,590 Operons, 287 Regulons, 3,718 Transcription Unit, 4050 Promoters</td>
<td valign="top" align="left">.tsv, .csv</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://epd.expasy.org/epd/">EPD2</ext-link></td>
<td valign="top" align="left">1998</td>
<td valign="top" align="left">DNA</td>
<td valign="top" align="left">Animals, Plant, Fungi, Protists</td>
<td valign="top" align="left">139 species</td>
<td valign="top" align="left">4,806 Promoters</td>
<td valign="top" align="left">.FASTA, EMBL</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/snp/">dbSNP</ext-link></td>
<td valign="top" align="left">1998</td>
<td valign="top" align="left">DNA</td>
<td valign="top" align="left">Human</td>
<td valign="top" align="left">Homo sapiens</td>
<td valign="top" align="left">Nearly 2 billion submissions representing more than 675 million distinct variants; 23.7 million refSNP entries (14.5 million validated)</td>
<td valign="top" align="left">ASN.1, FASTA, XML</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://www.kegg.jp/">KEGG</ext-link></td>
<td valign="top" align="left">1995</td>
<td valign="top" align="left">DNA, RNA, Protein</td>
<td valign="top" align="left">Animals, Plants, Fungi, Protists, Bacteria, Archaea</td>
<td valign="top" align="left">Euryarchaeota Candidatus, Thermoplasmatota, Thermoproteota, Chordata, Echinodermata, Hemichordata, Ascomycota, Basidiomycota, Atribacterota Candidatus, Saccharibacteria</td>
<td valign="top" align="left">Genes = 53,674,741, Addendum Proteins = 4,181, Viral Genes=688,823, Viral mature Peptides = 377</td>
<td valign="top" align="left">KGML, .FASTA, .txt</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/">NCBI</ext-link></td>
<td valign="top" align="left">1988</td>
<td valign="top" align="left">DNA, RNA, Protein</td>
<td valign="top" align="left">Multiple organisms</td>
<td valign="top" align="left">Multiple species</td>
<td valign="top" align="left">Hub of databases</td>
<td valign="top" align="left">.FASTA, XML</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://epd.expasy.org/epd/">Eukaryotic Promoter Database</ext-link></td>
<td valign="top" align="left">1986</td>
<td valign="top" align="left">DNA</td>
<td valign="top" align="left">Animals, Plants, Fungi, Invertebrates</td>
<td valign="top" align="left">Homo sapiens, Macaca mulatta, Mus musculus, Rattus norvegicus, Gallus gallus, Canis familiaris, Drosophila melanogaster, Apis mellifera, Danio rerio, Caenorhabditis elegans, Arabidopsis thaliana; Zea mays, Saccharomyces cerevisiae, Schizosaccharomyces pombe, Plasmodium falciparum</td>
<td valign="top" align="left">192,586 Promoters, 163,676 Genes</td>
<td valign="top" align="left">.FASTA, EMBL</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/genbank/">GenBank</ext-link></td>
<td valign="top" align="left">1982</td>
<td valign="top" align="left">DNA</td>
<td valign="top" align="left">Animals, Archaea, Bacteria, Fungi, Plants, Virus</td>
<td valign="top" align="left">557000 species</td>
<td valign="top" align="left">3,213,818,003,787 Bases, 250,803,006 Sequences</td>
<td valign="top" align="left">.gb</td>
</tr>
<tr>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://www.omim.org/">OMIM</ext-link></td>
<td valign="top" align="left">1960</td>
<td valign="top" align="left">DNA</td>
<td valign="top" align="left">Animals</td>
<td valign="top" align="left">Homo sapiens</td>
<td valign="top" align="left">17,290 Gene descriptions, 18 Gene and Phenotypes, 8,361 Phenotype description, 1,736 Phenotypes suspected mendelian</td>
<td valign="top" align="left">.txt</td>
</tr></tbody>
</table>
</table-wrap>
<p>A holistic view of the <xref ref-type="table" rid="T1">Table 1</xref> reveals that 12 databases provide RNA and protein sequences as well in addition to providing DNA sequences. As word embeddings methods and large language models are trained in unsupervised fashion and when they are trained on large sequence data usually, they produce better representations. To efficiently train word embedding methods and large language models, raw data can be acquired from these databases. To facilitate researchers, we have categorized 36 databases into three different categories on the basis of volume of raw sequences: low sequence facilitators, medium sequence facilitators, and high sequence facilitators. Specifically, 13 low sequence facilitators, namely, HOCOMOCO Human v11 database (<xref ref-type="bibr" rid="B182">182</xref>), Consensus Coding Sequence Database (<xref ref-type="bibr" rid="B183">183</xref>), MSigDB (<xref ref-type="bibr" rid="B184">184</xref>), Broad DepMap (<xref ref-type="bibr" rid="B185">185</xref>), JASPAR (<xref ref-type="bibr" rid="B186">186</xref>), Database of Essential Genes (<xref ref-type="bibr" rid="B187">187</xref>), ENCODE (<xref ref-type="bibr" rid="B188">188</xref>), MGC (<xref ref-type="bibr" rid="B189">189</xref>), Exon-Intron Database (<xref ref-type="bibr" rid="B190">190</xref>), Ensembl (<xref ref-type="bibr" rid="B191">191</xref>), RegulonDB (<xref ref-type="bibr" rid="B192">192</xref>), EPD2 (<xref ref-type="bibr" rid="B193">193</xref>), offer up to 100,000 DNA sequences each, while 9 medium sequence facilitators, namely, PPD (<xref ref-type="bibr" rid="B194">194</xref>), DREAM (<xref ref-type="bibr" rid="B195">195</xref>), EmExplorer database (<xref ref-type="bibr" rid="B196">196</xref>), GenomAD (<xref ref-type="bibr" rid="B197">197</xref>), DeOri (<xref ref-type="bibr" rid="B198">198</xref>), BioLip (<xref ref-type="bibr" rid="B199">199</xref>), DeOri6.0 (<xref ref-type="bibr" rid="B198">198</xref>), GWAS (<xref ref-type="bibr" rid="B200">200</xref>), Eukaryotic Promoter Database (<xref ref-type="bibr" rid="B193">193</xref>), provide up to 1 million DNA sequences. In contrast, 13 high sequence facilitators such as Descartes (<xref ref-type="bibr" rid="B201">201</xref>), EnhancerAtlas 2.0 (<xref ref-type="bibr" rid="B202">202</xref>), COSMIC (<xref ref-type="bibr" rid="B203">203</xref>), DisGeNet (<xref ref-type="bibr" rid="B204">204</xref>), ClinVar (<xref ref-type="bibr" rid="B205">205</xref>), CCLE (<xref ref-type="bibr" rid="B206">206</xref>), GENCODE (<xref ref-type="bibr" rid="B207">207</xref>), Gene Ontology (<xref ref-type="bibr" rid="B208">208</xref>), DataBase of Transcriptional Start Sites (<xref ref-type="bibr" rid="B209">209</xref>), GEO (<xref ref-type="bibr" rid="B210">210</xref>), KEGG (<xref ref-type="bibr" rid="B211">211</xref>), NCBI (<xref ref-type="bibr" rid="B212">212</xref>), GenBank (<xref ref-type="bibr" rid="B213">213</xref>), and dbSNP (<xref ref-type="bibr" rid="B214">214</xref>, <xref ref-type="bibr" rid="B215">215</xref>) offer more than 1 million DNA sequences each. These databases predominantly house DNA sequences from a diverse array of species, including humans, mice, plants, bacteria, and fungi. A comprehensive analysis reveals that approximately 22 databases, namely, Descartes (<xref ref-type="bibr" rid="B201">201</xref>), DREAM (<xref ref-type="bibr" rid="B195">195</xref>), EmExplorer database (<xref ref-type="bibr" rid="B196">196</xref>), COSMIC (<xref ref-type="bibr" rid="B203">203</xref>), DisGeNet (<xref ref-type="bibr" rid="B204">204</xref>), GenomAD (<xref ref-type="bibr" rid="B197">197</xref>), ClinVar (<xref ref-type="bibr" rid="B205">205</xref>), HOCOMOCO Human v11 database (<xref ref-type="bibr" rid="B182">182</xref>), DeOri (<xref ref-type="bibr" rid="B198">198</xref>), BioLip (<xref ref-type="bibr" rid="B199">199</xref>), GWAS (<xref ref-type="bibr" rid="B200">200</xref>), Broad DepMap (<xref ref-type="bibr" rid="B185">185</xref>), CCLE (<xref ref-type="bibr" rid="B206">206</xref>), GENCODE (<xref ref-type="bibr" rid="B207">207</xref>), Consensus Coding Sequence Database (<xref ref-type="bibr" rid="B183">183</xref>), MSigDB (<xref ref-type="bibr" rid="B184">184</xref>), ENCODE (<xref ref-type="bibr" rid="B188">188</xref>), DataBase of Transcriptional Start Sites (<xref ref-type="bibr" rid="B209">209</xref>), MGC (<xref ref-type="bibr" rid="B189">189</xref>), GEO (<xref ref-type="bibr" rid="B210">210</xref>), Ensembl (<xref ref-type="bibr" rid="B191">191</xref>), and OMIM (<xref ref-type="bibr" rid="B216">216</xref>), focus on animal DNA sequences, 4 databases including PPD (<xref ref-type="bibr" rid="B194">194</xref>), Database of Essential Genes (<xref ref-type="bibr" rid="B187">187</xref>), RegulonDB and (<xref ref-type="bibr" rid="B192">192</xref>) on bacterial sequences, and JASPAR (<xref ref-type="bibr" rid="B186">186</xref>) on plant DNA sequences. EnhancerAtlas 2.0 (<xref ref-type="bibr" rid="B202">202</xref>) is the only database that facilitates with both animal and bacterial DNA sequences, while 4 databases namely DeOri6.0 (<xref ref-type="bibr" rid="B198">198</xref>), Exon-Intron Database (<xref ref-type="bibr" rid="B190">190</xref>), EPD2 (<xref ref-type="bibr" rid="B193">193</xref>), and Eukaryotic Promoter Database (<xref ref-type="bibr" rid="B193">193</xref>) focus on animal and plant DNA sequences, whereas Gene Ontology (<xref ref-type="bibr" rid="B208">208</xref>), KEGG (<xref ref-type="bibr" rid="B211">211</xref>), and GenBank (<xref ref-type="bibr" rid="B213">213</xref>) provide DNA sequences for animal, plant, and bacteria. In addition, sequences from other organisms such as eukaryotes, invertebrates, fungi, and various microorganisms are also well-represented. Some databases encompass a broad spectrum of species. For instance, the EDP2 (<xref ref-type="bibr" rid="B193">193</xref>) database includes genomics data for 139 species, GenBank (<xref ref-type="bibr" rid="B213">213</xref>) houses sequences for 557,000 species, and PPD (<xref ref-type="bibr" rid="B194">194</xref>) has genomics data of 63 species.</p>
<p>Moreover, <xref ref-type="table" rid="T1">Table 1</xref> includes data formats utilized by various databases to manage and provide access to DNA sequences. TXT and FASTA format are universally accepted by almost all DNA sequence analysis programs. Each entry in both format types contains at least two lines: First line or header includes accession number, species name, or identification details, while next line contains nucleotide sequences. CSV and TSV are text-based formats in which values in rows are separated by commas or tabs, respectively. In both file formats, first row specifies headers which defines names of columns (&#x0201C;SeqID&#x0201D;, &#x0201C;SeqName&#x0201D;, &#x0201C;Type&#x0201D;, &#x0201C;Function&#x0201D;) and subsequent rows represent data. In VCF format, first row specifies headers which defines names of columns, but this format is specifically used to store genetic variation data including single nucleotide polymorphisms (SNPs), insertions, deletions, and structural variants. In addition, XLSX formats represent complex datasets that contain information computed with various formulas across multiple columns, whereas EMBL format includes structured sections for sequence data, feature annotations (genes and other biological features), organism information, references, and other details. An extensive analysis of <xref ref-type="table" rid="T1">Table 1</xref> reveals that most widely used data formats are FASTA, TXT, CSV, XLSX, and EMBL in DNA sequence analysis.</p>
<p>A rigorous analysis of <xref ref-type="table" rid="T1">Table 1</xref> reveals that out of 36 publicly accessible databases, several key categories of data emerge. Four databases, namely, Broad DepMap (<xref ref-type="bibr" rid="B185">185</xref>), genomAD, COSMIC, and MGC, provide data for DNA functional analysis tasks such as prediction of context-specific functional impact of genetic variants and conserved non-coding element classification. Seven databases, namely, BioLip, HOCOMOCO Human v11, GWAS, EnhancerAtlas 2.0, DataBase of Transcriptional Start Sites, Exon-Intron Database, and Eukaryotic Promoter Database, offer data on gene expression regulation. Three databases, namely, PPD, CCLE, and EmExplorer, focus on DNA modification data including methylcytosine and methyladenine modifications. In addition, DeOri, Descartes, DeOri6.0, and JASPAR provide information on gene structure and stability, including chromatin accessibility prediction, YY1-mediated chromatin loop identification, and DNA replication origins identification. GENCODE, Consensus Coding Sequence Database, MSigDB, Gene Ontology, DisGeNet, Database of Essential Genes, KEGG, and NCBI offer comprehensive gene analysis data. Furthermore, eight other databases, namely, EPD, ENCODE, RegulonDB, GEO, Ensembl, ClinVar, GenBank, and OMIM, provide a range of data on gene expression regulation, DNA modification prediction, genome structure and stability, DNA functional analysis, disease information, and gene analysis.</p>
</sec>
<sec id="s6">
<title>6 DNA sequence analysis benchmark datasets</title>
<p>The quality and quantity of datasets utilized in AI-driven DNA sequence analysis applications are vital determinants of their effectiveness and functionality. This section aims to provide a comprehensive overview of datasets relevant to 44 distinct DNA sequence analysis tasks. Overall, these datasets fall into two primary categories: publicly available datasets and in-house datasets. This categorization serves to illuminate the significance of dataset accessibility and its implications for the advancement of AI-driven DNA sequence analysis. Specifically, publicly available datasets are accessible to the wider research community and are commonly employed in the development of AI-based predictive models. They serve as foundational resources that facilitate the advancement of AI-driven DNA sequence analysis pipelines by ensuring accessibility, reusability, and transparency in research endeavors. Furthermore, the utilization of publicly available datasets fosters collaboration and knowledge exchange within the scientific community, thereby contributing to the overall progress of the field. In contrast, in-house datasets are proprietary in nature and are developed within specific research laboratories or institutions. These datasets often contain sensitive data tailored to particular research objectives. As in-house datasets cannot be shared publicly, their proprietary nature may limit broader access, reproducibility, and applicability of findings.</p>
<p>Rigorous assessment of <bold>127</bold> existing studies reveals that a total of <bold>242</bold> benchmark datasets related to 44 distinct DNA sequence analysis tasks are constructed or acquired from existing literature. Specifically, among these 242 benchmark datasets, <bold>199</bold> are publicly available and <bold>43</bold> are in-house datasets. <xref ref-type="table" rid="T2">Table 2</xref> provides the distribution of public and in-house datasets for 44 distinct DNA sequence analysis tasks. It provides information about which of these datasets are used by word embeddings, large language models, nucleotide composition, and positional information-based predictive pipelines.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Overview of 199 public and 43 in-house datasets used across 44 different DNA sequence analysis tasks.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Task name</bold></th>
<th valign="top" align="left"><bold>Task type</bold></th>
<th valign="top" align="left" colspan="2"><bold>Datasets used in language models</bold></th>
<th valign="top" align="left" colspan="2"><bold>Datasets used in word embeddings</bold></th>
<th valign="top" align="left" colspan="2"><bold>Datasets used in other methods</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#919498;color:#ffffff">
<td/>
<td/>
<td valign="top" align="left"><bold>Public</bold></td>
<td valign="top" align="left"><bold>In-house</bold></td>
<td valign="top" align="left"><bold>Public</bold></td>
<td valign="top" align="left"><bold>In-house</bold></td>
<td valign="top" align="left"><bold>Public</bold></td>
<td valign="top" align="left"><bold>In-house</bold></td>
</tr>
<tr>
<td valign="top" align="left">DNA replication origins identification</td>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Gao et al. <italic>(A. thaliana)</italic> (<xref ref-type="bibr" rid="B25">25</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Wu et al. datasets (<italic>S. cerevisiae</italic> Dataset, <italic>S. pombe</italic> Dataset, <italic>K. lactis</italic> Dataset, <italic>P. pastoris</italic> Dataset) (<xref ref-type="bibr" rid="B329">329</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Nucleosome position detection</td>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">Gangi et al. datasets (CE, DM, YS, HM, DM-5U, DM-PM, DM-LC, HM-5U, HM-LC, HM-PM, YS-PM) (<xref ref-type="bibr" rid="B330">330</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Chromatin accessibility prediction</td>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">DeepSEA datasets (TF, DHS) (<xref ref-type="bibr" rid="B32">32</xref>), DNase-Seq experiment data (<xref ref-type="bibr" rid="B31">31</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">YY1-Mediated chromatin loops prediction</td>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Dao et al. DeepYY1 datasets (HCT116, K562) (<xref ref-type="bibr" rid="B39">39</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Zhang et al. DeepYY1 datasets (HCT116, K562) (<xref ref-type="bibr" rid="B38">38</xref>)</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Genome structure analysis</td>
<td valign="top" align="left">Multi-class classification</td>
<td valign="top" align="left">NCycDB dataset (<xref ref-type="bibr" rid="B27">27</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Chromatin feature prediction</td>
<td valign="top" align="left">Multi-label classification</td>
<td valign="top" align="left">Logo919 (<xref ref-type="bibr" rid="B34">34</xref>), Logo2002 (<xref ref-type="bibr" rid="B34">34</xref>), Logo3357 (<xref ref-type="bibr" rid="B34">34</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Long-range chromatin interaction prediction</td>
<td valign="top" align="left">Interaction</td>
<td valign="top" align="left">Chip-seq dataset (<xref ref-type="bibr" rid="B36">36</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Enhancers identification</td>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">Liu et al. dataset (<xref ref-type="bibr" rid="B57">57</xref>), Liao et al. datasets (HEK293, NHEK, K652, GM12878, HMEC, HSMM, NHLF, HUVEC) (<xref ref-type="bibr" rid="B58">58</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Liu et al. dataset (<xref ref-type="bibr" rid="B57">57</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">DiseaseEnhancer (<xref ref-type="bibr" rid="B55">55</xref>), EnDisease (<xref ref-type="bibr" rid="B55">55</xref>), CancerEnD (<xref ref-type="bibr" rid="B55">55</xref>)</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Promoter identification</td>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">Yang et al. dataset (<xref ref-type="bibr" rid="B34">34</xref>), Ji et al. dataset (<xref ref-type="bibr" rid="B90">90</xref>), Xiao et al. dataset (<xref ref-type="bibr" rid="B331">331</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Wang et al. dataset [<italic>H. Sapiens</italic>-I (TATA-containing), <italic>H. Sapiens-II</italic> (TATA-less), <italic>R. Norvegicus-I</italic> (TATA-containing), <italic>R. Norvegicus-II</italic> (TATA-less), <italic>D. melanogaster-I</italic> (TATA-containing), <italic>D. melanogaster-II</italic> (TATA-less), <italic>Z. mays-I</italic> (TATA-containing), <italic>Z. mays-II</italic> (TATA-less)] (<xref ref-type="bibr" rid="B236">236</xref>), Zhang et al. dataset (K562, GM12878, HeLa-S3, HUVEC) (<xref ref-type="bibr" rid="B76">76</xref>), Xiao et al. dataset (<xref ref-type="bibr" rid="B331">331</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Yang et al. dataset (<xref ref-type="bibr" rid="B34">34</xref>), Xiao et al. (<xref ref-type="bibr" rid="B331">331</xref>)</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Enhancer-promoter interactions prediction</td>
<td valign="top" align="left">Interaction/binary classification</td>
<td valign="top" align="left">Yang et al. datasets (FoeT, Mon, nCD4, tB, tCD4, tCD8) (<xref ref-type="bibr" rid="B249">249</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Whalen et al. dataset (GM12878, HUVEC, HeLa-S3, IMR90, K562, NHEK) (<xref ref-type="bibr" rid="B329">329</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Whalen et al. dataset (GM12878, HUVEC, HeLa-S3, IMR90, K562, NHEK) (<xref ref-type="bibr" rid="B329">329</xref>)</td>
<td valign="top" align="left">Zhang et al. (GM12878 cell line, HeLa cell line) (<xref ref-type="bibr" rid="B332">332</xref>)</td>
</tr>
<tr>
<td valign="top" align="left">Transcription sites prediction</td>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">Clauwaert et al. dataset (<xref ref-type="bibr" rid="B50">50</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Transcription factor binding sites prediction</td>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">ChIP-Seq dataset (<xref ref-type="bibr" rid="B94">94</xref>), TSSs dataset (<xref ref-type="bibr" rid="B91">91</xref>), 497 TF ChIP-Seq dataset (<xref ref-type="bibr" rid="B90">90</xref>), 690 ChIP-Seq (<xref ref-type="bibr" rid="B51">51</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Shen et al. datasets (A549 dataset, MCF-7 Dataset, H1-HESCDataset, HUVEC dataset) (<xref ref-type="bibr" rid="B92">92</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Transcription factor binding affinity prediction</td>
<td valign="top" align="left">Multi-class classification</td>
<td valign="top" align="left">Weirauch et al. dataset (PBM Dataset), Jolma et al. dataset (HT-SELEX Dataset) (<xref ref-type="bibr" rid="B52">52</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Protein-DNA binding sites prediction</td>
<td valign="top" align="left">Interaction/binary classification</td>
<td valign="top" align="left">690 ChIP-Seq dataset (<xref ref-type="bibr" rid="B96">96</xref>), Patiyal et al. dataset (<xref ref-type="bibr" rid="B53">53</xref>), Xia et al. (dataset 2) (<xref ref-type="bibr" rid="B53">53</xref>), Liu and Tian (dataset 1, dataset 2) (<xref ref-type="bibr" rid="B95">95</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Splice sites prediction</td>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">Wang et al. dataset (<xref ref-type="bibr" rid="B100">100</xref>), Ji et al. dataset (<xref ref-type="bibr" rid="B93">93</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Splice-junction gene sequence dataset (<xref ref-type="bibr" rid="B97">97</xref>), Degroeve et al. (<xref ref-type="bibr" rid="B98">98</xref>), Liu et al. Datasets [<italic>O. sativa</italic> (Acceptor, Donor), <italic>A. Thaliana</italic> (Acceptor, Donor), <italic>H. sapiens</italic> (Acceptor, Donor)] (<xref ref-type="bibr" rid="B99">99</xref>)</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Translation initiation sites</td>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">Clauwaert et al. dataset (<xref ref-type="bibr" rid="B50">50</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Kalkatawi et al. TIS (<xref ref-type="bibr" rid="B101">101</xref>)</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Essential genes identification</td>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">Ma et al. datasets (<italic>S. cerevisiae, E. coli, H. sapiens, D. melanogaster</italic>) (<xref ref-type="bibr" rid="B109">109</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Ma et al. datasets (<italic>S. cerevisiae, E. coli, H. sapiens, D. melanogaster</italic>) (<xref ref-type="bibr" rid="B109">109</xref>), Campos et al. datasets (<italic>D. melanogaster, M. maripaludis, H. sapiens, C. elegans</italic>) (<xref ref-type="bibr" rid="B333">333</xref>), Zhang et al. (<xref ref-type="bibr" rid="B106">106</xref>), Xiao et al. dataset (<xref ref-type="bibr" rid="B107">107</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Campos et al. datasets (<italic>D. melanogaster, M. maripaludis, H. sapiens, C. elegans</italic>) (<xref ref-type="bibr" rid="B333">333</xref>), Sharma et al. (<xref ref-type="bibr" rid="B104">104</xref>)</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Disease genes prediction</td>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">Nunes et al. dataset (<xref ref-type="bibr" rid="B110">110</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Pseudogene function prediction</td>
<td valign="top" align="left">Interaction/binary classification</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Fan et al. dataset (CC, MF, BP) (<xref ref-type="bibr" rid="B111">111</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Target gene classification</td>
<td valign="top" align="left">Multi-class classification</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Arango et al. dataset (<xref ref-type="bibr" rid="B113">113</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Candidate gene prioritization &#x00026; Selection</td>
<td valign="top" align="left">Multi-class classification</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Toufiq et al. dataset (<xref ref-type="bibr" rid="B114">114</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Gene functions prediction</td>
<td valign="top" align="left">Multi-label classification</td>
<td valign="top" align="left">Hu et al. (Human gene set from gene ontology) (<xref ref-type="bibr" rid="B112">112</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">GTEx (CC, MF, BP) (<xref ref-type="bibr" rid="B111">111</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Gene expression prediction</td>
<td valign="top" align="left">Regression</td>
<td valign="top" align="left">Reddy et al. dataset (Jurkat, K-562, THP-1) (<xref ref-type="bibr" rid="B334">334</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Al Taweraqi et al., dataset (<xref ref-type="bibr" rid="B103">103</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Gene taxonomy classification</td>
<td valign="top" align="left">Multi-class classification</td>
<td valign="top" align="left">Mock et al. dataset (<xref ref-type="bibr" rid="B123">123</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Verma et al. dataset (<xref ref-type="bibr" rid="B121">121</xref>)</td>
<td valign="top" align="left">CAMI2 Airway dataset (<xref ref-type="bibr" rid="B122">122</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Gene network reconstruction</td>
<td valign="top" align="left">Multi-class classification</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">SynTReN dataset (<xref ref-type="bibr" rid="B128">128</xref>) DREAM5 dataset (<xref ref-type="bibr" rid="B128">128</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Pio et al. dataset (<xref ref-type="bibr" rid="B126">126</xref>), Schaffter et al. datasets (DREAM4 10, DREAM4 100) (<xref ref-type="bibr" rid="B127">127</xref>), Jozefczuk et al. datasets (<italic>E.coli</italic> cold, <italic>E.coli</italic> head, <italic>E.coli</italic> oxidative) (<xref ref-type="bibr" rid="B127">127</xref>)</td>
</tr>
<tr>
<td valign="top" align="left">4mc-Methyl-cytosine modification prediction</td>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">Xu et al. datasets (<italic>C. elegans, D. Malenogaster, A. Thaliana, E.coli, G. subterraneus, G. pickeringii</italic>) (<xref ref-type="bibr" rid="B141">141</xref>), Chen et al., datasets (<italic>C. elegans, D. Malenogaster, A. Thaliana, E.coli, G. subterraneus, G. pickeringii</italic>) (<xref ref-type="bibr" rid="B135">135</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Khanal et al. datasets (<italic>F. vesca, R. chinensis</italic>) (<xref ref-type="bibr" rid="B138">138</xref>), Zulfiqar et al. dataset (<xref ref-type="bibr" rid="B136">136</xref>), Zeng et al. dataset (<xref ref-type="bibr" rid="B137">137</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Khanal et al. (<xref ref-type="bibr" rid="B138">138</xref>), Chen et al., Datasets (<italic>C. elegans, D. Malenogaster, A. Thaliana, E.coli, G. subterraneus, G. pickeringii</italic>) (<xref ref-type="bibr" rid="B135">135</xref>)</td>
<td valign="top" align="left">Manavalan et al. (<xref ref-type="bibr" rid="B139">139</xref>)</td>
</tr>
<tr>
<td valign="top" align="left">5mc-Methyl-Cytosine modification prediction</td>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">Wang et al. dataset (<xref ref-type="bibr" rid="B282">282</xref>), Stanojevic et al. dataset (GM24385, NA12878, NA19240, H1ESc, K562, HX1) (<xref ref-type="bibr" rid="B152">152</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Wang et al. dataset (<xref ref-type="bibr" rid="B282">282</xref>), <italic>Hyb</italic>_2021 (<italic>C. elegans, D. Malenogaster, A. Thaliana, E.coli, G. subterraneus, G. pickeringii</italic>) (<xref ref-type="bibr" rid="B142">142</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Xu et al. (<xref ref-type="bibr" rid="B141">141</xref>), Rao Zeng et al. (<xref ref-type="bibr" rid="B140">140</xref>), Saha et al. (<xref ref-type="bibr" rid="B135">135</xref>)</td>
<td valign="top" align="left">Nguyen-Vo et al. (<xref ref-type="bibr" rid="B139">139</xref>)</td>
</tr>
<tr>
<td valign="top" align="left">5hmc-hydroxy-methylcytosine modification prediction</td>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Lv et al. dataset (<italic>M. musculus, H. sapiens</italic>) (<xref ref-type="bibr" rid="B156">156</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">6mA-methyladenine modification prediction</td>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">Abbas et al. dataset (<italic>A. thaliana, H. sapiens, M. musculus, S. cerevisiae</italic>) (<xref ref-type="bibr" rid="B335">335</xref>), DNA 6 mA dataset (<xref ref-type="bibr" rid="B281">281</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Lv et al. dataset (<xref ref-type="bibr" rid="B156">156</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Zhou et al. dataset (<italic>A. thaliana, C. elegans, C. equisetifolia, D. melanogaster, F. vesca, H. sapiens, R. chinensis, S. cerevisiae, T. thermophile</italic>, Xos. BLS256) (<xref ref-type="bibr" rid="B148">148</xref>), Fan et al. dataset (1, D.malenogaster, 3, 4, 5) (<xref ref-type="bibr" rid="B149">149</xref>)</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Methylation modification prediction</td>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">Lv et al. dataset 6mA (<italic>T. thermophile, A. thaliana, H. sapiens</italic>, Xos. BLS256, <italic>D. melanogaster, C. elegans, C. equisetifolia, S. cerevisiae</italic>, Tolypocladium, <italic>F. vesca, R. chinensis</italic>) 5hmC (<italic>M. musculus, H. sapiens</italic>) 4mC (<italic>F. vesca, Tolypcladium, S. cerevisiae, C. equisetifolia</italic>) (<xref ref-type="bibr" rid="B156">156</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Conserved non-coding element classification</td>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Polychronopoulos et al. dataset (<xref ref-type="bibr" rid="B163">163</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Functional prioritization of non-coding variants</td>
<td valign="top" align="left">Multi-class classification</td>
<td valign="top" align="left">Logo919 (<xref ref-type="bibr" rid="B34">34</xref>) Logo2002 (<xref ref-type="bibr" rid="B34">34</xref>) Logo3357 (<xref ref-type="bibr" rid="B34">34</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Exon &#x00026; Intron region classification</td>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">Akalin et al. dataset (<xref ref-type="bibr" rid="B164">164</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Recombination spots identification</td>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Liu et al. dataset (<xref ref-type="bibr" rid="B165">165</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Species classification</td>
<td valign="top" align="left">Multi-class classification</td>
<td valign="top" align="left">Mouse enhancers (<xref ref-type="bibr" rid="B44">44</xref>) Coding vs. intergenomic (<xref ref-type="bibr" rid="B44">44</xref>) human vs. worm (<xref ref-type="bibr" rid="B44">44</xref>) Human enhancers cohn (<xref ref-type="bibr" rid="B44">44</xref>) human enhancers ensembl (<xref ref-type="bibr" rid="B44">44</xref>) Human Regulatory (<xref ref-type="bibr" rid="B44">44</xref>) human nontata promoter (<xref ref-type="bibr" rid="B44">44</xref>) human OCR ensembl ()</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Prediction of context-specific functional impact of genetic variants</td>
<td valign="top" align="left">Multi-class classification</td>
<td valign="top" align="left">eQTLs dataset (<xref ref-type="bibr" rid="B36">36</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Nitrogen cycle prediction</td>
<td valign="top" align="left">Multi-class classification</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">NCycDB (<xref ref-type="bibr" rid="B27">27</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Pathogen signature identification</td>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">DS500 dataset (<xref ref-type="bibr" rid="B171">171</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Phage-host interactions prediction</td>
<td valign="top" align="left">Interaction/binary classification</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">ESKAPE dataset (<xref ref-type="bibr" rid="B177">177</xref>), Wang et al. dataset (<xref ref-type="bibr" rid="B176">176</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Qiu et al. dataset (Kingdom, Phylum, Class, Order, Family, Genus) (<xref ref-type="bibr" rid="B175">175</xref>)</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Mutation susceptibility analysis</td>
<td valign="top" align="left">Multi-class classification</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Yilmaz et al. dataset (Human, Mouse) (<xref ref-type="bibr" rid="B173">173</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Tumor type prediction</td>
<td valign="top" align="left">Multi-class classification</td>
<td valign="top" align="left">TCGA pan-cancer dataset (<xref ref-type="bibr" rid="B180">180</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Pathogenicity potential assessment</td>
<td valign="top" align="left">Multi-class classification</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">E-K12 (<xref ref-type="bibr" rid="B27">27</xref>), CARD-A (<xref ref-type="bibr" rid="B27">27</xref>), CARD-D (<xref ref-type="bibr" rid="B27">27</xref>), CARD-R (<xref ref-type="bibr" rid="B27">27</xref>), VFDB (<xref ref-type="bibr" rid="B27">27</xref>), ENZYME (<xref ref-type="bibr" rid="B27">27</xref>), PATRIC (<xref ref-type="bibr" rid="B27">27</xref>), NCycDB (<xref ref-type="bibr" rid="B27">27</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Phylogenetic analysis</td>
<td valign="top" align="left">Clustering</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Ren et al. datasets (<xref ref-type="bibr" rid="B111">111</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Disease risks estimation</td>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">HSCR-RET (<xref ref-type="bibr" rid="B90">90</xref>), HSCR-RET-Long (<xref ref-type="bibr" rid="B90">90</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr></tbody>
</table>
</table-wrap>
<p>For each DNA sequences analysis task, public and in-house datasets are distributed as DNA Replication Origins Identification (0, 5), Nucleosome Position Detection (11, 0), Chromatin Accessibility Prediction (2, 0), YY1-Mediated Chromatin Loop Prediction (4, 0), Genome structure analysis (0, 1), Chromatin Feature Prediction (3, 0), Long-range chromatin interaction prediction (1, 0), Enhancers Identification (12, 0), Promoter Identification (15, 0), Enhancer-Promoter Interactions Prediction (18, 2), Transcription Site Prediction (1, 0), Transcription Factor Binding Site Prediction (4, 4), Transcription Factor Binding Affinity Prediction (2, 0), Protein-DNA Binding Site Prediction (5, 0), Splice Site Prediction (10, 0), Translation Initiation Sites (1, 1), Essential Gene Identification (6, 5), Disease Gene Prediction (1, 0), Pseudogene Function Prediction (3, 0), Target Gene Classification (1, 0), Candidate Gene Prioritization/ Identification (0, 1), Gene Function Prediction (4, 0), Gene Expression Prediction (4, 0), Gene Taxonomy Classification (2, 1), Gene Network Reconstruction (2, 6), 4mc-Methylcytosine Site Prediction (16, 0), 6mA-Methyladenine Site Prediction (5, 0), 5mc-Methylcytosine Site Prediction (24, 1), 5hmc-Methylcytosine Site Prediction (2, 0), Methylation Site Prediction (17, 0), Conserved Non-Coding Elements Classification (0, 1), Functional Priorizitation of non-coding variants (3, 0), Exon and Intron Region Classification (0, 1), Recombination Spots Identification (1, 0), Species Classification (8, 0), Prediction of context-specific functional impact of genetic variant (1, 0), Nitrogen Cycle Prediction (0, 1), Pathogen Signatures Identification (0, 1), Phage-Host Interactions Prediction (8, 0), Mutation Susceptibility Analysis (0, 2), Tumor Type Prediction (1, 0), Pathogenicity Potential Assessment (0, 8), Phylogenetic Analysis (0, 1), and Disease Risks Estimation (2, 0). First entry in brackets refers to count of public datasets, and second entry indicates total number of in-house datasets for a particular task. For example, in &#x0201C;Essential Gene Identification (6, 5)&#x0201D; task, 6 refers to public datasets while 5 represents in-house datasets.</p>
<p>A holistic view of <xref ref-type="table" rid="T2">Table 2</xref> reveals 110 public and 18 in-house datasets are employed to develop both word embeddings and language models based predictive pipelines for 12 DNA sequence analysis tasks, namely, DNA replication origins identification, enhancers identification, promoters identification, enhancer-promoter interaction prediction, transcription factor binding site prediction, essential gene identification, gene function prediction, gene expression prediction, gene taxonomy classification, 4mC-methyl cytosine modification prediction, 5mC-methl cytosine modification, and 6mA-methyl modification prediction. Notably, both types of predictive pipelines have utilized 1 common dataset to evaluate the performance of predictive models developed for three tasks, namely, enhancer identification, essential gene identification, and 5mC-methyl cytosine modification prediction.</p>
<p>Furthermore, 112 public and 15 in-house datasets are used to develop both word embedding and nucleotide compositional and positional information-based predictive pipelines for 11 DNA sequence analysis tasks including essential gene identification, gene network reconstruction, 4mC-methyl cytosine modification prediction, 5mC-modification prediction, 6mA-methyl adenine modification prediction, and phage-host interaction prediction. However, both predictive pipelines have used 9 common pubic dataset for only three tasks. Specifically, six public datasets for enhancer-promoter interactions prediction, one public data for essential gene identification, and two public datasets for 4mC-Methyl cytosine modification prediction are commonly employed by both predictive pipelines.</p>
<p>Moreover, <xref ref-type="table" rid="T2">Table 2</xref> highlights that 107 public and 9 in-house datasets are utilized by 9 DNA sequence analysis tasks, namely, enhancers identification, promoters identification, enhancer-promoter interaction prediction, splice site prediction, translation initiation sites identification, essential gene identification, 4mC-methyl cytosine modification prediction, 5mC-methl cytosine modification, and 6mA-methyl modification prediction for developing both language models and nucleotide compositional and positional information-based predictive pipelines. Merely, 7 public datasets are used commonly by both predictive pipelines for two tasks: one for promoter identification and six for 4mC-methyl cytosine modification prediction.</p>
<p>Although all three different types of representation learning-based predictive pipelines are employed across six different DNA sequence analysis tasks, namely, enhancers identification, promoters identification, enhancer-promoter interaction prediction 4mC-methyl cytosine modification prediction, 5mC-methl cytosine modification, and 6mA-methyl modification prediction, only on one task, namely, promoters identification, all three kinds of predictive pipelines are evaluated on one common dataset. These statistics reveal that researchers have focused on creating new datasets for each kind of predictive pipelines instead of using existing datasets. Consequently, this domain lacks a fair performance comparison between different kinds of predictive pipelines.</p>
</sec>
<sec id="s7">
<title>7 A brief look on representation learning methods and predictors used in DNA sequence analysis predictive pipelines</title>
<p>This section dives into 12 most commonly used word embedding approaches, 8 large language models, 9 machine learning, 8 deep learning, and 3 statistical algorithms that are used in development of predictive pipelines for 44 different DNA sequence analysis tasks.</p>
<sec>
<title>7.1 DNA sequence representation learning using word embeddings</title>
<p>In the domain of natural language processing (NLP), the introduction of word embedding techniques represented a significant advancement by enabling the development of more accurate machine and deep learning predictive models. These approaches assign statistical vectors to words by capturing contextual representations of words within extensive, unlabelled corpora (<xref ref-type="bibr" rid="B217">217</xref>, <xref ref-type="bibr" rid="B218">218</xref>). The primary objective is to assign comparable vectors to semantically similar words and distinct vectors to dissimilar words (<xref ref-type="bibr" rid="B217">217</xref>, <xref ref-type="bibr" rid="B218">218</xref>). Leveraging transfer learning strategies, these contextual word representations have empowered data-hungry deep learning models to achieve exceptional performance, even with limited training data. Following the success of word embeddings in various NLP tasks (<xref ref-type="bibr" rid="B217">217</xref>&#x02013;<xref ref-type="bibr" rid="B220">220</xref>), researchers have adopted these approaches for genomic and proteomic sequence analysis tasks, which share similarities with NLP tasks. This section offers a comprehensive overview of 12 distinct word embedding approaches that are utilized in DNA sequence analysis predictive pipelines. <xref ref-type="fig" rid="F5">Figure 5</xref> visually illustrates the utilization of various word embedding methods in conjunction with different machine and deep learning algorithms.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Utilization of 12 different word embedding approaches and 8 large language models in diverse DNA sequence analysis pipelines based on a variety of machine and deep learning predictors such that RF, random forest; DF, deep forest; SVM, support vector machine; LogR, logistic regression; NB, Naive Bayes; kNN, k-nearest neighbors; MLP, multilayer perceptron; CNN, convolutional neural network; GNN, graph neural network; GCN, graph convolutional network; TCN, temporal convolutional network; GAT, graph attention network; LSTM, long short-term memory; BiLSTM, bidirectional Llong short-term memory; BiGRU, bidirectional gated recurrent unit; PCT, predictive clustering tree; CRF, conditional random field; FGM, fast gradient method. All language models are used with self-classifiers and few language models like transformer, ULMFiT, GPT, and BERT are also used with separate standalone or hybrid algorithms.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1503229-g0005.tif"/>
</fig>
<p>These word embedding approaches leveraged for DNA sequence analysis tasks can be categorized into two types: (1) non-graph-based methods and (2) graph-based methods. Non-graph-based methods segregate DNA sequences into overlapping or non-overlapping k-mers. Specifically, overlapping k-mers are generated by sliding a fixed-size window over sequence with a smaller stride as compared to window size. For instance, for ACGTG sequence with a window size of 4 and a stride of 1, the k-mers generated are ACGT and CGTA. Alternatively, in non-overlapping k-mers generation, window and stride size must be equal in size. For same sequence used in the overlapping case, this non-overlapping approach generates only one k-mer, such as ACGT. The length of the k-mer is determined by the window size. Researchers often create pre-trained embeddings with different k-mer sizes and then select the size which yields best performance in downstream tasks. Once k-mers are generated, these k-mers sequences are passed to traditional word embedding models (Word2vec, FastText, GloVe) to generate representation.</p>
<p>A high level overview of <xref ref-type="fig" rid="F5">Figure 5</xref> indicates that various studies have explored the potential of the Word2Vec embedding method in combination with 13 different machine and deep learning algorithms as well as 2 statistical algorithms. The era of word embedding approaches begin in 2013 with introduction of Word2Vec (<xref ref-type="bibr" rid="B221">221</xref>). Word2vec has two different embeddings generation paradigm: (1) SkipGram and (2) Continuous Bag of Words (CBoW). SkipGram learns representations of k-mers by predicting surrounding k-mers for every k-mers of corpus. The number of surrounding k-mers is a hyper-parameter that can be adjusted according to available data. Contrarily, CBoW model learns k-mers representations by predicting single k-mer based on the context of its surrounding k-mers. Similar to SkipGram model, here context of surrounding k-mers is a hyper-parameter. Word2vec architecture is comprised of input layer, hidden layer, and an output layer. At input layer, a random d-dimensional vector is initialized for each k-mer, while the hidden layer extracts relationships between k-mers. These relationships are further passed to output layer, which predicts probabilities of output k-mers based on the context of input k-mers. The predicted probabilities are passed to loss function which computes loss value. To facilitate readers, <xref ref-type="disp-formula" rid="E10">Equation 10</xref> embodies mathematical expressions for computing loss values of both variants.</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M7"><mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>k</mml:mi><mml:mi>i</mml:mi><mml:mi>p</mml:mi><mml:mi>G</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:msubsup><mml:mrow><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x0007C;</mml:mo><mml:msub><mml:mi>W</mml:mi><mml:mi>s</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>C</mml:mi><mml:mi>B</mml:mi><mml:mi>o</mml:mi><mml:mi>W</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:msubsup><mml:mrow><mml:mstyle displaystyle='true'><mml:msub><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mi>&#x003F5;</mml:mi><mml:msub><mml:mi>W</mml:mi><mml:mi>s</mml:mi></mml:msub></mml:mrow></mml:msub><mml:mrow><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo>&#x0007C;</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>In above expression, N refer to number of k-mers, <italic>w</italic><sub><italic>i</italic></sub> indicates target k-mers, <italic>w</italic><sub><italic>j</italic></sub> is one of k-mers within contextual window, and <italic>W</italic><sub><italic>s</italic></sub> refers to set of k-mers in contextual windows of k-mers <italic>w</italic><sub><italic>i</italic></sub>. After computing loss, weights are updated during back propagation which eventually helps in generating similar vectors for similar k-mers and distinct vectors for dissimilar k-mers.</p>
<p>Pennington et al. (<xref ref-type="bibr" rid="B222">222</xref>) proposed another k-mers embedding approach named Global Vectors (GloVe) which generates k-mers vectors by capturing both global and local contextual information of k-mer within corpora. It can be seen in <xref ref-type="fig" rid="F5">Figure 5</xref>, in the context of DNA sequence analysis, the potential of Glove k-mers embedding method is explored with two distinct deep learning methods. Primarily, this embedding generation method computes local and global contextual information by incorporating occurrence frequencies of k-mer pairs into an objective function shown in <xref ref-type="disp-formula" rid="E7">Equation 7</xref>.</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M8"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>J</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x02200;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:munder></mml:mstyle><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mtext>&#x000A0;</mml:mtext><mml:mi>&#x003F5;</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mi>G</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi><mml:mi>P</mml:mi><mml:mi>a</mml:mi><mml:mi>i</mml:mi><mml:mi>r</mml:mi><mml:mi>s</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>In above expression, <italic>w</italic><sub><italic>i</italic></sub> and <italic>w</italic><sub><italic>j</italic></sub> are k-mers within a pair, <italic>b</italic><sub><italic>i</italic></sub> and <italic>b</italic><sub><italic>j</italic></sub> are corresponding biases, and <italic>f</italic>(<italic>C</italic><sub><italic>ij</italic></sub>) is a weighted function to normalize co-occurrence matrix values and eradicate biases and their impact of noise on k-mers embeddings.</p>
<p><xref ref-type="fig" rid="F5">Figure 5</xref> shows that Word2Vec is the most commonly explored word embedding method, followed by FastText. Mikolov et al. (<xref ref-type="bibr" rid="B223">223</xref>) proposed FastText approach by extending the working paradigm of Word2Vec model. Primarily, this approach handles out-of-vocabulary (OOV) k-mers by discretizing k-mers into sub k-mers. After generating sub k-mers. it takes average of sub k-mers vectors to generate k-mers vectors and passed them to word2vec model. During back propagation, it updates vectors of both k-mers and sub k-mers. Through this strategy, vectors are generated for both k-mers and sub k-mers.</p>
<p>Furthermore, in NLP domain, with an aim to generate more comprehensive vectors of k-mers by capturing k-mers informative patterns from textual corpora, researchers have proposed different graph-based methods. These approaches include DeepWalk (<xref ref-type="bibr" rid="B224">224</xref>), Node2Vec (<xref ref-type="bibr" rid="B225">225</xref>), Graph2Vec (<xref ref-type="bibr" rid="B226">226</xref>), SDNE (<xref ref-type="bibr" rid="B227">227</xref>), SocDim (<xref ref-type="bibr" rid="B228">228</xref>), GraRep (<xref ref-type="bibr" rid="B229">229</xref>), Laplacian Eigenmaps (<xref ref-type="bibr" rid="B230">230</xref>), Locally Linear Embedding (<xref ref-type="bibr" rid="B231">231</xref>), and OPA2Vec (<xref ref-type="bibr" rid="B232">232</xref>). <xref ref-type="fig" rid="F5">Figure 5</xref> highlights that within the context of DNA sequence analysis, the potential of the 7 graph-based methods is less explored compared to the two foundational word embedding methods, Word2vec and FastText. In addition, among the graph-based methods, Node2Vec (<xref ref-type="bibr" rid="B225">225</xref>) has been investigated more extensively than DeepWalk (<xref ref-type="bibr" rid="B224">224</xref>), Graph2Vec (<xref ref-type="bibr" rid="B226">226</xref>), SDNE (<xref ref-type="bibr" rid="B227">227</xref>), SocDim (<xref ref-type="bibr" rid="B228">228</xref>), GraRep (<xref ref-type="bibr" rid="B229">229</xref>), Laplacian Eigenmaps (<xref ref-type="bibr" rid="B230">230</xref>), Locally Linear Embedding (<xref ref-type="bibr" rid="B231">231</xref>), and OPA2Vec (<xref ref-type="bibr" rid="B232">232</xref>). Similar to non-graph-based methods, graph-based methods segregate sequences into k-mers and generate k-mers pairs by sliding a 2 size window over k-mers sequences. By using k-mers paris, a graph is formed where nodes represent k-mers, and edges represent relationships between the k-mers. For example, to generate a graph from the DNA sequence ACTGCA with k = 3, first, overlapping k-mers (ACT, CTG, TGC, GCA) are generated. By sliding a window of size 2 over these k-mers sequence, k-mers pairs [(ACT, CTG), (CTG, TGC), and (TGC, GCA)] are created. These pairs form edges of graph, with k-mers serving as nodes. Perrozi et al. (<xref ref-type="bibr" rid="B224">224</xref>) proposed DeepWalk approach that utilizes graphical space to generate new sequences by capturing extensive relationships between k-mers. After generating new sequences, it makes use of Word2Vec model for generation of k-mers vectors. In contrast, Grover et al. (<xref ref-type="bibr" rid="B225">225</xref>) proposed Node2Vec approach that utilizes a distinct strategy for generation of new sequences. Primarily, Node2Vec employs second order random walk sampling strategy which reaps the benefits of breath first search (BFS) and depth first search (DFS) algorithms. This strategy computes probability of visiting next node depending on the previously visited nodes rather than just randomly selecting one of neighboring nodes. Naeayanan et al., (<xref ref-type="bibr" rid="B226">226</xref>) introduced another embedding generation approach namely Graph2Vec. It extracts root node, its sub-graph, and degree of intended sub-graph to generate a sorted list of nodes which is then passed to SkipGram with negative sampling (SGNS) model.</p>
<p>Matrix factorization embedding approaches extend graph-based embedding approaches by using adjacency matrix rather than generating new sequences directly from graph. Adjacency matrix encodes the relationships between nodes within the graph which is then decomposed using matrix factorization methods namely SVD and NMF. These approaches also decompose adjacency matrix of graph into lower-dimensional matrices which represents node embeddings. These embeddings extract nodes latent features and relationships between them. Mainly, matrix factorization methods aim to minimize reconstruction error between original adjacency matrix and reconstructed matrix from node embeddings. These methods include Laplacian Eigenmaps (<xref ref-type="bibr" rid="B230">230</xref>), Locally Linear Embedding (<xref ref-type="bibr" rid="B231">231</xref>), SDNE (<xref ref-type="bibr" rid="B227">227</xref>), SocDim (<xref ref-type="bibr" rid="B228">228</xref>), GraRep (<xref ref-type="bibr" rid="B229">229</xref>), and OPA2Vec (<xref ref-type="bibr" rid="B232">232</xref>). A closer view of <xref ref-type="fig" rid="F5">Figure 5</xref> indicates that 6 matrix factorization embedding approaches method are least explored as compared to foundational word embedding methods (Word2vec, FastText) and graph-based methods.</p>
<p>Laplacian Eigenmaps (<xref ref-type="bibr" rid="B230">230</xref>) approach derives degree matrix from adjacency matrix and computes graph Laplacian matrix by computing the difference between degree matrix and adjacency matrix. Next, it computes eigen values and constructs eigenvectors corresponding to smallest non-zero eigenvalues which results in generating lower-dimensional k-mer embeddings and preserving local k-mers relationships. Another matrix representation approach graph representations (GraRep) (<xref ref-type="bibr" rid="B229">229</xref>) make use of adjacency (<italic>A</italic><sub><italic>i, j</italic></sub>) and degree (<italic>D</italic><sub><italic>i, j</italic></sub>) matrices driven from nodes and edges of graph. <xref ref-type="disp-formula" rid="E8">Equation 8</xref> depicts mathematical expression for computing proximity matrix from <italic>A</italic><sub><italic>i, j</italic></sub> and <italic>D</italic><sub><italic>i, j</italic></sub> matrices.</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M9"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>V</mml:mi></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where V represents total number of nodes in graphs. Calculated proximity matrix is further passed to singular value decomposition approach for generating k-mer embeddings. Moreover, this approach focuses on extracting similarities between nodes by using k-step information of neighbors where levels of neighbors can be represented through k-steps. Similar to GraRep approach, SocDim (<xref ref-type="bibr" rid="B228">228</xref>) generates k-mer representations by incorporating social dimensions, namely, attributes and network structures. Specifically in SocDim, adjacency and degree matrices are used to compute modularity matrix by using <xref ref-type="disp-formula" rid="E9">Equation 9</xref>.</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M10"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>A</mml:mi><mml:mo>-</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mi>m</mml:mi></mml:mrow></mml:mfrac><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>D</mml:mi><mml:msup><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where m represents edges, and D represents degree matrix. Similar to GraRep, modularity matrix is passed to SVD for k-mers embeddings generation.</p>
<p>Moreover, structural deep network embedding (SDNE) (<xref ref-type="bibr" rid="B227">227</xref>) leverages deep auto-encoders to generate k-mer embeddings by determining first and second order proximities to ensure connected k-mers have similar embeddings. SDNE model architecture is trained to optimize combined loss function that incorporates both proximities and finally generates low-dimensional representations by capturing non-linear relationships between nodes and encoding structural information into embeddings. Afterward, structural embedding aims to address limitations of k-mer embeddings approaches in capturing structural and semantic information of nodes and edges in heterogeneous networks. Among structural embedding approaches, Opa2Vec (<xref ref-type="bibr" rid="B232">232</xref>) makes use of individual entities containing structured knowledge or characterized classes axioms and unstructured information or metadata, i.e., textual annotations and passes them semantic reasoner tool (Elk/HermiT) for generating ontology sequence which is then passed to Word2Vec model for generating representations. Locally linear embedding (LLE) (<xref ref-type="bibr" rid="B231">231</xref>) method identifies neighboring k-mers for each k-mer in the sequence and determines weights by employing graph Laplacian concept which linearly reconstructs each k-mer from its neighbors. Afterward, it computes sum of edges between close k-mers by using heat-kernel method which ensures weights of connected k-mers as 1 and unconnected k-mers as 0, ultimately maintaining the reconstruction relationship. These weights extract both semantic and syntactic information and maintain the reconstruction relationship. By optimizing reconstruction error and computing eigenvectors, LLE generates embeddings for each k-mer in the sequence. These embeddings represent the k-mers in a reduced-dimensional space, where similar k-mers in context are closer together.</p>
<p>Specifically, for DNA sequence analysis tasks, word embeddings methods are being utilized to generate pre-trained embeddings in 2 different ways: In one way, sequences are segregated into k-mers and embeddings of k-mers are generated. In second way, embeddings are generated for whole DNA sequence. Moreover, most of the DNA sequence analysis predictors follow first way to generate embeddings (<xref ref-type="bibr" rid="B21">21</xref>, <xref ref-type="bibr" rid="B39">39</xref>, <xref ref-type="bibr" rid="B45">45</xref>, <xref ref-type="bibr" rid="B47">47</xref>&#x02013;<xref ref-type="bibr" rid="B49">49</xref>, <xref ref-type="bibr" rid="B58">58</xref>, <xref ref-type="bibr" rid="B59">59</xref>, <xref ref-type="bibr" rid="B64">64</xref>&#x02013;<xref ref-type="bibr" rid="B66">66</xref>, <xref ref-type="bibr" rid="B76">76</xref>, <xref ref-type="bibr" rid="B78">78</xref>, <xref ref-type="bibr" rid="B79">79</xref>, <xref ref-type="bibr" rid="B82">82</xref>, <xref ref-type="bibr" rid="B92">92</xref>, <xref ref-type="bibr" rid="B105">105</xref>, <xref ref-type="bibr" rid="B106">106</xref>, <xref ref-type="bibr" rid="B113">113</xref>, <xref ref-type="bibr" rid="B121">121</xref>, <xref ref-type="bibr" rid="B122">122</xref>, <xref ref-type="bibr" rid="B136">136</xref>&#x02013;<xref ref-type="bibr" rid="B138">138</xref>, <xref ref-type="bibr" rid="B151">151</xref>, <xref ref-type="bibr" rid="B153">153</xref>, <xref ref-type="bibr" rid="B163">163</xref>, <xref ref-type="bibr" rid="B165">165</xref>, <xref ref-type="bibr" rid="B171">171</xref>, <xref ref-type="bibr" rid="B173">173</xref>, <xref ref-type="bibr" rid="B233">233</xref>&#x02013;<xref ref-type="bibr" rid="B235">235</xref>), but second way is utilized by only few tasks including gene-disease association prediction (<xref ref-type="bibr" rid="B110">110</xref>), pseudogene function prediction (<xref ref-type="bibr" rid="B111">111</xref>), promoter identification (<xref ref-type="bibr" rid="B236">236</xref>), essential gene prediction (<xref ref-type="bibr" rid="B107">107</xref>, <xref ref-type="bibr" rid="B108">108</xref>), gene network reconstruction (<xref ref-type="bibr" rid="B128">128</xref>), and gene expression prediction (<xref ref-type="bibr" rid="B103">103</xref>). In this section, we have defined methods from first way perspective. A comprehensive detail about second way is available in following articles (<xref ref-type="bibr" rid="B103">103</xref>, <xref ref-type="bibr" rid="B107">107</xref>, <xref ref-type="bibr" rid="B108">108</xref>, <xref ref-type="bibr" rid="B110">110</xref>, <xref ref-type="bibr" rid="B111">111</xref>, <xref ref-type="bibr" rid="B128">128</xref>, <xref ref-type="bibr" rid="B236">236</xref>).</p>
<p>In a nutshell, word embedding approaches have significantly propelled 44 distinct DNA sequence analysis tasks, enriching the research community with the development of robust and precise models. Notably, conventional word embedding techniques such as Word2Vec, GloVe, and FastText excel in capturing k-mers context and sub k-mers information effectively. In contrast, innovative techniques such as Graph2Vec, Node2Vec, DeepWalk, and GraRep harness graph-based methodologies to enhance embeddings based on connectivity and proximities. In addition, SocDim and OPA2Vec offer distinctive perspectives by integrating social and ontological elements, while SDNE combines local and global structural insights through deep autoencoders. Locally linear embedding (LLE) and Laplacian eigenmaps are dedicated to preserving local geometric properties. Ultimately, each approach makes a distinctive contribution to driving significant progress in DNA sequence analysis.</p>
</sec>
<sec>
<title>7.2 DNA sequence representation learning using language models</title>
<p>In the evolving landscape of natural language processing (NLP), the inception of the Transformer model has announced a new era of advancements, setting the precedent for subsequent developments in language models (<xref ref-type="bibr" rid="B237">237</xref>, <xref ref-type="bibr" rid="B238">238</xref>). The Transformer and distinct language models, including BERT, GPT-3, and ELECTRA, have significantly contributed to pushing the boundaries of what machines can understand and generate in terms of human language (<xref ref-type="bibr" rid="B237">237</xref>, <xref ref-type="bibr" rid="B238">238</xref>). The importance of these models lies not only in their ability to comprehend and produce text but also in their application across different domains including genomics and proteomics sequence analysis (<xref ref-type="bibr" rid="B239">239</xref>). These models have found multifarious applications in genomics and proteomics sequence analysis tasks by generating highly effective representations of biological sequences (<xref ref-type="bibr" rid="B239">239</xref>). To facilitate DNA sequence analysis researchers, here we briefly delve into the key features, advantages, and disadvantages of commonly used eight modern sophisticated language models, namely, Transformer (<xref ref-type="bibr" rid="B102">102</xref>), Transformer-XL (<xref ref-type="bibr" rid="B50">50</xref>), XLNet (<xref ref-type="bibr" rid="B156">156</xref>), ULMFIT, BERT (<xref ref-type="bibr" rid="B156">156</xref>), ALBERT (<xref ref-type="bibr" rid="B156">156</xref>), ELECTRA (<xref ref-type="bibr" rid="B156">156</xref>), and GPT-3 (<xref ref-type="bibr" rid="B240">240</xref>). <xref ref-type="table" rid="T3">Table 3</xref> presents 8 distinct language models and their variants, categorized into 4 different groups based on their architectures. These architectures include trivial LSTM-based language model, encoder-decoder architecture, encoder-only architecture, and decoder-only architecture. <xref ref-type="table" rid="T3">Table 3</xref> also provides information about language model architecture and outlines number of layers as well as count of encoders or decoders and their respective layers.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Summary of 8 contemporary language models used in DNA sequence analysis.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Architecture type</bold></th>
<th valign="top" align="left"><bold>Language model, release year</bold></th>
<th valign="top" align="left"><bold>Language model variants</bold></th>
<th valign="top" align="left"><bold>Number of layers in encoders</bold></th>
<th valign="top" align="left"><bold>Number of layers in decoders</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Trivial LSTM based language model</td>
<td valign="top" align="left">ULMFiT, (<xref ref-type="bibr" rid="B243">243</xref>), 2018</td>
<td valign="top" align="left">AWD-LSTM language model</td>
<td valign="top" align="left">1</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Encoder-decoder</td>
<td valign="top" align="left">Transformer_XL (<xref ref-type="bibr" rid="B336">336</xref>), 2019</td>
<td valign="top" align="left">Transformer_XL Large (WikiText-103)</td>
<td valign="top" align="left">24</td>
<td valign="top" align="left">24</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">24L Transformer_XL (text8)</td>
<td valign="top" align="left">24</td>
<td valign="top" align="left">24</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">12L Transformer (enwik8)</td>
<td valign="top" align="left">12</td>
<td valign="top" align="left">12</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">18L Transformer (enwik8)</td>
<td valign="top" align="left">18</td>
<td valign="top" align="left">18</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">24L Transformer (enwik8)</td>
<td valign="top" align="left">24</td>
<td valign="top" align="left">24</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">Transformer_XL base (Billion Word)</td>
<td valign="top" align="left">12</td>
<td valign="top" align="left">12</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">Transformer_XL large (Billion Word)</td>
<td valign="top" align="left">24</td>
<td valign="top" align="left">24</td>
</tr>
<tr>
<td valign="top" align="left">Encoder-decoder</td>
<td valign="top" align="left">Transformer, (<xref ref-type="bibr" rid="B241">241</xref>), 2017</td>
<td valign="top" align="left">Base</td>
<td valign="top" align="left">6</td>
<td valign="top" align="left">6</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">Big</td>
<td valign="top" align="left">6</td>
<td valign="top" align="left">6</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">ELECTRA, (<xref ref-type="bibr" rid="B246">246</xref>), 2020</td>
<td valign="top" align="left">Small</td>
<td valign="top" align="left">12</td>
<td valign="top" align="left">_</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">Base</td>
<td valign="top" align="left">12</td>
<td valign="top" align="left">_</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">Large</td>
<td valign="top" align="left">24</td>
<td valign="top" align="left">_</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">ALBERT, (<xref ref-type="bibr" rid="B245">245</xref>), 2020</td>
<td valign="top" align="left">Base</td>
<td valign="top" align="left">12</td>
<td valign="top" align="left">_</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">Large</td>
<td valign="top" align="left">24</td>
<td valign="top" align="left">_</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">xLarge</td>
<td valign="top" align="left">24</td>
<td valign="top" align="left">_</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">xxLarge</td>
<td valign="top" align="left">12</td>
<td valign="top" align="left">_</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">BERT, (<xref ref-type="bibr" rid="B244">244</xref>), 2019</td>
<td valign="top" align="left">Base</td>
<td valign="top" align="left">12</td>
<td valign="top" align="left">_</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">Large</td>
<td valign="top" align="left">24</td>
<td valign="top" align="left">_</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">XL-Net, (<xref ref-type="bibr" rid="B242">242</xref>), 2019</td>
<td valign="top" align="left">Base</td>
<td valign="top" align="left">12</td>
<td valign="top" align="left">_</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">Large</td>
<td valign="top" align="left">24</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Decoder-only</td>
<td valign="top" align="left">GPT, 2018</td>
<td valign="top" align="left">GPT-1 (<xref ref-type="bibr" rid="B337">337</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">12</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">GPT-2 small (<xref ref-type="bibr" rid="B338">338</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">12</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">GPT-2 medium (<xref ref-type="bibr" rid="B338">338</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">24</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">GPT-2 Large (<xref ref-type="bibr" rid="B338">338</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">36</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">GPT-3 (<xref ref-type="bibr" rid="B247">247</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">96</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">GPT-4 (<xref ref-type="bibr" rid="B339">339</xref>)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">120</td>
</tr></tbody>
</table>
</table-wrap>
<p>The Transformer model, introduced in 2017 by Vaswani et al., (<xref ref-type="bibr" rid="B241">241</xref>) marks a significant departure from previous models that relied on recurrent or convolutional neural networks for processing sequential data. This model utilizes a unique architecture that focuses on attention mechanisms which allows to handle long-range dependencies and understand the context and semantics of sequences more effectively (<xref ref-type="bibr" rid="B102">102</xref>, <xref ref-type="bibr" rid="B241">241</xref>). Key innovations of the Transformer include positional encoding and self-attention mechanisms (<xref ref-type="bibr" rid="B102">102</xref>, <xref ref-type="bibr" rid="B241">241</xref>). Positional encoding assigns a unique number to each individual k-mer or group of k-mers and helps in grasping k-mers order and sequence context. The self-attention mechanism allows the model to weigh the importance of each k-mer in relation to others, enhancing its ability to process and predict scientific language patterns (<xref ref-type="bibr" rid="B102">102</xref>, <xref ref-type="bibr" rid="B241">241</xref>). The main advantage of the Transformer is its efficiency in training and inference due to parallel processing of sequences (<xref ref-type="bibr" rid="B102">102</xref>, <xref ref-type="bibr" rid="B241">241</xref>). However, it requires substantial computational resources, which can be a limiting factor in resource-constrained environments. Despite this, its flexibility and scalability in handling diverse genomics tasks make it a preferred choice in many advanced AI applications (<xref ref-type="bibr" rid="B102">102</xref>).</p>
<p>Transformer-XL extends the Transformer architecture to address the limitation of fixed-length context by incorporating mechanisms that capture long-range dependencies more effectively (<xref ref-type="bibr" rid="B50">50</xref>). This model enhances the ability to maintain context over longer sequences than standard Transformer models, which significantly improves performance in various genomics and proteomics sequence analysis tasks (<xref ref-type="bibr" rid="B50">50</xref>). The core innovations of Transformer-XL include the introduction of a segment-level recurrence mechanism and a novel relative positional encoding (<xref ref-type="bibr" rid="B50">50</xref>). These features allow the model to reuse past information and thereby extend the context window across different segments. This design enables Transformer-XL to handle longer biological sequences efficiently and provides a substantial improvement over traditional models where each segment is processed in isolation (<xref ref-type="bibr" rid="B50">50</xref>). One of the main advantages of Transformer-XL is its capability to learn dependencies that are significantly longer than those captured by traditional models, leading to improvements in both short and long sequence analysis tasks (<xref ref-type="bibr" rid="B50">50</xref>). However, the model demands more memory due to its recurrence mechanism and larger context handling, which could be a limitation in resource-constrained environments.</p>
<p>XLNet extends the Transformer-XL model using an autoregressive method (<xref ref-type="bibr" rid="B242">242</xref>). This approach allows XLNet to learn bidirectional contexts by maximizing the expected likelihood over all permutations of the input sequence order which significantly enhances its scientific language understanding capabilities (<xref ref-type="bibr" rid="B242">242</xref>). Primary innovation of XLNet is its permutation language modeling (PLM), which enables the model to predict the likelihood of a sequence by considering all permutations of the k-mers within it (<xref ref-type="bibr" rid="B156">156</xref>, <xref ref-type="bibr" rid="B242">242</xref>). This method allows XLNet to capture a comprehensive bidirectional context, unlike traditional autoregressive models only consider a single direction. In addition, XLNet incorporates a two-stream self-attention mechanism which enhances its ability to manage the context more effectively during the prediction process (<xref ref-type="bibr" rid="B156">156</xref>, <xref ref-type="bibr" rid="B242">242</xref>). One of the main advantages of XLNet is its robustness in modeling bidirectional contexts, which significantly outperforms previous models such as BERT in numerous genomics sequence analysis tasks (<xref ref-type="bibr" rid="B156">156</xref>, <xref ref-type="bibr" rid="B242">242</xref>). However, the complexity of its training process, which involves permutation of input sequences and a two-stream attention mechanism, may pose challenges in terms of computational resources and time (<xref ref-type="bibr" rid="B156">156</xref>, <xref ref-type="bibr" rid="B242">242</xref>).</p>
<p>Universal Language Model Fine-tuning (ULMFiT) has revolutionized natural language processing by introducing effective transfer learning techniques for various NLP tasks. It is developed by Jeremy Howard and Sebastian Ruder in 2018 (<xref ref-type="bibr" rid="B243">243</xref>), and it typically leverages a pre-trained language model which is fine-tuned on specific DNA sequence analysis tasks having minimal sequences (<xref ref-type="bibr" rid="B243">243</xref>). ULMFiT utilizes Average Stochastic Gradient Descent - Long Short-Term Memory (AWD-LSTM) architecture to learn the distribution and contextual relationships of k-mers in DNA sequences (<xref ref-type="bibr" rid="B57">57</xref>). It employs self-supervised learning that predicts the next k-mer based on the previous known k-mers and enables the model to capture the semantics and discriminative potential of the sequences (<xref ref-type="bibr" rid="B57">57</xref>). The core innovation of ULMFiT lies in its ability to fine-tune pre-trained language models using techniques such as discriminative fine-tuning and the slanted triangular learning rates policy. Discriminative fine-tuning considers that different layers of neural network capture different kind of information; hence, it tunes every layer with distinct learning rates (<xref ref-type="bibr" rid="B243">243</xref>), whereas slanted triangular learning rate describes a unique learning rate scheduler that initially increases the learning rate and afterward drops it in a linear fashion (<xref ref-type="bibr" rid="B243">243</xref>). The short increase stage enables the model to quickly converge to a parameter space suitable for the task, while the extended decay period allows for more effective fine-tuning (<xref ref-type="bibr" rid="B243">243</xref>). By adjusting the learning rate for different layers, it prevents catastrophic forgetting and stabilizes the training process across various tasks (<xref ref-type="bibr" rid="B243">243</xref>). ULMFiT incorporates dropout techniques to regularize learnable parameters and prevent overfitting which ensures model&#x00027;s generalization ability (<xref ref-type="bibr" rid="B57">57</xref>). Another advantage of ULMFiT is its ability to achieve high performance with significantly less data compared to traditional models. However, the complexity of fine-tuning and the need for careful calibration of learning rates can be challenging, requiring a nuanced understanding of model behavior across different layers (<xref ref-type="bibr" rid="B57">57</xref>).</p>
<p>Bidirectional Encoder Representations from Transformers (BERT) is developed by Google in 2018 (<xref ref-type="bibr" rid="B244">244</xref>). It is pretrained on a large corpus of text data, such as Wikipedia and books (<xref ref-type="bibr" rid="B244">244</xref>). It has revolutionized NLP tasks by employing a transformer-based architecture that enables the model to consider the context of k-mers from both directions simultaneously, rather than a single direction at a time (<xref ref-type="bibr" rid="B244">244</xref>). BERT is distinctive for its deep bidirectional nature, achieved through the application of the transformer model, specifically using mechanisms such as Masked Language Modeling (MLM) and Next Sentence Prediction (NSP) (<xref ref-type="bibr" rid="B244">244</xref>). This approach allows BERT to understand the context of a k-mer based on all other k-mers in a sequence, rather than just those preceding it. Specifically, it learns to capture the semantics and contextual information of the input text exceptionally well through self-supervised learning tasks such as MLP and NSP (<xref ref-type="bibr" rid="B244">244</xref>). In the case of DNA sequence analysis, BERT is used to transform DNA sequences into statistical feature space and then fine-tuned on specific downstream tasks, such as enhancer identification and strength prediction (<xref ref-type="bibr" rid="B156">156</xref>). BERT captures the semantics of DNA sequences by dynamically learning their representations through a multihead self-attention mechanism. BERT leverages transfer learning by pre-training on a large corpus and then fine-tuning on specific DNA sequence analysis task, allowing it to adapt to different applications (<xref ref-type="bibr" rid="B156">156</xref>). BERT uses MLM and NSP tasks during pre-training to learn the contextual relationships between k-mers in DNA sequences (<xref ref-type="bibr" rid="B156">156</xref>).</p>
<p>The primary advantages of BERT include its high accuracy and efficiency across various DNA sequence analysis tasks, due to its robust handling of context and bidirectional training (<xref ref-type="bibr" rid="B156">156</xref>). BERT captures both discriminative and semantical relationships of k-mers, making it effective in characterizing DNA sequences (<xref ref-type="bibr" rid="B156">156</xref>). BERT-based models have shown improved performance compared to traditional approaches in different DNA sequence analysis tasks such as enhancer identification and strength prediction (<xref ref-type="bibr" rid="B156">156</xref>). In addition, BERT can be adapted to specific application scenarios by pre-training on domain-specific custom corpora (<xref ref-type="bibr" rid="B156">156</xref>). BERT is a large model that requires significant computational resources for training and inference on extensive datasets. BERT performs best when trained on large and diverse datasets, which may not always be available for specific DNA sequence analysis tasks. In addition, while BERT provides state-of-the-art results in many scenarios, it requires fine-tuning for specific tasks, which can be resource-intensive. BERT performance can degrade with longer texts and the complex architecture of BERT makes it challenging to interpret the learned representations and understand the underlying biological mechanisms (<xref ref-type="bibr" rid="B156">156</xref>).</p>
<p>ALBERT, introduced by Google researchers, is a streamlined version of BERT designed to provide state-of-the-art results in NLP with significantly fewer parameters (<xref ref-type="bibr" rid="B245">245</xref>). This model enhances the efficiency and scalability of BERT by incorporating innovative techniques such as factorized embedding parameterization, cross-layer parameter sharing, and sentence order prediction (<xref ref-type="bibr" rid="B156">156</xref>, <xref ref-type="bibr" rid="B245">245</xref>). Factorized embedding parameterization technique reduces the size of the embedding matrix by separating the vocabulary and hidden layer sizes, which decreases the number of parameters significantly (<xref ref-type="bibr" rid="B156">156</xref>, <xref ref-type="bibr" rid="B245">245</xref>). In cross-layer parameter sharing, parameters are shared across all layers of the model, reducing the total parameter count and improving training efficiency (<xref ref-type="bibr" rid="B156">156</xref>, <xref ref-type="bibr" rid="B245">245</xref>). It replaces the next sequence prediction with sequence order prediction to enhance the model&#x00027;s ability to understand sequence coherence without requiring task prediction, making it more effective for downstream tasks (<xref ref-type="bibr" rid="B156">156</xref>, <xref ref-type="bibr" rid="B245">245</xref>). The primary advantage of ALBERT is its reduced parameter size, which allows for faster training times and less memory usage compared to BERT, without a significant loss in performance. However, the extensive parameter sharing might lead to a slight decrease in model flexibility, potentially affecting task-specific fine-tuning (<xref ref-type="bibr" rid="B156">156</xref>, <xref ref-type="bibr" rid="B245">245</xref>). Efficiently Learning an Encoder that Classifies Token Replacements Accurately (ELECTRA) (<xref ref-type="bibr" rid="B156">156</xref>, <xref ref-type="bibr" rid="B246">246</xref>) has introduced a novel pre-training method for language models. ELECTRA operates on a replaced token detection (RTD) mechanism, where it differs from traditional masked language models such as BERT (<xref ref-type="bibr" rid="B156">156</xref>, <xref ref-type="bibr" rid="B246">246</xref>). Instead of masking k-mers, ELECTRA corrupts the input by replacing tokens or k-mers with outputs from a generator model, challenging the discriminator to identify changes (<xref ref-type="bibr" rid="B156">156</xref>, <xref ref-type="bibr" rid="B246">246</xref>). This approach allows the model to learn from the entire input sequence, enhancing training efficiency. The primary advantage of ELECTRA lies in its efficiency, requiring less computational power and time to reach or exceed the benchmarks set by larger models (<xref ref-type="bibr" rid="B156">156</xref>, <xref ref-type="bibr" rid="B246">246</xref>). However, the complexity of its dual-model architecture, involving both a generator and a discriminator, might pose challenges in training stability and hyperparameter tuning.<xref ref-type="fn" rid="fn0008"><sup>8</sup></xref></p>
<p>GPT-3 is one of the most advanced AI language models developed by OpenAI (<xref ref-type="bibr" rid="B247">247</xref>). It is recognized for its ability to generate text that closely mimics human writing, making it a pivotal development in natural language processing. GPT-3 builds upon the transformer architecture, which utilizes self-attention mechanisms to process input data (<xref ref-type="bibr" rid="B247">247</xref>). Unlike GPT-2, which had 1.5 billion parameters, GPT-3 boasts a staggering 175 billion parameters. This exponential increase in parameters enhances its ability to generate coherent and contextually relevant text (<xref ref-type="bibr" rid="B247">247</xref>). GPT-3 differs from models such as BERT and XLNet by maintaining an autoregressive nature. From scientific perspective, this implies that it predicts the next k-mer in a sequence based on the previous k-mers, while BERT uses bidirectional context (<xref ref-type="bibr" rid="B240">240</xref>, <xref ref-type="bibr" rid="B247">247</xref>). One of the innovative aspects of GPT-3 is its use of alternating dense and locally banded sparse attention patterns. Dense attention considers all input k-mers simultaneously, while sparse attention focuses on a subset of k-mers, making the model more efficient and scalable. This combination enables GPT-3 to handle long-range dependencies and maintain computational efficiency (<xref ref-type="bibr" rid="B240">240</xref>, <xref ref-type="bibr" rid="B247">247</xref>). One of GPT-3&#x00027;s standout capabilities is its performance in few-shot settings. Unlike fine-tuned models that require large amounts of task-specific data, GPT-3 can perform well on new tasks with minimal sequences. This flexibility is a significant advantage over models such as BERT, which typically require extensive fine-tuning for each specific task. GPT-3 demonstrates strong performance across various tasks, often matching or exceeding that of fine-tuned models. This capability makes it a versatile tool for a wide range of applications (<xref ref-type="bibr" rid="B240">240</xref>, <xref ref-type="bibr" rid="B247">247</xref>).</p>
<p>For instance, in context of cell biology, scientific researchers have used GPT-3 to learn gene and cell embeddings effectively (<xref ref-type="bibr" rid="B240">240</xref>). Scientific researchers have utilized the text summaries of genes from the NCBI database, which contain curated information about gene functionalities and properties. The gene text summaries are passed through the GPT-3 language model, which generates gene embeddings that capture the underlying biology described in the gene summaries (<xref ref-type="bibr" rid="B240">240</xref>). The gene embeddings are averaged, weighted by the expression levels of each gene in the cell. These averaged embeddings are then normalized to a unit l2 norm to generate single-cell embeddings (<xref ref-type="bibr" rid="B240">240</xref>). In another strategy, each cell is represented by a natural language sentence constructed based on the ranked gene expressions. The gene names are ordered by descending normalized expression levels, and this sentence representation is passed through the GPT-3 model to obtain the cell embeddings (<xref ref-type="bibr" rid="B240">240</xref>). Extrinsic performance analysis of GPT-3 embeddings on tasks such as classifying gene properties or cell types has shown supreme effectiveness (<xref ref-type="bibr" rid="B240">240</xref>). While GPT-3&#x00027;s capabilities are groundbreaking, it faces challenges such as potential biases in training data and high computational demands. Moreover, its &#x0201C;black box&#x0201D; nature makes it difficult to discern how decisions are made, posing ethical and operational concerns. GPT-3&#x00027;s massive size requires significant computational resources for both training and inference. This makes it less accessible for smaller organizations or researchers without high-end hardware.</p>
</sec>
<sec>
<title>7.3 Machine and deep learning predictors</title>
<p>Machine and deep learning algorithms need statistical vectors to extract useful patterns for specific sequence analysis task. A comprehensive literature review of 127 studies reveals that 12 word embedding and 8 large language models have been used to generate statistical vectors of raw sequences to feed 28 different algorithm available within predictive pipelines of 44 DNA sequence analysis tasks. Based on working paradigms, these algorithms are categorized into 3 different categories, namely, statistical algorithms, machine learning algorithms, and deep learning algorithms. From 28 algorithms, 3 algorithms, namely, conditional random fields (CRF) (<xref ref-type="bibr" rid="B248">248</xref>), k-means clustering algorithm (<xref ref-type="bibr" rid="B21">21</xref>), and cosine similarity algorithm (<xref ref-type="bibr" rid="B173">173</xref>), belong to statistical algorithms. Machine learning algorithms involves 8 algorithms, namely, support vector machine (SVM) (<xref ref-type="bibr" rid="B352">352</xref>), Naive Bayes (NB) (<xref ref-type="bibr" rid="B95">95</xref>), multilayer perceptron (MLP) (<xref ref-type="bibr" rid="B77">77</xref>), predictive clustering tree (PCT) (<xref ref-type="bibr" rid="B128">128</xref>), random forest (RF) (<xref ref-type="bibr" rid="B103">103</xref>), deep forest (DF) (<xref ref-type="bibr" rid="B61">61</xref>), XGBoost (<xref ref-type="bibr" rid="B352">352</xref>), and CatBoost (<xref ref-type="bibr" rid="B143">143</xref>). Furthermore, deep learning algorithms include convolutional neural network (CNN) (<xref ref-type="bibr" rid="B91">91</xref>), graph neural network (GNN) (<xref ref-type="bibr" rid="B104">104</xref>), temporal convolutional network (TCN) (<xref ref-type="bibr" rid="B235">235</xref>), graph convolutional network (GCN) (<xref ref-type="bibr" rid="B55">55</xref>), graph attention network (GAT) (<xref ref-type="bibr" rid="B108">108</xref>), long short-term memory (LSTM) (<xref ref-type="bibr" rid="B97">97</xref>), bidirectional long short-term memory (BiLSTM) (<xref ref-type="bibr" rid="B58">58</xref>), and bidirectional gated recurrent unit (BiGRU) (<xref ref-type="bibr" rid="B82">82</xref>). Similarly, 8 algorithms, namely, ELECTRA (<xref ref-type="bibr" rid="B89">89</xref>), ALBERT (<xref ref-type="bibr" rid="B249">249</xref>), Transformer-XL (<xref ref-type="bibr" rid="B50">50</xref>), XL-Net (<xref ref-type="bibr" rid="B156">156</xref>), Transformer (<xref ref-type="bibr" rid="B250">250</xref>), ULMFit (<xref ref-type="bibr" rid="B146">146</xref>), GPT-3 (<xref ref-type="bibr" rid="B114">114</xref>), and BERT (<xref ref-type="bibr" rid="B251">251</xref>), belong to language modeling algorithms, five algorithms, namely, LSTM &#x0002B; CNN, CNN &#x0002B; BiLSTM, CNN &#x0002B; BiLSTM &#x0002B; BiGRU, RF &#x0002B; CNN, and CNN &#x0002B; BiGRU, belong to hybrid algorithms, whereas 1 meta-predictor reaps benefits of both machine and deep learning algorithms, namely, KNN, RF, SVM, MLP, and CNN.</p>
<p>Statistical algorithms provide a framework for understanding DNA sequence distribution and characteristics. They offer valuable advantages in terms of interpretability by facilitating researchers to assess statistical significance of genomic features. Among three statistical algorithms, Conditional Random Fields (CRF) (<xref ref-type="bibr" rid="B248">248</xref>) calculate the conditional probability of class labels of sequences by using neighboring k-mers. By capturing dependencies between adjacent labels, CRF allows for more accurate predictions of sequence features by taking into account both local sequence context and broader genomic patterns. K-means clustering algorithm (<xref ref-type="bibr" rid="B21">21</xref>) groups sequences into k distinct clusters based on similarity. It starts by initializing k centroids and assigns each sequence to the nearest cluster by calculating Euclidean distance between sequence and centroid. These centroids are updated iteratively by averaging the sequences in each cluster until they stabilize. Cosine similarity can be advantageous in DNA sequence analysis for tasks such as similarity comparison and clustering, where measuring the similarity between sequences is essential (<xref ref-type="bibr" rid="B252">252</xref>). Cosine similarity can handle high-dimensional data efficiently and is suitable for tasks requiring similarity-based analysis (<xref ref-type="bibr" rid="B252">252</xref>). These models also have limitations, including potential difficulties in managing complex and high-dimensional data. In addition, they rely on strong assumptions about underlying data distribution, which may not always align with real-world DNA sequence analysis scenarios. Despite these challenges, statistical models remain indispensable tools in DNA sequence analysis and provide valuable insights.</p>
<p>From 8 different machine learning algorithms, support vector machine (SVM) operates by finding the optimal hyperplane that best separates data points into different classes. SVMs are known for their ability to handle high-dimensional data and work well in cases where the data are not linearly separable, as they can use kernel functions to transform the data into higher dimensions where separation is possible (<xref ref-type="bibr" rid="B62">62</xref>). However, SVMs can have limitations in terms of training time, especially with large datasets, as they need to solve a complex optimization problem to find the best hyperplane that separates the classes. Naive Bayes (NB) is a probabilistic algorithm based on Bayes&#x00027; theorem with the assumption of independence between features. NB is efficient, simple to implement, and works well with high-dimensional data, making it suitable for tasks where feature independence assumptions hold (<xref ref-type="bibr" rid="B253">253</xref>). However, it may not always hold true in practice, especially in complex biological datasets where features are correlated.</p>
<p>In addition to SVM, tree-based algorithms are fundamentally built upon decision tree algorithm. Decision tree algorithm uses independent variables to construct a tree-like structure, where data are split at decision nodes into branches connected to leaf nodes to make predictions. This foundational algorithm is extended into more advanced algorithms, namely, Random Forest (RF), Deep Forest (DF), XgBoost, CatBoost, and Predictive Clustering Tree (PCT) (<xref ref-type="bibr" rid="B128">128</xref>). All of these advanced algorithms enhance basic decision tree by incorporating techniques such as ensembling, and boosting for improved accuracy and generalization. Random Forest (RF) algorithm is an ensemble learning method that constructs a multitude of decision trees during training and outputs the mode of the classes as the prediction. RF is known for its robustness to overfitting, feature importance estimation, and ability to handle high-dimensional data with ease (<xref ref-type="bibr" rid="B254">254</xref>). However, RF may not perform as well when dealing with imbalanced datasets or when there are many irrelevant features present in the data. Deep forest (DF) algorithm is another ensemble learning method that utilizes a cascade structure of multiple random forests to make predictions. DF can be advantageous in DNA sequence analysis for tasks such as clustering and species classification based on DNA barcodes (<xref ref-type="bibr" rid="B255">255</xref>). DFs are capable of learning hierarchical representations of data and can capture complex patterns in high-dimensional spaces effectively (<xref ref-type="bibr" rid="B255">255</xref>). Nonetheless, the main drawback of DF lies in its computational complexity and the need for substantial computational resources, which can limit its practicality in large-scale DNA sequence analysis projects. XGBoost combines multiple weak learners to create a strong predictive model. XGBoost can handle large datasets with high dimensionality and is known for its efficiency in boosting the performance of weak learners (<xref ref-type="bibr" rid="B256">256</xref>). However, XgBoost may require fine-tuning of hyperparameters to achieve optimal performance, and it could be sensitive to noisy data. CatBoost is another ensemble learning method designed to handle categorical features efficiently. CatBoost can automatically handle categorical features and is known for its robustness to overfitting and efficiency in training models with categorical data (<xref ref-type="bibr" rid="B256">256</xref>). Nevertheless, CatBoost&#x00027;s training time might be longer compared to other algorithms, especially when dealing with large genetic datasets.</p>
<p>Predictive clustering tree (PCT) (<xref ref-type="bibr" rid="B128">128</xref>) is a versatile predictor that integrates elements of both clustering and supervised learning. Unlike traditional decision trees, random forests, or support vector machines, PCTs are designed to handle hierarchical multi-label classification tasks, making them particularly effective for complex, high-dimensional data (<xref ref-type="bibr" rid="B128">128</xref>). PCTs operate by viewing a decision tree as a hierarchy of clusters. The root node represents a single cluster containing all training examples, which is recursively partitioned into smaller clusters as one moves down the tree. This approach allows PCTs to simultaneously perform clustering and classification, leveraging the hierarchical structure to predict multiple labels for each instance (<xref ref-type="bibr" rid="B128">128</xref>). One of the key strengths of PCTs is their ability to manage complex data with multiple interrelated labels. They can identify relevant features across different levels of the hierarchy, providing interpretable results that are valuable for domain experts. In addition, PCTs are capable of handling large datasets efficiently, making them suitable for various real-world applications (<xref ref-type="bibr" rid="B128">128</xref>). Despite their strengths, they can be computationally intensive, especially for large and deep hierarchies, and may require careful parameter tuning to avoid overfitting. In addition, while PCTs offer interpretability, the complexity of the hierarchical structure can sometimes make the results harder to interpret compared to simpler models (<xref ref-type="bibr" rid="B128">128</xref>). Apart from this, researchers have also designed customized meta-predictors which utilize the powers of five or more than five distinct algorithms, namely, kNN, RF, SVM, MLP, and CNN (<xref ref-type="bibr" rid="B105">105</xref>).</p>
<p>Multilayer perceptron (MLP) is composed of multiple layers of nodes that can learn complex patterns in data. MLPs are powerful algorithms for feature extraction and predictive modeling in DNA sequence analysis, capable of capturing intricate relationships in the data (<xref ref-type="bibr" rid="B252">252</xref>). MLPs excel in tasks requiring non-linear decision boundaries and can handle large amounts of data effectively. However, training MLPs can be computationally expensive, especially with large datasets, and they are prone to overfitting if not properly regularized.</p>
<p>Among all categories, deep learning algorithms are most extensively used for efficient DNA sequence analysis. A total of eight deep learning algorithms are most commonly used by scientific community for DNA sequence analysis. Convolutional neural network (CNN) is a deep learning algorithm designed to process structured grid-like data, such as images. In DNA sequence analysis, CNNs can be applied to DNA sequence analysis tasks to capture spatial dependencies in data. They are effective for tasks that require feature hierarchies and translation invariance (<xref ref-type="bibr" rid="B257">257</xref>). However, CNNs may struggle with capturing long-range dependencies in sequences, which can be crucial in DNA analysis where distant k-mers may interact. Graph neural network (GNN) is a type of neural network designed to operate on graph-structured data. GNNs are suitable for tasks involving relational data, such as molecular structures, making them applicable to DNA sequence analysis for tasks such as clustering (<xref ref-type="bibr" rid="B258">258</xref>). GNNs can effectively capture dependencies between nodes in a graph and are capable of learning representations that incorporate both local and global information (<xref ref-type="bibr" rid="B258">258</xref>). However, GNNs may encounter challenges in efficiently scaling to large graphs, and interpreting the learned representations in GNNs can be complex, limiting their interpretability. Temporal convolutional network (TCN) is a type of neural network designed to process sequential data efficiently. TCNs are suitable for tasks involving temporal dependencies, making them applicable to DNA sequence analysis for tasks like predicting DNA binding sites for transcription factors. TCNs can capture long-range dependencies in sequential data and are known for their parallel processing capabilities, enabling faster training times (<xref ref-type="bibr" rid="B259">259</xref>). However, TCNs may struggle with modeling complex temporal dynamics compared to recurrent models such as LSTMs. Graph convolutional network (GCN) is a type of neural network designed to operate on graph-structured data. GCNs can leverage graph structures to learn representations of nodes and edges, enabling tasks such as node classification and link prediction in DNA sequences (<xref ref-type="bibr" rid="B256">256</xref>). However, GCNs may require meticulous graph construction and preprocessing, and they can be computationally intensive, especially for large graphs, which can hinder their scalability.</p>
<p>Graph attention network (GAT) is a type of neural network that incorporates attention mechanisms to learn the importance of neighboring nodes in a graph. GATs are suitable for tasks involving relational data, as they can adaptively weigh the contributions of neighboring nodes, enabling more flexible and accurate learning on graph-structured data (<xref ref-type="bibr" rid="B260">260</xref>). However, GATs may be sensitive to noisy or sparse graphs, and designing optimal attention mechanisms can be challenging, impacting their performance in certain scenarios. Long short-term memory (LSTM) is a type of recurrent neural network designed to capture long-term dependencies in sequential data. LSTMs are effective in DNA sequence analysis for tasks such as hypersensitive DNA sequence classification (<xref ref-type="bibr" rid="B261">261</xref>). LSTMs can retain information over long sequences and are suitable for tasks requiring memory of past events, making them ideal for tasks such as classification of DNA sequences (<xref ref-type="bibr" rid="B261">261</xref>). However, LSTMs may encounter vanishing or exploding gradient problems during training, which can affect their ability to capture long-term dependencies accurately. Bidirectional long short-term memory (BiLSTM) is an extension of LSTM that processes sequences in both forward and backward directions. BiLSTMs are advantageous in DNA sequence analysis for tasks where contextual information from both past and future is essential (<xref ref-type="bibr" rid="B262">262</xref>). BiLSTMs can capture dependencies in both directions and are effective in tasks requiring bidirectional context understanding (<xref ref-type="bibr" rid="B262">262</xref>). However, BiLSTMs may be computationally intensive due to processing sequences in two directions, which can impact their training and inference speed. Bidirectional gated recurrent unit (BiGRU) is another type of recurrent neural network that combines the advantages of bidirectionality and gating mechanisms. BiGRUs can capture bidirectional dependencies efficiently and are known for their simpler architecture compared to LSTMs, making them computationally more efficient (<xref ref-type="bibr" rid="B256">256</xref>). However, BiGRUs may struggle with capturing very long-term dependencies compared to LSTMs, which can limit their effectiveness in tasks requiring extensive memory retention.</p>
<p>For different DNA sequence analysis tasks, eight contemporary language models, namely, ELECTRA, ALBERT, Transformer-XL, XLnet, Transformer, ULMFit, GPT-3, and BERT have been used in two different settings. In first setting, the addition of classification layers to these language models adapts the general-purpose language models to specific classification tasks by learning to map the rich contextual embeddings to the desired output classes. In second setting, rich contextual embeddings of these 8 language models are passed to standalone machine learning algorithms, deep learning algorithms, and ensemble or hybrid algorithms for accurate classification of DNA sequences.</p>
<p>A total of five hybrid algorithms combine different types of models to leverage the strengths of each component. LSTM &#x0002B; CNN, CNN &#x0002B; BiLSTM, CNN &#x0002B; BiLSTM &#x0002B; BiGRU, RF &#x0002B; CNN (<xref ref-type="bibr" rid="B76">76</xref>), and CNN &#x0002B; BiGRU are some of the examples of hybrid algorithms that integrate deep learning and traditional machine learning techniques to enhance predictive performance (<xref ref-type="bibr" rid="B263">263</xref>) for different DNA sequence analysis tasks. These hybrid models aim to capitalize on the complementary advantages of different algorithms to achieve superior results in various tasks.</p>
</sec>
</sec>
<sec id="s8">
<title>8 Uncovering evaluation measures for DNA sequence analysis predictive pipelines</title>
<p>AI-driven DNA sequence analysis predictive pipelines are evaluated using two different experimental settings: (1) k-fold cross-validation (<xref ref-type="bibr" rid="B48">48</xref>, <xref ref-type="bibr" rid="B78">78</xref>) and (2) Train-test split (<xref ref-type="bibr" rid="B108">108</xref>, <xref ref-type="bibr" rid="B110">110</xref>). In k-fold cross-validation, dataset is splitted into k folds, where <italic>k</italic>&#x02212;1 folds are used for training and one fold is used for testing. In next iterations, from k-folds, another fold is reserved for testing whereas remaining <italic>k</italic>&#x02212;1 folds are used for training. In this way, pipelines are trained and tested k times on whole data. This method offers more precise assessment of model generalization capability. Specifically for deep learning models (<xref ref-type="bibr" rid="B236">236</xref>), an additional set, known as validation set, is created from training set which typically uses 10% of training data. This validation set is used to optimize the model&#x00027;s hyperparameters. In train-test split experimental setting, data are divided into two distinct sets: (a) train set and (b) test set. Train set comprises majority of data (usually 70%&#x02013;80%), while test set contains remaining 20%&#x02013;30%. Similar to k-fold cross-validation, validation set is also created from train set for deep learning models.</p>
<p>Among 127 DNA sequence analysis studies, 67 studies have utilized 5-fold cross-validation-based experimental setting. Thirty five studies have used 10-fold cross-validation-based setting and 17 studies have used train test split-based setting. Eight studies have used both k-fold cross validation and train test split-based setting. Performance and effectiveness of trained predictive pipelines highly depends on ability to handles new and unseen data. To assess effectiveness and performance of predictive pipelines from different perspectives, various evaluation measures have been proposed. Based on task type, these measures are categorized into four classes: binary (<xref ref-type="bibr" rid="B92">92</xref>, <xref ref-type="bibr" rid="B153">153</xref>)/multi-class classification (<xref ref-type="bibr" rid="B27">27</xref>, <xref ref-type="bibr" rid="B173">173</xref>), multi-label classification (<xref ref-type="bibr" rid="B111">111</xref>, <xref ref-type="bibr" rid="B112">112</xref>), regression (<xref ref-type="bibr" rid="B102">102</xref>, <xref ref-type="bibr" rid="B103">103</xref>), and clustering (<xref ref-type="bibr" rid="B21">21</xref>). Following subsections summarize details of all four types of evaluation measures.</p>
<sec>
<title>8.1 Binary or multi-class classification evaluation criteria</title>
<p>Most commonly used evaluation measures in this category are accuracy (<xref ref-type="bibr" rid="B152">152</xref>, <xref ref-type="bibr" rid="B153">153</xref>, <xref ref-type="bibr" rid="B264">264</xref>), precision (<xref ref-type="bibr" rid="B137">137</xref>, <xref ref-type="bibr" rid="B152">152</xref>), recall (<xref ref-type="bibr" rid="B152">152</xref>, <xref ref-type="bibr" rid="B153">153</xref>), specificity (<xref ref-type="bibr" rid="B137">137</xref>, <xref ref-type="bibr" rid="B153">153</xref>), F1 Score (<xref ref-type="bibr" rid="B137">137</xref>, <xref ref-type="bibr" rid="B152">152</xref>), and MCC (<xref ref-type="bibr" rid="B137">137</xref>, <xref ref-type="bibr" rid="B153">153</xref>). These measures are typically calculated using confusion matrix, which consists of four entities: true positives (<italic>T</italic><sub><italic>P</italic></sub>), false positives (<italic>F</italic><sub><italic>P</italic></sub>), true negatives (<italic>T</italic><sub><italic>N</italic></sub>), and false negatives (<italic>F</italic><sub><italic>N</italic></sub>) (<xref ref-type="bibr" rid="B265">265</xref>). <xref ref-type="fig" rid="F6">Figure 6</xref> makes use of aforementioned four entities to compute distinct evaluation measures.</p>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>Overview of confusion matrix.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1503229-g0006.tif"/>
</fig>
<p>It can be seen in <xref ref-type="fig" rid="F6">Figure 6</xref> that <italic>T</italic><sub><italic>P</italic></sub> and <italic>T</italic><sub><italic>N</italic></sub> indicate correct positive and negative predictions, while <italic>F</italic><sub><italic>P</italic></sub> and <italic>F</italic><sub><italic>N</italic></sub> signify incorrect positive and negative predictions. <xref ref-type="disp-formula" rid="E10">Equation 10</xref> embodies mathematical expressions to compute aforementioned measures.</p>
<disp-formula id="E10"><label>(10)</label><mml:math id="M11"><mml:mrow><mml:mi>f</mml:mi><mml:msub><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mi>b</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>c</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:mi>A</mml:mi><mml:mi>c</mml:mi><mml:mi>c</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>y</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mo stretchy='false'>(</mml:mo><mml:mi>A</mml:mi><mml:mi>c</mml:mi><mml:mi>c</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:msub></mml:mrow></mml:mfrac></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mo stretchy='false'>(</mml:mo><mml:mi>P</mml:mi><mml:mi>R</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:msub></mml:mrow></mml:mfrac></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mo stretchy='false'>(</mml:mo><mml:mi>R</mml:mi><mml:mi>E</mml:mi><mml:mi>C</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:msub></mml:mrow></mml:mfrac></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>F</mml:mi><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:mi>S</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mo>&#x0002A;</mml:mo><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mo>&#x0002A;</mml:mo><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mo>+</mml:mo><mml:mi>R</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>S</mml:mi><mml:mi>p</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>y</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mo stretchy='false'>(</mml:mo><mml:mi>S</mml:mi><mml:mi>p</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:msub></mml:mrow></mml:mfrac></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>M</mml:mi><mml:mi>C</mml:mi><mml:mi>C</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>An in-depth assessment of existing DNA sequence analysis predictive pipelines reveals that most widely used evaluation measures for balanced datasets are F1-score, precision, accuracy, recall, specificity, and Matthews correlation coefficient (MCC). However, for imbalanced datasets, micro, macro, and weighted versions of these measures are used. To address class imbalance issue, weighted precision (<xref ref-type="bibr" rid="B266">266</xref>) considers both precision and relative weight of each class. Precision of a class is ratio of true positives to total number of positives for that class, while relative weight is proportion of samples of that class relative to total number of samples. Similarly, weighted recall (<xref ref-type="bibr" rid="B266">266</xref>) and weighted F1-score (<xref ref-type="bibr" rid="B27">27</xref>) are calculated by determining weights, recall, and F1-score for each class. Macro precision (<xref ref-type="bibr" rid="B146">146</xref>) calculates precision for each class independently and then averages these values. Macro recall (<xref ref-type="bibr" rid="B146">146</xref>) and macro F1-score (<xref ref-type="bibr" rid="B146">146</xref>) average recall and F1-score across all classes by considering each class equally regardless of size. In contrast, micro precision (<xref ref-type="bibr" rid="B146">146</xref>) calculates precision globally by considering all true positives and false positives across all classes together. Micro recall (<xref ref-type="bibr" rid="B146">146</xref>) and micro F1-score (<xref ref-type="bibr" rid="B146">146</xref>) aggregate <italic>T</italic><sub><italic>P</italic></sub>, <italic>F</italic><sub><italic>P</italic></sub>, and <italic>F</italic><sub><italic>N</italic></sub> across all classes and provides a fair and balanced evaluation of predictor performance. <xref ref-type="disp-formula" rid="E11">Equation 11</xref> provides mathematical expressions for computing these measures.</p>
<disp-formula id="E11"><label>(11)</label><mml:math id="M12"><mml:mrow><mml:mi>f</mml:mi><mml:msub><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mi>b</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>c</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:mi>M</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mi>r</mml:mi><mml:mi>o</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup><mml:mrow><mml:msubsup><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mrow><mml:mtext>&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:mstyle></mml:mrow><mml:mrow><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msubsup><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mrow><mml:mtext>&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>+</mml:mo><mml:msubsup><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mrow><mml:mtext>&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mfrac></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>M</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mi>r</mml:mi><mml:mi>o</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup><mml:mrow><mml:msubsup><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mrow><mml:mtext>&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:mstyle></mml:mrow><mml:mrow><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msubsup><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mrow><mml:mtext>&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>+</mml:mo><mml:msubsup><mml:mi>F</mml:mi><mml:mi>N</mml:mi><mml:mrow><mml:mtext>&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mfrac></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>M</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mi>r</mml:mi><mml:mi>o</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mi>F</mml:mi><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:mi>S</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup><mml:mrow><mml:mn>2.</mml:mn><mml:msubsup><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mrow><mml:mtext>&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:mstyle></mml:mrow><mml:mrow><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mn>2.</mml:mn><mml:msubsup><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mrow><mml:mtext>&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>+</mml:mo><mml:msubsup><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mrow><mml:mtext>&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>+</mml:mo><mml:msubsup><mml:mi>F</mml:mi><mml:mi>N</mml:mi><mml:mrow><mml:mtext>&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mfrac></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>r</mml:mi><mml:mi>o</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>n</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup><mml:mrow><mml:mi>P</mml:mi><mml:msup><mml:mi>r</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow></mml:mstyle></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>r</mml:mi><mml:mi>o</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>n</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow></mml:mstyle></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>r</mml:mi><mml:mi>o</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mi>F</mml:mi><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:mi>S</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>n</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup><mml:mrow><mml:mi>F</mml:mi><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:mi>S</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:msup><mml:mi>e</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow></mml:mstyle></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>W</mml:mi><mml:mi>e</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>h</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle='true'><mml:msub><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mi>n</mml:mi><mml:mi>P</mml:mi><mml:msup><mml:mi>r</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>.</mml:mo><mml:msup><mml:mi>w</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow></mml:mstyle></mml:mrow><mml:mrow><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup><mml:mrow><mml:msup><mml:mi>w</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow></mml:mstyle></mml:mrow></mml:mfrac></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>W</mml:mi><mml:mi>e</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>h</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle='true'><mml:msub><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mi>n</mml:mi><mml:msup><mml:mi>R</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>.</mml:mo><mml:msup><mml:mi>w</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow></mml:mstyle></mml:mrow><mml:mrow><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup><mml:mrow><mml:msup><mml:mi>w</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow></mml:mstyle></mml:mrow></mml:mfrac></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>W</mml:mi><mml:mi>e</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>h</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mi>F</mml:mi><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:mi>S</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup><mml:mrow><mml:mi>F</mml:mi><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:mi>s</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:msup><mml:mi>e</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>.</mml:mo><mml:msup><mml:mi>w</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow></mml:mstyle></mml:mrow><mml:mrow><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup><mml:mrow><mml:msup><mml:mi>w</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow></mml:mstyle></mml:mrow></mml:mfrac></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M13"><mml:msup><mml:mrow><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>P</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>, <inline-formula><mml:math id="M14"><mml:msup><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>P</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>, and <inline-formula><mml:math id="M15"><mml:msup><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> refer to true positives, false positive, and false negatives in class i, respectively. <italic>Pr</italic><sup><italic>i</italic></sup>, <italic>R</italic><sup><italic>i</italic></sup>, and <italic>F</italic>1&#x02212;<italic>score</italic><sup><italic>i</italic></sup> signify precision, recall, and F1-score of class <italic>i</italic>. <italic>w</italic><sub><italic>i</italic></sub> indicates relative weight of class i, and <italic>i</italic> refers to <italic>i</italic><sup><italic>th</italic></sup> class among <italic>n</italic> classes.</p>
</sec>
<sec>
<title>8.2 Multi-label classification evaluation measures</title>
<p>Performance evaluation of multi-label classification predictive pipelines is more challenging compared to binary and multi-class classification predictive pipelines (<xref ref-type="bibr" rid="B267">267</xref>). In binary or multi-class classification, each sample is assigned to only one class at a time, so predicted class label will be either true or false. Contrarily, in multi-label classification, a sample belongs to two or more labels simultaneously and predictive pipelines predicts multiple labels (<xref ref-type="bibr" rid="B267">267</xref>). Among predicted labels, some labels can be correct, some labels can be incorrect, or all predicted labels can be correct or incorrect. This partial correctness introduces complexity. To address this problem, researchers have developed various evaluation measures, namely, accuracy (<xref ref-type="bibr" rid="B268">268</xref>), precision (<xref ref-type="bibr" rid="B268">268</xref>), recall (<xref ref-type="bibr" rid="B268">268</xref>), and hamming loss (<xref ref-type="bibr" rid="B269">269</xref>).</p>
<p><xref ref-type="disp-formula" rid="E12">Equation 12</xref> illustrates the mathematical expressions for these evaluation measures.</p>
<disp-formula id="E12"><label>(12)</label><mml:math id="M16"><mml:mrow><mml:mi>f</mml:mi><mml:msub><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>u</mml:mi><mml:mi>l</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>b</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo> <mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:mi>A</mml:mi><mml:mi>c</mml:mi><mml:mi>c</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>y</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mo stretchy='false'>(</mml:mo><mml:mi>A</mml:mi><mml:mi>c</mml:mi><mml:mi>c</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:msubsup><mml:mrow><mml:mrow><mml:mo>|</mml:mo> <mml:mrow><mml:mfrac><mml:mrow><mml:msup><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>&#x02227;</mml:mo><mml:msup><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow><mml:mrow><mml:msup><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>&#x02228;</mml:mo><mml:msup><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow></mml:mfrac></mml:mrow> <mml:mo>|</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mo stretchy='false'>(</mml:mo><mml:mi>R</mml:mi><mml:mi>E</mml:mi><mml:mi>C</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:msubsup><mml:mrow><mml:mfrac><mml:mrow><mml:mrow><mml:mo>|</mml:mo> <mml:mrow><mml:msup><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>&#x02227;</mml:mo><mml:msup><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow> <mml:mo>|</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mo>|</mml:mo> <mml:mrow><mml:msup><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow> <mml:mo>|</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mo stretchy='false'>(</mml:mo><mml:mi>P</mml:mi><mml:mi>R</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:msubsup><mml:mrow><mml:mfrac><mml:mrow><mml:mrow><mml:mo>|</mml:mo> <mml:mrow><mml:msup><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>&#x02227;</mml:mo><mml:msup><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow> <mml:mo>|</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mo>|</mml:mo> <mml:mrow><mml:msup><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow> <mml:mo>|</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>F</mml:mi><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:mi>S</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:msubsup><mml:mrow><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mo>&#x0002A;</mml:mo><mml:mrow><mml:mo>|</mml:mo> <mml:mrow><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x0002A;</mml:mo><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo stretchy='false'>)</mml:mo></mml:mrow> <mml:mo>|</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mo>|</mml:mo> <mml:mrow><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mo>+</mml:mo><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo stretchy='false'>)</mml:mo></mml:mrow> <mml:mo>|</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>H</mml:mi><mml:mi>a</mml:mi><mml:mi>m</mml:mi><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>g</mml:mi><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mi>N</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:msubsup><mml:mrow><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>l</mml:mi></mml:msubsup><mml:mrow><mml:mrow><mml:mo>[</mml:mo> <mml:mrow><mml:mo>&#x0007C;</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:msubsup><mml:mi>A</mml:mi><mml:mi>j</mml:mi><mml:mrow><mml:mtext>&#x000A0;&#x000A0;</mml:mtext><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02260;</mml:mo><mml:msubsup><mml:mi>P</mml:mi><mml:mi>j</mml:mi><mml:mrow><mml:mtext>&#x000A0;&#x000A0;</mml:mtext><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x0007C;</mml:mo></mml:mrow> <mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow> </mml:mrow></mml:mrow></mml:math></disp-formula>
<p>In these equations, <italic>N</italic> represents total number of samples, <italic>n</italic><sup><italic>i</italic></sup> denotes <italic>i</italic><sup><italic>th</italic></sup> sample out of <italic>N</italic> samples, <italic>A</italic><sup><italic>i</italic></sup> is actual class label, and <italic>P</italic><sup><italic>i</italic></sup> is the predicted label for <italic>n</italic><sup><italic>i</italic></sup> sample. <italic>l</italic> represents sample length, <italic>j</italic> denotes class index, &#x02228; signifies logical OR operator, and &#x02227; represents logical AND operator. Similar to evaluation measures in binary or multi-class classification and includes micro precision, micro recall, micro F1-score, macro precision, macro recall, macro F1-score, weighted precision, weighted recall, and weighted F1-score. A rigorous analysis of existing literature on multi-label classification tasks in DNA sequence analysis reveals that most widely used evaluation metrics are accuracy, precision, recall, F1-score, MCC, sensitivity, and specificity.</p>
</sec>
<sec>
<title>8.3 Regression evaluation criteria</title>
<p>Regression tasks differ fundamentally from classification tasks where model predicts continuous numerical values rather than discrete class labels. Regression task-related predictive pipelines are evaluated using distinct evaluation measures including mean squared error (MSE) (<xref ref-type="bibr" rid="B270">270</xref>), mean absolute error (MAE) (<xref ref-type="bibr" rid="B271">271</xref>), mean bias error (MBE) (<xref ref-type="bibr" rid="B272">272</xref>), mean absolute percentage error (MAPE) (<xref ref-type="bibr" rid="B273">273</xref>), root mean square error (RMSE) (<xref ref-type="bibr" rid="B271">271</xref>), <italic>R</italic><sup>2</sup> (<xref ref-type="bibr" rid="B274">274</xref>), relative mean absolute error (rMAE) (<xref ref-type="bibr" rid="B275">275</xref>), relative mean square error (rMSE) (<xref ref-type="bibr" rid="B275">275</xref>), relative root mean square error (rRMSE) (<xref ref-type="bibr" rid="B275">275</xref>), and relative mean bias error (rMBE) (<xref ref-type="bibr" rid="B275">275</xref>).</p>
<p>MAE assesses predictor performance by measuring absolute difference between predicted and actual values (<xref ref-type="bibr" rid="B271">271</xref>). MSE quantifies deviation by averaging squared differences between actual and predicted values (<xref ref-type="bibr" rid="B270">270</xref>). Similarly, RMSE calculates standard deviation of prediction errors and demonstrates how tightly data points cluster around regression line (<xref ref-type="bibr" rid="B271">271</xref>). MBE assesses predictor performance in terms of under and overfitting by enumerating average difference between predicted and actual value (<xref ref-type="bibr" rid="B272">272</xref>). MAPE calculates percentage variation between predicted and actual values (<xref ref-type="bibr" rid="B273">273</xref>). Smaller the values of MAE, MBE, MSE, and MAPE, better will be predictor performance. Higher value of <italic>R</italic><sup>2</sup> score signifies promising predictor performance as it measures proportion of variance in predicted dependent variable explained by independent variable to determine strength of relationship.</p>
<p>MAE, MSE, RMSE, and MAPE measures compute average error value for N number of data points. Relative performance evaluation can improve quality of performance evaluation by reducing the noise from data. For relative performance evaluation, the percentage error of each metric is computed relative to the average of actual values (<xref ref-type="bibr" rid="B275">275</xref>). It facilitates in controlling factors that influence predictor performance by relatively calculating ratio of particular error with average of actual values (<xref ref-type="bibr" rid="B275">275</xref>). Since data continuously vary and produce varying predicted values at different time intervals, an overall percentage error is computed to obtain relative error of all data points. <xref ref-type="disp-formula" rid="E13">Equation 13</xref> embodies mathematical expressions for aforementioned evaluation metrics.</p>
<disp-formula id="E13"><label>(13)</label><mml:math id="M17"><mml:mrow><mml:mi>f</mml:mi><mml:msub><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>g</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>M</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>M</mml:mi></mml:msubsup><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:msup><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>&#x02212;</mml:mo><mml:msup><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>M</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>M</mml:mi></mml:msubsup><mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msup><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>&#x02212;</mml:mo><mml:msup><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mstyle></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>R</mml:mi><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:msqrt><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mi>M</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>M</mml:mi></mml:msubsup><mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msup><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>&#x02212;</mml:mo><mml:msup><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mstyle></mml:mrow></mml:msqrt></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>M</mml:mi><mml:mi>B</mml:mi><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>M</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>M</mml:mi></mml:msubsup><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msup><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>&#x02212;</mml:mo><mml:msup><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:mi>P</mml:mi><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>m</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>M</mml:mi></mml:msubsup><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msup><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msup><mml:mo>&#x02212;</mml:mo><mml:msup><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow><mml:mrow><mml:msup><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msup></mml:mrow></mml:mfrac></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:mrow></mml:mstyle><mml:mo>&#x000D7;</mml:mo><mml:mn>100</mml:mn></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:msup><mml:mi>R</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mi>S</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>M</mml:mi></mml:msubsup><mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>P</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mi>A</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mstyle></mml:mrow><mml:mrow><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>M</mml:mi></mml:msubsup><mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>A</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mi>a</mml:mi><mml:mi>v</mml:mi><mml:mi>g</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>A</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mstyle></mml:mrow></mml:mfrac></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>r</mml:mi><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:mi>E</mml:mi></mml:mrow><mml:mover accent='true'><mml:mi>A</mml:mi><mml:mo>&#x000AF;</mml:mo></mml:mover></mml:mfrac><mml:mo>&#x000D7;</mml:mo><mml:mn>100</mml:mn></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>r</mml:mi><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:mi>E</mml:mi></mml:mrow><mml:mover accent='true'><mml:mi>A</mml:mi><mml:mo>&#x000AF;</mml:mo></mml:mover></mml:mfrac><mml:mo>&#x000D7;</mml:mo><mml:mn>100</mml:mn></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>r</mml:mi><mml:mi>M</mml:mi><mml:mi>B</mml:mi><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>M</mml:mi><mml:mi>B</mml:mi><mml:mi>E</mml:mi></mml:mrow><mml:mover accent='true'><mml:mi>A</mml:mi><mml:mo>&#x000AF;</mml:mo></mml:mover></mml:mfrac><mml:mo>&#x000D7;</mml:mo><mml:mn>100</mml:mn></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>r</mml:mi><mml:mi>R</mml:mi><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>R</mml:mi><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:mi>E</mml:mi></mml:mrow><mml:mover accent='true'><mml:mi>A</mml:mi><mml:mo>&#x000AF;</mml:mo></mml:mover></mml:mfrac><mml:mo>&#x000D7;</mml:mo><mml:mn>100</mml:mn></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>In <xref ref-type="disp-formula" rid="E13">Equation 13</xref>, <italic>M</italic> denotes total number of samples, <italic>A</italic><sup><italic>i</italic></sup> represents actual value, and <italic>P</italic><sup><italic>i</italic></sup> is predicted value where i denotes the sample number and &#x00100; is the average of total actual values.</p>
</sec>
<sec>
<title>8.4 Clustering evaluation measures</title>
<p>In contrast to first three categories explained, clustering tasks aim to group similar samples based on their features without predefined class labels. In this task, prime objective is to use clustering algorithms and identify inherent patterns or structures within data. In these tasks, clusters of data samples with similar features are created, and predictors assign new data points to appropriate clusters (<xref ref-type="bibr" rid="B276">276</xref>). A higher similarity to a cluster indicates that data sample belongs to that cluster (<xref ref-type="bibr" rid="B276">276</xref>). To assess clustering predictive pipeline performance, researchers have introduced various evaluation measures including accuracy (<xref ref-type="bibr" rid="B277">277</xref>), normalized mutual information (NMI) (<xref ref-type="bibr" rid="B277">277</xref>), silhouette score (SS) (<xref ref-type="bibr" rid="B278">278</xref>), dunn index (DI) (<xref ref-type="bibr" rid="B279">279</xref>), and Davies-Bouldin index (DBI) (<xref ref-type="bibr" rid="B280">280</xref>).</p>
<p>Accuracy (<xref ref-type="bibr" rid="B277">277</xref>) is the proportion of correctly predicted samples to total number of samples. NMI (<xref ref-type="bibr" rid="B277">277</xref>) quantifies quality of predictor by measuring mutual information between predicted clusters and actual clusters. Mutual information refers to computed joint probability between predicted clusters and actual clusters. Silhouette score (<xref ref-type="bibr" rid="B278">278</xref>) measures how similar data samples are within a cluster compared to other clusters. BDI (<xref ref-type="bibr" rid="B280">280</xref>) evaluates average similarity ratio of each cluster with its most similar cluster. DI (<xref ref-type="bibr" rid="B279">279</xref>) computes ratio of minimum inter-cluster distance to maximum intra-cluster distance. <xref ref-type="disp-formula" rid="E14">Equation 14</xref> embodies mathematical expressions for these evaluations measures.</p>
<disp-formula id="E14"><label>(14)</label><mml:math id="M18"><mml:mrow><mml:mi>f</mml:mi><mml:msub><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>l</mml:mi><mml:mi>u</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:mi>A</mml:mi><mml:mi>c</mml:mi><mml:mi>c</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>y</mml:mi><mml:mo>=</mml:mo><mml:munder><mml:mi>m</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:munder><mml:mfrac><mml:mrow><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup><mml:mn>1</mml:mn></mml:mstyle><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mi>m</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow><mml:mi>n</mml:mi></mml:mfrac></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>N</mml:mi><mml:mi>M</mml:mi><mml:mi>I</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>I</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mn>2</mml:mn></mml:mfrac><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>H</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>+</mml:mo><mml:mi>H</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>S</mml:mi><mml:mi>S</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>d</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mi>a</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>d</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mi>a</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>D</mml:mi><mml:mi>B</mml:mi><mml:mi>I</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>n</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup><mml:mrow><mml:munder><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>&#x02260;</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:munder></mml:mrow></mml:mstyle><mml:mo stretchy='false'>(</mml:mo><mml:mfrac><mml:mrow><mml:mover accent='true'><mml:mrow><mml:msub><mml:mi>S</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo stretchy='true'>&#x000AF;</mml:mo></mml:mover><mml:mo>+</mml:mo><mml:mover accent='true'><mml:mrow><mml:msub><mml:mi>S</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow><mml:mo stretchy='true'>&#x000AF;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mfrac><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>D</mml:mi><mml:mi>I</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x02264;</mml:mo><mml:mi>i</mml:mi><mml:mo>&#x0003C;</mml:mo><mml:mi>j</mml:mi><mml:mo>&#x02264;</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mi>d</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>c</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x02264;</mml:mo><mml:mi>k</mml:mi><mml:mo>&#x02264;</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mi>d</mml:mi><mml:mo>&#x02032;</mml:mo></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:mi>c</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mfrac></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>In <xref ref-type="disp-formula" rid="E14">Equation 14</xref>, <italic>y</italic><sub><italic>i</italic></sub> refers to predicted cluster, <italic>c</italic><sub><italic>i</italic></sub> and <italic>c</italic><sub><italic>j</italic></sub> indicate <italic>i</italic><sup><italic>th</italic></sup> and <italic>j</italic><sup><italic>th</italic></sup> clusters among <italic>n</italic> clusters. Moreover, <italic>I</italic>(<italic>y</italic><sub><italic>i</italic></sub>, <italic>c</italic><sub><italic>i</italic></sub>) signifies mutual information, <italic>H</italic>(<italic>y</italic><sub><italic>i</italic></sub>) and <italic>H</italic>(<italic>c</italic><sub><italic>i</italic></sub>) show entropy of predicted and actual clusters. <italic>d</italic>(<italic>y</italic><sub><italic>i</italic></sub>) is the average distance from <italic>y</italic><sub><italic>i</italic></sub> to all points in other clusters, and <italic>a</italic>(<italic>y</italic><sub><italic>i</italic></sub>) is the average distance of <italic>y</italic><sub><italic>i</italic></sub> to all points in that clusters. <italic>d</italic>(<italic>c</italic><sub><italic>i</italic></sub>, <italic>c</italic><sub><italic>j</italic></sub>) represents inter-cluster distance between cluster i and cluster j, <inline-formula><mml:math id="M19"><mml:mover accent="true"><mml:mrow><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover></mml:math></inline-formula> represents mean distance from cluster mean for all observations in cluster i, while <inline-formula><mml:math id="M20"><mml:mover accent="true"><mml:mrow><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover></mml:math></inline-formula> denotes mean distance from cluster median for all observations in cluster j. An extensive analysis of existing literature reveals that most commonly used evaluation measures are accuracy and normalized mutual information.</p>
</sec>
</sec>
<sec id="s9">
<title>9 Open-source DNA sequence analysis predictive models</title>
<p>The public availability of predictor source codes, pretrained language models, and word embeddings significantly benefits researchers by preventing the need to reinvent the wheel. These resources enable researchers to build on existing work, utilizing pre-trained models and complete predictive pipelines to develop new, enhanced applications. By integrating new strategies into these established pipelines, they can create more powerful predictors. In addition, open-source access to these codes allows for the reproduction of predictor performance, fostering transparency and reliability in research. To expedite the establishment of more precise, robust, reliable, and efficient AI models for DNA sequence analysis and ultimately accelerate advancements in genomics and bioinformatics research, this section provides a summary of open source predictive pipelines developed using two representation learning approaches: word embeddings and large language models for 44 different DNA sequence analysis tasks.</p>
<p>Our analysis reveals that, out of 39 existing word embedding based DNA sequence analysis studies, only <bold>25</bold> studies have made the source codes of their predictive pipelines publicly accessible. In addition, source code of only 38 studies is publicly available out of 67 existing DNA sequence analysis studies based on large language models. <xref ref-type="table" rid="T4">Tables 4</xref>, <xref ref-type="table" rid="T5">5</xref> offer details on open-source codes for DNA sequence analysis predictive pipelines based on word embeddings and large language models, respectively. They also provide a summary of the representation learning methods and machine/deep learning predictors employed, along with links to the corresponding source codes.</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Summary of open-source word embedding based models in existing studies.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>References</bold></th>
<th valign="top" align="left"><bold>Task</bold></th>
<th valign="top" align="left"><bold>Embedding approach</bold></th>
<th valign="top" align="left"><bold>Classifier</bold></th>
<th valign="top" align="left"><bold>Source Code</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Ratajczak et al. (<xref ref-type="bibr" rid="B340">340</xref>)</td>
<td valign="top" align="left">Disease genes identification</td>
<td valign="top" align="left">Node2Vec</td>
<td valign="top" align="left">MLP</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/fratajcz/speos">https://github.com/fratajcz/speos</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Pan et al. (<xref ref-type="bibr" rid="B177">177</xref>)</td>
<td valign="top" align="left">Phage-host interactions prediction</td>
<td valign="top" align="left">SDNE, Word2Vec</td>
<td valign="top" align="left">MLP</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/NWUJiePan/Code">https://github.com/NWUJiePan/Code</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Chen et al. (<xref ref-type="bibr" rid="B341">341</xref>)</td>
<td valign="top" align="left">Chromatin accessibility prediction</td>
<td valign="top" align="left">Graph2Vec</td>
<td valign="top" align="left">NA</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/pinellolab/simba">https://github.com/pinellolab/simba</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Han et al. (<xref ref-type="bibr" rid="B342">342</xref>)</td>
<td valign="top" align="left">Nucleosome position detection</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">CNN &#x0002B; BiLSTM &#x0002B; BiGRU</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/lliqi-echo/Nucleosome-positioning-based-on-DNA-sequence-word-vector-and-deep-learning">https://github.com/lliqi-echo/Nucleosome-positioning-based-on</ext-link></td>
</tr>
<tr>
<td/>
<td/>
<td/>
<td/>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/lliqi-echo/Nucleosome-positioning-based-on-DNA-sequence-word-vector-and-deep-learning">-DNA-sequence-word-vector-and-deep-learning</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Liao et al. (<xref ref-type="bibr" rid="B58">58</xref>)</td>
<td valign="top" align="left">Enhancer identification</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">CNN &#x0002B; BiLSTM</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/WamesM/iEnhancer-DCLA">https://github.com/WamesM/iEnhancer-DCLA</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Inayat et al. (<xref ref-type="bibr" rid="B59">59</xref>)</td>
<td valign="top" align="left">Enhancer identification</td>
<td valign="top" align="left">FastText</td>
<td valign="top" align="left">MLP</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/salman-khan-mrd/IEnhancer-DFH">https://github.com/salman-khan-mrd/IEnhancer-DFH</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Zhang et al. (<xref ref-type="bibr" rid="B76">76</xref>)</td>
<td valign="top" align="left">Promoter identification</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">RF &#x0002B; CNN</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/HaoWuLab-Bioinformatics/iPro-WAEL">https://github.com/HaoWuLab-Bioinformatics/iPro-WAEL</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Le et al. (<xref ref-type="bibr" rid="B78">78</xref>)</td>
<td valign="top" align="left">Promoter identification</td>
<td valign="top" align="left">FastText</td>
<td valign="top" align="left">CNN</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/khanhlee/deepPromoter">https://github.com/khanhlee/deepPromoter</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Hong et al. (<xref ref-type="bibr" rid="B82">82</xref>)</td>
<td valign="top" align="left">Enhancer-promoter interactions prediction</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">CNN &#x0002B; BiGRU</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/hzy95/EPIVAN">https://github.com/hzy95/EPIVAN</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Min et al. (<xref ref-type="bibr" rid="B49">49</xref>)</td>
<td valign="top" align="left">Enhancer-promoter interactions prediction</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">CNN &#x0002B; BiGRU &#x0002B; Matching Heauristic</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/Xzenglab/EPI-DLMH">https://github.com/Xzenglab/EPI-DLMH</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Dao et al. (<xref ref-type="bibr" rid="B45">45</xref>)</td>
<td valign="top" align="left">YY1-mediated chromatin loops prediction</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">CNN</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="http://lin-group.cn/server/DeepYY1">http://lin-group.cn/server/DeepYY1</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Tran et al. (<xref ref-type="bibr" rid="B153">153</xref>)</td>
<td valign="top" align="left">Methylcytosine sites prediction</td>
<td valign="top" align="left">FastText</td>
<td valign="top" align="left">XGBoost</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/khucnam/5mC_Pred">https://github.com/khucnam/5mC_Pred</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Zulfiqar et al. (<xref ref-type="bibr" rid="B136">136</xref>)</td>
<td valign="top" align="left">Methylcytosine sites prediction</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">CNN</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/linDing-groups/Deep-4mCW2V">https://github.com/linDing-groups/Deep-4mCW2V</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Fang et al. (<xref ref-type="bibr" rid="B137">137</xref>)</td>
<td valign="top" align="left">Methylcytosine sites prediction</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">CNN</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/mat310/W2VC">https://github.com/mat310/W2VC</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Khanal et al. (<xref ref-type="bibr" rid="B138">138</xref>)</td>
<td valign="top" align="left">Methylcytosine sites prediction</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">CNN</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="http://nsclbio.jbnu.ac.kr/tools/4mC-w2vec/">http://nsclbio.jbnu.ac.kr/tools/4mC-w2vec/</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Huang et al. (<xref ref-type="bibr" rid="B151">151</xref>)</td>
<td valign="top" align="left">Methyladenine sites prediction</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">BiLSTM</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="http://39.100.246.211:5004/6mA_Pred/">http://39.100.246.211:5004/6mA_Pred/</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Le et al. (<xref ref-type="bibr" rid="B105">105</xref>)</td>
<td valign="top" align="left">Essential genes identification</td>
<td valign="top" align="left">FastText</td>
<td valign="top" align="left">Ensemble (kNN &#x0002B; RF &#x0002B; SVM &#x0002B; MLP &#x0002B; CNN)</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/khanhlee/eDNN-EG">https://github.com/khanhlee/eDNN-EG</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Zhang et al. (<xref ref-type="bibr" rid="B106">106</xref>)</td>
<td valign="top" align="left">Essential genes identification</td>
<td valign="top" align="left">Node2Vec</td>
<td valign="top" align="left">MLP</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/xzhang2016/DeepHE">https://github.com/xzhang2016/DeepHE</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Nunes et al. (<xref ref-type="bibr" rid="B110">110</xref>)</td>
<td valign="top" align="left">Disease genes prediction</td>
<td valign="top" align="left">OPA2Vec</td>
<td valign="top" align="left">RF</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/liseda-lab/KGE_Predictions_GD">https://github.com/liseda-lab/KGE_Predictions_GD</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Fan et al. (<xref ref-type="bibr" rid="B111">111</xref>)</td>
<td valign="top" align="left">Pseudogene function prediction</td>
<td valign="top" align="left">Node2Vec</td>
<td valign="top" align="left">GCN</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/yanzhanglab/Pseudo2GO">https://github.com/yanzhanglab/Pseudo2GO</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Yilmaz et al. (<xref ref-type="bibr" rid="B173">173</xref>)</td>
<td valign="top" align="left">Mutation susceptibility analysis</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">Cosine similarity</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/alperyilmaz/dna2vec_snp">https://github.com/alperyilmaz/dna2vec_snp</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Arango et al. (<xref ref-type="bibr" rid="B113">113</xref>)</td>
<td valign="top" align="left">Target gene classification</td>
<td valign="top" align="left">FastText</td>
<td valign="top" align="left">MLP</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://bitbucket.org/gaarangoa/metamlp/src/master">https://bitbucket.org/gaarangoa/metamlp/src/master</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Shi et al. (<xref ref-type="bibr" rid="B122">122</xref>)</td>
<td valign="top" align="left">Gene taxonomy classification</td>
<td valign="top" align="left">LSH &#x0002B; FastText</td>
<td valign="top" align="left">MLP</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/Lizhen0909/LSHVec">https://github.com/Lizhen0909/LSHVec</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Li et al. (<xref ref-type="bibr" rid="B23">23</xref>)</td>
<td valign="top" align="left">DNA-binding proteins binding sites identification</td>
<td valign="top" align="left">Word2Vec, FastText, GloVe</td>
<td valign="top" align="left">CNN, DCNN, CNN &#x0002B; BiLSTM, DCNN &#x0002B; BiLSTM</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="http://bliulab.net/BioSeq-BLM/">http://bliulab.net/BioSeq-BLM/</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Do et al. (<xref ref-type="bibr" rid="B165">165</xref>)</td>
<td valign="top" align="left">Recombination spots identification</td>
<td valign="top" align="left">FastText</td>
<td valign="top" align="left">SVM</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/khanhlee/fastspot">https://github.com/khanhlee/fastspot</ext-link></td>
</tr></tbody>
</table>
</table-wrap>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>Summary of open-source language model-based models in existing studies.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>References</bold></th>
<th valign="top" align="left"><bold>Task name</bold></th>
<th valign="top" align="left"><bold>Language model</bold></th>
<th valign="top" align="left"><bold>Classifier</bold></th>
<th valign="top" align="left"><bold>Pre-train/Self-train</bold></th>
<th valign="top" align="left"><bold>Source code</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Zhang et al. (<xref ref-type="bibr" rid="B31">31</xref>)</td>
<td valign="top" align="left">Chromatin accessibility prediction</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">Self-train</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/ykzhang0126/semanticCAP">https://github.com/ykzhang0126/semantic-CAP</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Nguyen et al. (<xref ref-type="bibr" rid="B44">44</xref>)</td>
<td valign="top" align="left">Species classification, chromatin accessibility prediction</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">Self-train</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/HazyResearch/hyena-dna">https://github.com/HazyResearch/hyena-dna</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Luo et al. (<xref ref-type="bibr" rid="B96">96</xref>)</td>
<td valign="top" align="left">Protein-DNA binding sites prediction</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">Self-train</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/lhy0322/TFBert">https://github.com/lhy0322/TFBert</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Liu et al. (<xref ref-type="bibr" rid="B53">53</xref>)</td>
<td valign="top" align="left">Protein-DNA binding sites prediction</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">CNN</td>
<td valign="top" align="left">Pretrained</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/YAndrewL/clape">https://github.com/YAndrewL/clape</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Gao et al. (<xref ref-type="bibr" rid="B36">36</xref>)</td>
<td valign="top" align="left">Long-range chromatin interaction prediction, prediction of context-specific functional impact of genetic variants</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">Self-train</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/ZjGaothu/EpiGePT">https://github.com/ZjGaothu/EpiGePT</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Ni et al. (<xref ref-type="bibr" rid="B83">83</xref>)</td>
<td valign="top" align="left">Enhancer-promoter interaction prediction</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">Self-train</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/NWAFUniyu/EPI-Mind">https://github.com/NWAFUniyu/EPI-Mind</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Reddy et al. (<xref ref-type="bibr" rid="B102">102</xref>)</td>
<td valign="top" align="left">Gene expression prediction</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">CNN</td>
<td valign="top" align="left">Self-train</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/anikethjr/promoter_models">https://github.com/anikethjr/promoter-models</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Fishman et al. (<xref ref-type="bibr" rid="B32">32</xref>)</td>
<td valign="top" align="left">Gene expression prediction</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">Self-train</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/AIRI-Institute/GENA_LM">https://github.com/AIRI-Institute/GENA_LM</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Osseni et al. (<xref ref-type="bibr" rid="B180">180</xref>)</td>
<td valign="top" align="left">Tumor type prediction</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">Self-train</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/dizam92/multiomic_predictions">https://github.com/dizam92/multiomic-predictions</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Le et al. (<xref ref-type="bibr" rid="B281">281</xref>)</td>
<td valign="top" align="left">Methyladenine sites prediction</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">CNN</td>
<td valign="top" align="left">Self-train</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/khanhlee/bert-dna">https://github.com/khanhlee/bert-dna</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Tsukiyama et al. (<xref ref-type="bibr" rid="B144">144</xref>)</td>
<td valign="top" align="left">Methyladenine sites prediction</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">CNN &#x0002B; BiLSTM</td>
<td valign="top" align="left">Self-train</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/kuratahiroyuki/BERT6mA.git">https://github.com/kuratahiroyuki/BERT-6mA.git</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Wang et al. (<xref ref-type="bibr" rid="B158">158</xref>)</td>
<td valign="top" align="left">Methylation sites prediction</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">Self-train</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/sb111169/tf-5mc">https://github.com/sb111169/tf-5mc</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Zhou et al. (<xref ref-type="bibr" rid="B250">250</xref>)</td>
<td valign="top" align="left">Methylation sites prediction</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">Self-train</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/LieberInstitute/INTERACT">https://github.com/LieberInstitute/INTERACT</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Jeong et al. (<xref ref-type="bibr" rid="B159">159</xref>)</td>
<td valign="top" align="left">Methylation sites prediction</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">Self-train</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/CompEpigen/methylseq_simulation">https://github.com/CompEpigen/methyl-seq_simulation</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Huson et al. (<xref ref-type="bibr" rid="B248">248</xref>)</td>
<td valign="top" align="left">Methylation sites prediction</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">CRF</td>
<td valign="top" align="left">Self-train</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/husonlab/MR-DNA">https://github.com/husonlab/MR-DNA</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Yu et al. (<xref ref-type="bibr" rid="B154">154</xref>)</td>
<td valign="top" align="left">Methylation sites prediction</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">Pretrained</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/YUYING07/iDNA_ABT">https://github.com/YUYING07/iDNA_ABT</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Zhou et al. (<xref ref-type="bibr" rid="B155">155</xref>)</td>
<td valign="top" align="left">Methylation sites prediction</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">Pretrained</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/wrab12/StableDNAm">https://github.com/wrab12/StableDNAm</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Jin et al. (<xref ref-type="bibr" rid="B157">157</xref>)</td>
<td valign="top" align="left">Methylation sites prediction</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">FGM</td>
<td valign="top" align="left">Pretrained</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/FakeEnd/iDNA_ABF">https://github.com/FakeEnd/iDNA_ABF</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Stanojevic et al. (<xref ref-type="bibr" rid="B152">152</xref>)</td>
<td valign="top" align="left">Methylcytosine sites prediction</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">Pretrained</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/lbcb-sci/rockfish">https://github.com/lbcb-sci/rockfish</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Wang et al. (<xref ref-type="bibr" rid="B282">282</xref>)</td>
<td valign="top" align="left">Methylcytosine site prediction</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">Self-train</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://zenodo.org/records/10143217">https://zenodo.org/records/10143217</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Yang et al. (<xref ref-type="bibr" rid="B143">143</xref>)</td>
<td valign="top" align="left">Methylcytosine sites prediction</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">CatBoost</td>
<td valign="top" align="left">Pretrained</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/abcair/4mCBERT">https://github.com/abcair/4mCBERT</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Yang et al. (<xref ref-type="bibr" rid="B249">249</xref>)</td>
<td valign="top" align="left">Conserved non-coding element classification</td>
<td valign="top" align="left">Transformer, ALBERT</td>
<td valign="top" align="left">Transformer, ALBERT</td>
<td valign="top" align="left">Self-train</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/melobio/LOGO">https://github.com/melobio/LOGO</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Fazeel et al. (<xref ref-type="bibr" rid="B29">29</xref>)</td>
<td valign="top" align="left">Nucleosome positioning prediction</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">LSTM</td>
<td valign="top" align="left">Self-train</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/FAhtisham/Nucleosome-position-prediction">https://github.com/FAhtisham/Nucleosome-position-prediction</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Dalla et al. (<xref ref-type="bibr" rid="B100">100</xref>)</td>
<td valign="top" align="left">Promoter identification, enhancers identification, splice site identification, chromatin accessibility prediction</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">Self-train</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/instadeepai/nucleotide-transformer">https://github.com/instadeepai/nucleotide-transformer</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Ji et al. (<xref ref-type="bibr" rid="B93">93</xref>)</td>
<td valign="top" align="left">Promoter prediction, splice sites prediction, transcription factor binding sites prediction</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">Self-train</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/jerryji1993/DNABERT">https://github.com/jerryji1993/DNABERT</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Xu et al. (<xref ref-type="bibr" rid="B52">52</xref>)</td>
<td valign="top" align="left">Transcription factor binding affinity prediction</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">Self-train</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/ericcombiolab/TRAFICA">https://github.com/ericcombiolab/TRAFICA</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Kabir et al. (<xref ref-type="bibr" rid="B94">94</xref>)</td>
<td valign="top" align="left">Transcription factor binding affinity prediction</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">CNN</td>
<td valign="top" align="left">Pretrained</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/lanl/EPBD-BERT">https://github.com/lanl/EPBD-BERT</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Chen et al. (<xref ref-type="bibr" rid="B343">343</xref>)</td>
<td valign="top" align="left">Transcription factor binding affinity prediction</td>
<td valign="top" align="left">GPT</td>
<td valign="top" align="left">RF</td>
<td valign="top" align="left">Pretrained</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/yiqunchen/GenePT">https://github.com/yiqunchen/GenePT</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Clauwaert et al. (<xref ref-type="bibr" rid="B50">50</xref>)</td>
<td valign="top" align="left">Transcription sites prediction, translation initiation sites, methylation sites prediction</td>
<td valign="top" align="left">Transformer-XL</td>
<td valign="top" align="left">CNN</td>
<td valign="top" align="left">Self-train</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/jdcla/DNA-transformer">https://github.com/jdcla/DNA-transformer</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Zhang et al. (<xref ref-type="bibr" rid="B344">344</xref>)</td>
<td valign="top" align="left">Conserved non-coding element classification</td>
<td valign="top" align="left">GPT</td>
<td valign="top" align="left">GPT</td>
<td valign="top" align="left">Self-train</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="http://TencentAILabHealthcare/DNAGPT(github.com)">TencentAILabHealthcare/DNAGPT(github.com)</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Zhang et al. (<xref ref-type="bibr" rid="B344">344</xref>)</td>
<td valign="top" align="left">Translation initiation sites identification</td>
<td valign="top" align="left">GPT</td>
<td valign="top" align="left">GPT</td>
<td valign="top" align="left">Self-train</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/TencentAILabHealthcare/DNAGPT">https://github.com/TencentAILabHealth-care/DNAGPT</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Wang et al. (<xref ref-type="bibr" rid="B61">61</xref>)</td>
<td valign="top" align="left">DNA replication origins prediction</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">Pretrained</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/CongWang3/PLANNER">https://github.com/CongWang3/PLANNER</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Le et al. (<xref ref-type="bibr" rid="B60">60</xref>)</td>
<td valign="top" align="left">Enhancer identification</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">CNN</td>
<td valign="top" align="left">Pretrained</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/khanhlee/bert-enhancer">https://github.com/khanhlee/bert-enhancer</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Wang et al. (<xref ref-type="bibr" rid="B61">61</xref>)</td>
<td valign="top" align="left">Enhancer identification</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">DF</td>
<td valign="top" align="left">Pretrained</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/no-banana/SMFM-master">https://github.com/no-banana/SMFM-master</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Clauwaert et al. (<xref ref-type="bibr" rid="B91">91</xref>)</td>
<td valign="top" align="left">Gene functions prediction</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">CNN</td>
<td valign="top" align="left">Pretrained</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/jdcla/DNA-transformer">https://github.com/jdcla/DNA-transformer</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Martinek et al. (<xref ref-type="bibr" rid="B283">283</xref>)</td>
<td valign="top" align="left">Enhancers identification, promoter identification</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">Pretrained</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/ML-Bioinfo-CEITEC/">https://github.com/ML-Bioinfo-CEITEC/</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Li et al. (<xref ref-type="bibr" rid="B284">284</xref>)</td>
<td valign="top" align="left">Protein-DNA interface hotspots prediction</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">RF</td>
<td valign="top" align="left">Pretrained</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/lixiangli01/PDH-EH">https://github.com/lixiangli01/PDH-EH</ext-link></td>
</tr>
<tr>
<td valign="top" align="left">Toufiq et al. (<xref ref-type="bibr" rid="B114">114</xref>)</td>
<td valign="top" align="left">Candidate gene prioritization and selection</td>
<td valign="top" align="left">GPT</td>
<td valign="top" align="left">GPT</td>
<td valign="top" align="left">Pretrained</td>
<td valign="top" align="left"><ext-link ext-link-type="uri" xlink:href="https://github.com/Drinchai/A37_LLM">https://github.com/Drinchai/A37_LLM</ext-link></td>
</tr></tbody>
</table>
</table-wrap>
<p>A close look at <xref ref-type="table" rid="T4">Table 4</xref> reveals that 25 AI-driven predictive pipelines are developed for <bold>16</bold> unique DNA sequence analysis tasks. These tasks include disease gene identification, phage-host interactions prediction, nucleosome position detection, enhancer identification, promoter identification, enhancer-promoter interactions prediction, YY1-mediated chromatin loop prediction, methylcytosine site prediction, methyladenine site prediction, essential gene identification, disease gene prediction, pseudogene function prediction, mutation susceptibility analysis, target gene classification, gene taxonomy classification, protein-DNA binding sites identification, and recombination spots identification. In addition, a high-level overview of <xref ref-type="table" rid="T4">Table 4</xref> illustrates that a total of 2 node2vec and OPA2Vec word embedding approaches along with MLP and RF classifiers have made their source code publicly available for disease gene identification. In addition, source code of 4 Word2vec and FastText word embedding approach-based predictive pipelines is publicly available for promoter and enhancer identification tasks. Furthermore, two open-source FastText and node2vec word embedding approach-based predictive pipelines are developed for essential gene identification. Moreover, four Word2vec and one FastText word embeddings based predictive pipelines are developed for DNA methylation modification predictive pipelines.</p>
<p>Overall, <xref ref-type="table" rid="T4">Table 4</xref> encompasses source codes of <bold>7</bold> unique word embedding approaches (Word2Vec, FastText, GloVe, Node2Vec, OPA2Vec, Graph2Vec, SDNE). Furthermore, a total of <bold>3</bold> machine learning classifiers, namely, RF, SVM, and XGBoost, <bold>4</bold> standalone deep learning classifiers, namely, MLP, CNN, GCN, and BiLSTM, and <bold>6</bold> hybrid deep learning models are used for the development of 26 predictive pipelines for 19 distinct DNA sequence analysis tasks.</p>
<p>Analysis of <xref ref-type="table" rid="T5">Table 5</xref> demonstrates that 38 predictive pipelines are developed using <bold>4</bold> unique large language models, namely, BERT, ALBERT, GPT, and Transformer, and <bold>9</bold> unique classifiers, namely, RF, CatBoost, DF, MLP, CRF, CNN, LSTM, FGM, and hybrid (CNN, BiLSTM). Overall, these 38 large language models based predictive models are evaluated across <bold>24</bold> unique DNA sequence analysis tasks. These 24 tasks include chromatin accessibility prediction, species classification, protein-DNA binding site prediction, long-range chromatin interaction prediction, prediction of context-specific functional impact of genetic variants, enhancer-promoter interaction prediction, gene expression prediction, tumor type prediction, methyladenine modification prediction, methylation modification prediction, methylcytosine modification prediction, conserved non-coding element classification, nucleosome position prediction, promoter identification, splice site prediction, transcription factor binding site prediction, transcription factor binding affinity prediction, transcription site prediction, translation initiation site prediction, dna replication origin prediction, enhancer identification, gene function prediction, protein-dna interface hotspots prediction, and candidate gene prioritization and selection. A high level overview of <xref ref-type="table" rid="T5">Table 5</xref> reveals that a total of two open-source chromatin accessibility predictive pipelines and two open-source gene expression prediction pipelines use transformers. In contrast, two open-source protein-DNA binding site identification pipelines and two open-source transcription factor binding site identification pipelines use BERT. In addition, three open-source transcription factor binding affinity prediction pipelines use GPT and BERT language models, whereas 8 methyl-adenine and 4 methyl-cytosine modification prediction pipelines use BERT and transformers.</p>
<p>Predictive pipelines can use language models in two different ways: (1) training a language model from scratch (self-training) on a large corpus and (2) leveraging a pre-trained open-source language model and fine-tuning it for specific downstream tasks. Overall, a critical analysis of existing studies reveals that source codes of 20 BERT, 13 Transformer, 4 GPT, and 1 Transformer-XL based predictive pipelines are publicly available. A holistic view of <xref ref-type="table" rid="T5">Table 5</xref> reveals that 23 open-source predictive pipelines perform self-training of different language models from scratch for 20 tasks, whereas 15 open-source predictive pipelines have used pre-trained language models for 11 different tasks.</p>
<p>Specifically in 20 BERT-based predictive pipelines, 9 BERT models are self-trained from scratch for nine different tasks, namely, protein-DNA binding site prediction (<xref ref-type="bibr" rid="B96">96</xref>), 6mA-methyl adenine modification prediction (<xref ref-type="bibr" rid="B144">144</xref>, <xref ref-type="bibr" rid="B281">281</xref>), DNA methylation modification prediction (<xref ref-type="bibr" rid="B159">159</xref>, <xref ref-type="bibr" rid="B248">248</xref>), 5mC-methyl cytosine modification prediction (<xref ref-type="bibr" rid="B282">282</xref>), nucleosome positioning prediction (<xref ref-type="bibr" rid="B29">29</xref>), promoter prediction (<xref ref-type="bibr" rid="B93">93</xref>), splice site prediction (<xref ref-type="bibr" rid="B93">93</xref>), transcription factor binding site prediction (<xref ref-type="bibr" rid="B93">93</xref>), and transcription factor binding affinity prediction (<xref ref-type="bibr" rid="B52">52</xref>). In contrast, 11 pre-trained BERT models are utilized to perform 7 downstream tasks, namely, protein-DNA binding site prediction (<xref ref-type="bibr" rid="B53">53</xref>), DNA methylation modification (<xref ref-type="bibr" rid="B154">154</xref>, <xref ref-type="bibr" rid="B155">155</xref>, <xref ref-type="bibr" rid="B157">157</xref>), 4mC-methyl cytosine modification prediction (<xref ref-type="bibr" rid="B143">143</xref>), transcription factor binding affinity prediction (<xref ref-type="bibr" rid="B94">94</xref>), DNA replication origin prediction (<xref ref-type="bibr" rid="B25">25</xref>), enhancer identification (<xref ref-type="bibr" rid="B60">60</xref>, <xref ref-type="bibr" rid="B61">61</xref>, <xref ref-type="bibr" rid="B283">283</xref>), and protein-DNA interface hotspots prediction (<xref ref-type="bibr" rid="B284">284</xref>). To facilitate readers, we have summarized uniquely pre-trained language models along with pre-training data for DNA sequence analysis tasks in <xref ref-type="table" rid="T6">Table 6</xref>.</p>
<table-wrap position="float" id="T6">
<label>Table 6</label>
<caption><p>Summary of uniquely pre-trained language models along with pre-training data for DNA sequence analysis tasks.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Language model</bold></th>
<th valign="top" align="left"><bold>Pre-trained data</bold></th>
<th valign="top" align="left"><bold>Language model</bold></th>
<th valign="top" align="left"><bold>Pre-trained data</bold></th>
<th valign="top" align="left"><bold>Language model</bold></th>
<th valign="top" align="left"><bold>Pre-trained data</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Zhang et al., Transformer (<xref ref-type="bibr" rid="B31">31</xref>)</td>
<td valign="top" align="left">Human reference Genome Sequences GRCh37 data</td>
<td valign="top" align="left">Stanojevic et al., Transformer (<xref ref-type="bibr" rid="B152">152</xref>)</td>
<td valign="top" align="left">893k sequences from ONT GM24385 Dataset</td>
<td valign="top" align="left">Elnaggar et al. BERT (<xref ref-type="bibr" rid="B345">345</xref>)</td>
<td valign="top" align="left">UniRef100 and BFD-100 datasets</td>
</tr>
<tr>
<td valign="top" align="left">Nguyen et al., Transformer (<xref ref-type="bibr" rid="B44">44</xref>)</td>
<td valign="top" align="left">Human reference Genome Sequences GRCh37 data</td>
<td valign="top" align="left">Clauwaert et al., Transformer-XL (<xref ref-type="bibr" rid="B50">50</xref>)</td>
<td valign="top" align="left">2.7M Genome Data for TSSs, TISs, MethSMRT for the 4mC-Methylations</td>
<td valign="top" align="left">Devlin et al., BERT (<xref ref-type="bibr" rid="B244">244</xref>)</td>
<td valign="top" align="left">BooksCorpus (800M words), English Wikipedia (2,500M words)</td>
</tr>
<tr>
<td valign="top" align="left">Gao et al., Transformer (<xref ref-type="bibr" rid="B36">36</xref>)</td>
<td valign="top" align="left">EpiGenomic data</td>
<td valign="top" align="left">Luo et al., BERT (<xref ref-type="bibr" rid="B96">96</xref>)</td>
<td valign="top" align="left">690 ChIP-Seq Dataset (20,464,149 Samples)</td>
<td valign="top" align="left">Lai et al., BERT (<xref ref-type="bibr" rid="B346">346</xref>)</td>
<td valign="top" align="left">PubMed abstracts</td>
</tr>
<tr>
<td valign="top" align="left">Ni et al., Transformer (<xref ref-type="bibr" rid="B83">83</xref>)</td>
<td valign="top" align="left">Human reference Genome Sequences</td>
<td valign="top" align="left">Le at al., BERT (<xref ref-type="bibr" rid="B281">281</xref>)</td>
<td valign="top" align="left">Cased Text in the top 104 Languages with the Largest Corpus</td>
<td valign="top" align="left">Yang et al., Transformer, ALBERT (<xref ref-type="bibr" rid="B249">249</xref>)</td>
<td valign="top" align="left">Human reference Genome Sequences hg19 data</td>
</tr>
<tr>
<td valign="top" align="left">Reddy et al., Transformer (<xref ref-type="bibr" rid="B102">102</xref>)</td>
<td valign="top" align="left">MPRA data</td>
<td valign="top" align="left">Tsukiyama et al., BERT (<xref ref-type="bibr" rid="B144">144</xref>)</td>
<td valign="top" align="left"><italic>R. chinensis</italic></td>
<td valign="top" align="left">Zhang et al., GPT (<xref ref-type="bibr" rid="B344">344</xref>)</td>
<td valign="top" align="left">Genomes from 9 species: (<italic>Arabidopsis thaliana, Caenorhabditis elegans, Bos taurus, Danio rerio, Drosophila melanogaster, Escherichia coli gca 001721525, Homo sapiens, Mus musculus, Saccharomyces cerevisiae</italic>)</td>
</tr>
<tr>
<td valign="top" align="left">Fishman et al., Transformer (<xref ref-type="bibr" rid="B32">32</xref>)</td>
<td valign="top" align="left">Human T2T v2 Genome</td>
<td valign="top" align="left">Jeong et al., BERT (<xref ref-type="bibr" rid="B159">159</xref>)</td>
<td valign="top" align="left">HG19, MM10</td>
<td valign="top" align="left">Zhang et al., GPT (<xref ref-type="bibr" rid="B344">344</xref>)</td>
<td valign="top" align="left">(a) approx.10B bps (b) approx. 200B bps</td>
</tr>
<tr>
<td valign="top" align="left">Osseni et al., Transformer (<xref ref-type="bibr" rid="B180">180</xref>)</td>
<td valign="top" align="left">Omics dataset</td>
<td valign="top" align="left">Huson et al., BERT (<xref ref-type="bibr" rid="B248">248</xref>)</td>
<td valign="top" align="left">DNA Methylation and taxonomy Data</td>
<td valign="top" align="left">Cui et al., GPT (<xref ref-type="bibr" rid="B347">347</xref>)</td>
<td valign="top" align="left">NCBI text descriptions of individual genes</td>
</tr>
<tr>
<td valign="top" align="left">Wang et al., Transformer (<xref ref-type="bibr" rid="B158">158</xref>)</td>
<td valign="top" align="left">WGBS dataset</td>
<td valign="top" align="left">Wang et al., BERT (<xref ref-type="bibr" rid="B282">282</xref>)</td>
<td valign="top" align="left">1,825,095 Promoter Sequences</td>
<td valign="top" align="left">Toufiq et al., GPT 4 (<xref ref-type="bibr" rid="B114">114</xref>)</td>
<td valign="top" align="left">Co&#x02013;expression gene set (M9.2) from the BloodGen3 repertoire associated with circulating erythroid cells</td>
</tr>
<tr>
<td valign="top" align="left">Zhou et al., Transformer (<xref ref-type="bibr" rid="B250">250</xref>)</td>
<td valign="top" align="left">WGBS dataset</td>
<td valign="top" align="left">Fazeel et al., BERT (<xref ref-type="bibr" rid="B29">29</xref>)</td>
<td valign="top" align="left">Human reference Genome Sequences with the length of sequences between 5 and 510 with 3-mer</td>
<td valign="top" align="left">Toufiq et al., Claud (<xref ref-type="bibr" rid="B114">114</xref>)</td>
<td valign="top" align="left">Co-expression gene set (M9.2) from the BloodGen3 repertoire associated with circulating erythroid cells</td>
</tr>
<tr>
<td valign="top" align="left">Dalla et al., Transformer (<xref ref-type="bibr" rid="B100">100</xref>)</td>
<td valign="top" align="left">A total of 850 species, whose Genomes add up to 174B nucleotides</td>
<td valign="top" align="left">Ji et al., BERT (<xref ref-type="bibr" rid="B93">93</xref>)</td>
<td valign="top" align="left">Human reference Genome Sequences</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Clauwaert et al., Transformer (<xref ref-type="bibr" rid="B91">91</xref>)</td>
<td valign="top" align="left">9 283 204 full genome sequence</td>
<td valign="top" align="left">Xu et al., BERT (<xref ref-type="bibr" rid="B52">52</xref>)</td>
<td valign="top" align="left">ATAC-Seq dataset with over 13M nucleotide sequences</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr></tbody>
</table>
</table-wrap>
</sec>
<sec id="s10">
<title>10 DNA sequence analysis predictive pipeline performance analysis</title>
<p>This section facilitates AI researchers by providing details of performance figures achieved over diverse benchmark datasets for all three kinds of predictive pipelines, namely, word embedding, language models, and nucleotide compositional and positional information across 44 distinct DNA sequence analysis tasks. To assist researchers for developing novel predictive pipelines, we have thoroughly analyzed literature and identified the current state-of-the-art predictor for each task. Section 3 provides categorization of 44 DNA sequence analysis tasks into 8 different categories. In this section, we have summarized the performance values of the predictive pipelines for these 44 tasks into 7 different Tables. Each Table corresponds to the predictive pipelines of tasks within one category, except for 1 Table that includes predictive pipelines related to tasks from 3 different categories. Within these Tables, highlighted predictors represent state-of-the-art performance values on public datasets across each task. Furthermore, this section also facilitates crucial information that which of the tasks of every goal offer more room for improvement through the development of more robust and effective predictive pipelines.</p>
<p><xref ref-type="table" rid="T7">Table 7</xref> summarizes the crucial details of seven DNA sequence analysis tasks classified under the hood of genome structure and stability. Overall, for genome structure and stability goal, four unique representation learning methods, namely, BERT, Transformer, Word2vec, and multi-scale convolution, in conjunction with bi-directional gated recurrent unit methods are used across seven different tasks. Similarly, six unique classifiers, namely, BERT, LSTM, Transformer, CNN&#x0002B;LSTM, CapsNet, and CNN, are used in seven different task predictive pipelines. Most commonly used representation learning scheme for this goal is BERT followed by Transformer. BERT is most commonly used with a self-classifier for three different tasks and used with LSTM classifier for one task. Transformer is used with only self-classifier for three different tasks. Word2vec potential is explored with CNN-based classifiers for two different tasks and multi-scale convolution in conjunction with bi-directional gated recurrent unit method is only explored with CNN classifier for one task. Overall, among all predictive pipelines, BERT with self-classifier or LSTM classifier manages to achieve top performance figures as compared to transformer-based predictive pipelines. Among all 7 tasks, genome structure analysis and long range chromatin interaction prediction tasks provide a lot of room for improvement as the performance of their predictive models fall below 70%. BERT or Transformer with CapsNet classifier-based predictive pipeline can potentially enhance the performance on either or both of these tasks.</p>
<table-wrap position="float" id="T7">
<label>Table 7</label>
<caption><p>Genome structure and stability related 10 distinct DNA sequence analysis task predictive pipeline performance.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Task type</bold></th>
<th valign="top" align="left"><bold>Task name</bold></th>
<th valign="top" align="left"><bold>References</bold></th>
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="left"><bold>Representation learning</bold></th>
<th valign="top" align="left"><bold>Classifier</bold></th>
<th valign="top" align="left"><bold>Performance evaluation</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">DNA replication origins prediction</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B25"><bold>25</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Gao et al. Dataset (</bold><italic><bold>A. thaliana</bold></italic><bold>)</bold></td>
<td valign="top" align="left"><bold>BERT</bold></td>
<td valign="top" align="left">_</td>
<td valign="top" align="left"><italic><bold>A. thaliana</bold></italic><bold>: AUROC = 0.9811</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B329">329</xref>)</td>
<td valign="top" align="left">Wu et al. Datasets (<italic>S. cerevisiae</italic> Dataset, <italic>S. pombe</italic> Dataset, <italic>K. lactis</italic> Dataset, <italic>P. pastoris</italic> Dataset)</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">CNN</td>
<td valign="top" align="left">Accuracy [<italic>S. cerevisiae</italic> (S1): 0.975, <italic>S. pombe</italic> (S2): 0.765, <italic>K. lactis</italic> (S3): 0.885, <italic>P. pastoris</italic> (S4): 0.967]; MCC [<italic>S. cerevisiae</italic> (S1): 0.940, <italic>S. pombe</italic> (S2): 0.530, <italic>K. lactis</italic> (S3): 0.771, <italic>P. pastoris</italic> (S4): 0.934]; AUC [<italic>S. cerevisiae</italic> (S1): 0.975, <italic>S. pombe</italic> (S2): 0.800, <italic>K. lactis</italic> (S3): 0.888, <italic>P. pastoris</italic> (S4): 0.981]</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Nucleosome position detection</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B29"><bold>29</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Gangi et al. Dataset 1 [a. C. elegans (CE), b. D. melanogester (DM), c. S. cerevisiae (YS), d</bold>. <italic><bold>H. sapiens</bold></italic> <bold>(HM)], Gangi et al. Dataset 2 (DM-5U, DM-PM, DM-LC, HM-5U, HM-PM, HM-LC, YS-PM, YS-WG)</bold></td>
<td valign="top" align="left"><bold>BERT</bold></td>
<td valign="top" align="left"><bold>LSTM</bold></td>
<td valign="top" align="left"><bold>Dataset 1 a. CE: Acc = 90.5, Sn = 91.8, Sp = 92.1, Precision = 91.8, MCC = 80.5, AUROC = 95.8 b. DM: Acc = 85.1, Sn = 84.8, Sp = 85.6, Precision = 85.3, MCC = 70.5, AUROC = 92.4 c. YS: Acc = 100, Sn = 100, Sp = 99.8, Precision = 99.8, MCC = 99.82, AUROC = 100 d. HM: Acc = 88.3, Sn = 88.3, Sp = 88.4, Precision = 88.5, MCC = 76.8, AUROC = 94.4 Dataset 2 a. DM-5U: Acc = 69.5, Sn = 41.1, Sp = 85.8, Precision = 63.8, MCC = 30.8, AUROC = 68.3 b. DM-PM: Acc = 73.6, Sn = 40.1, Sp = 93.6, Precision = 80.4, MCC = 42.0, AUROC = 73.7 c. DM-LC: Acc = 71.3, Sn = 43.1, Sp = 90.0, Precision = 75.2, MCC = 38.7, AUROC = 72.0 d. HM-5U: Acc = 81.8, Sn = 51.6, Sp = 94.3, Precision = 80.0, MCC = 53.4, AUROC = 80.2 e. HM-LC: Acc = 91.1, Sn = 83.7, Sp = 96.1, Precision = 93.7, MCC = 81.7, AUROC = 95.1 f. HM-PM: Acc = 85.1, Sn = 75.8, Sp = 92.4, Precision = 89.1, MCC = 70.1, AUROC = 90.4 g. YS-PM: Acc = 92.4, Sn = 63.1, Sp = 97.2, Precision = 79.6, MCC = 66.3, AUROC = 93.5 h. YS-WG: Acc = 94.3, Sn = 60.3, Sp = 98.4, Precision = 94.3, MCC = 67.2, AUROC = 94.5</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Chromatin accessibility prediction</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B32"><bold>32</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>DeepSEA dataset (TF &#x00026; DHS)</bold></td>
<td valign="top" align="left"><bold>BERT</bold></td>
<td valign="top" align="left">_</td>
<td valign="top" align="left"><bold>TF: AUROC = 96.81</bold> <bold>&#x000B1;</bold> <bold>0.1 DHS: AUROC = 92.8</bold> <bold>&#x000B1;</bold> <bold>0.03</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B31"><bold>31</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>DNase-Seq experiment Data</bold></td>
<td valign="top" align="left"><bold>Transformer</bold></td>
<td valign="top" align="left">_</td>
<td valign="top" align="left"><bold>Average AUROC = 0.8977 Average AUPRC = 0.8983</bold></td>
</tr>
<tr>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">YY1-mediated chromatin loops prediction</td>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B39">39</xref>)</td>
<td valign="top" align="left">1. HCT116 2. K562</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">CNN&#x0002B;LSTM</td>
<td valign="top" align="left">K562: AUROC = 98.2, Acc = 92.9 HCT116: AUROC = 95.7, Acc = 88.5</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B38"><bold>38</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>1. HCT116 2. K562</bold></td>
<td valign="top" align="left"><bold>MSC&#x0002B;BiGRU</bold></td>
<td valign="top" align="left"><bold>CapsNet</bold></td>
<td valign="top" align="left"><bold>10-fold cross-validation HCT116: Acc = 0.9544, AUROC = 0.9886 K562: Acc = 0.9680, AUROC = 0.9924</bold> Independent test setting HCT116: Acc = 0.9622, AUROC = 0.9913, AUPRC = 0.9917, F1-score = 0.9564 K562: Acc = 0.9560, AUROC = 0.9912, AUPRC = 0.9920, F1-score = 0.9583</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B45">45</xref>)</td>
<td valign="top" align="left">1. HCT116 2. K562</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">CNN</td>
<td valign="top" align="left">1. AUROC = 0.93 2. AUROC = 0.93</td>
</tr>
<tr>
<td valign="top" align="left">Multi-class classification</td>
<td valign="top" align="left">Genome structure analysis</td>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B27">27</xref>)</td>
<td valign="top" align="left">NCycDB</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Macro F1-score = 61.8, Weighted F1-score = 65.4</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Chromatin feature prediction</td>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B34">34</xref>)</td>
<td valign="top" align="left">Yang et al. Datasets 1. LOGO-919, 2. LOGO-2002 (TF &#x00026; DHS), LOGO-3357 (TF &#x00026; DHS)</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">LOGO-919: AUROC = 0.703 LOGO-2002: TF: AUROC = 0.954, DHSs: AUROC = 0.913, HM: AUROC = 0.883 LOGO-3357: TF: AUROC = 0.926, DHSs: AUROC = 0.928, HM: AUROC = 0.883</td>
</tr>
<tr>
<td valign="top" align="left">Interaction</td>
<td valign="top" align="left">Long range chromatin interaction prediction</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B36"><bold>36</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>ChIP-seq Dataset</bold></td>
<td valign="top" align="left"><bold>Transformer</bold></td>
<td valign="top" align="left">_</td>
<td valign="top" align="left"><bold>AUPRC = 0.647</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Bold values highlight top performers on unique datasets related to distinct tasks.</p>
</table-wrap-foot>
</table-wrap>
<p>Furthermore, <xref ref-type="table" rid="T8">Table 8</xref> provides a high-level overview of the performance achieved by 48 predictors for 9 DNA sequence analysis tasks classified under the hood of gene expression regulation. Overall, for gene expression regulation goal, 10 unique representation learning methods, namely, ULMFiT, BERT, One-hot encoding, Word2vec, FastText, C2&#x0002B;NCP, Node2vec&#x0002B;SocDim&#x0002B;Grarep, ELECTRA, ALBERT, and Transformer, are used across 9 different tasks. Overall, 14 unique classifiers are employed in these predictive pipelines, namely, CNN, MLP, DF, CNN&#x0002B;BiLSTM, CNN&#x0002B;LSTM, SVM, Siamese network, DenseNet, CatBoost, TCN, XGBoost, RF&#x0002B;CNN, LSTM, BiGRU, and CNN&#x0002B;BiGRU.</p>
<table-wrap position="float" id="T8">
<label>Table 8</label>
<caption><p>Gene expression regulation related 48 distinct DNA sequence analysis task predictive pipeline performance.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Task type</bold></th>
<th valign="top" align="left"><bold>Task name</bold></th>
<th valign="top" align="left"><bold>References</bold></th>
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="left"><bold>Representation learning</bold></th>
<th valign="top" align="left"><bold>Classifier</bold></th>
<th valign="top" align="left"><bold>Performance evaluation</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">Enhancer identification</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B57"><bold>57</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Liu et al. dataset (1, 2)</bold></td>
<td valign="top" align="left"><bold>ULMFiT</bold></td>
<td valign="top" align="left"><bold>CNN</bold></td>
<td valign="top" align="left">1. Enhancer/ Non-Enhancer Cross-Validation: Acc = 0.946, Sn = 0.946, Sp = 0.949, MCC = 0.892 Independent: Acc = 0.843, Sn = 0.842, Sp = 0.87, MCC = 0.686 <bold>2. Weak/ Strong-Enhancer Cross-Validation: Acc = 0.90, Sn = 0.90, Sp = 0.896, MCC = 0.8 Independent: Acc = 0.875, Sn = 0.873, Sp = 0.75, MCC = 0.774</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B56"><bold>56</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Liu et al. dataset 1</bold></td>
<td valign="top" align="left"><bold>BERT</bold></td>
<td valign="top" align="left"><bold>CNN</bold></td>
<td valign="top" align="left"><bold>Sn = 1.00, Sp = 1.00, Acc = 1.00, MCC = 1.00, AUROC = 1.00</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B55"><bold>55</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>1. DiseaseEnhancer 2. EnDisease 3. CancerEnD</bold></td>
<td valign="top" align="left"><bold>One-hot Encoding</bold></td>
<td valign="top" align="left"><bold>MLP</bold></td>
<td valign="top" align="left"><bold>1. AUROC = 0.9645</bold> <bold>&#x000B1;</bold> <bold>0.0057, AUPRC = 0.9647</bold> <bold>&#x000B1;</bold> <bold>0.0043, Acc = 0.8986</bold> <bold>&#x000B1;</bold> <bold>0.0169, Precision = 0.8765</bold> <bold>&#x000B1;</bold> <bold>0.0266, Recall = 0.9290</bold> <bold>&#x000B1;</bold> <bold>0.0132, F1-score = 0.9018</bold> <bold>&#x000B1;</bold> <bold>0.0169 2. AUROC = 0.9546</bold> <bold>&#x000B1;</bold> <bold>0.0036, AUPRC = 0.9474</bold> <bold>&#x000B1;</bold> <bold>0.0118, Acc = 0.8959</bold> <bold>&#x000B1;</bold> <bold>0.0141, Precision = 0.8583</bold> <bold>&#x000B1;</bold> <bold>0.0261, Recall = 0.9469</bold> <bold>&#x000B1;</bold> <bold>0.0103, F1-score = 0.9003</bold> <bold>&#x000B1;</bold> <bold>0.0182 3. AUROC = 0.9755</bold> <bold>&#x000B1;</bold> <bold>0.0026, AUPRC = 0.9673</bold> <bold>&#x000B1;</bold> <bold>0.0047, Acc = 0.9306</bold> <bold>&#x000B1;</bold> <bold>0.0053, Precision = 0.9373</bold> <bold>&#x000B1;</bold> <bold>0.0051, Recall = 0.9261</bold> <bold>&#x000B1;</bold> <bold>0.0085, F1-score = 0.9317</bold> <bold>&#x000B1;</bold> <bold>0.0053</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B67"><bold>67</bold></xref><bold>)</bold></td>
<td valign="top" align="left">Liu et al. dataset 1, <bold>Basith et al. dataset (HEK293, NHEK, K652, GN12878, HMEC, HSMM, NHLF, HUVEC)</bold></td>
<td valign="top" align="left"><bold>BERT</bold></td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">(Liu et al. dataset 1) Acc = 0.8300, Sn = 0.8000, Sp = 0.8600, MCC = 0.6612, AUROC = 0.8560 <bold>(Basith&#x00027;s dataset) HEK293: Acc = 0.8732, Sn = 0.8666, Sp = 0.8798, MCC = 0.7283, AUROC = 0.9443 NHEK: Acc = 0.7766, Sn = 0.7229, Sp = 0.8303, MCC = 0.5453, AUROC = 0.8716 K652: Acc = 0.7974, Sn = 0.8180, Sp = 0.7767, MCC = 0.5679, AUROC = 0.8712 GM12878: Acc = 0.8222, Sn = 0.7564, Sp = 0.8879, MCC = 0.6475, AUROC = 0.9179 HMEC: Acc = 0.7645, Sn = 0.7638, Sp = 0.7652, MCC = 0.5068, AUROC = 0.8631 HSMM: Acc = 0.7193, Sn = 0.7191, Sp = 0.7194, MCC = 0.4179, AUROC = 0.7948 NHLF: Acc = 0.7884, Sn = 0.8236, Sp = 0.7532, MCC = 0.5479, AUROC = 0.8623 HUVEC: Acc = 0.7334, Sn = 0.7691, Sp = 0.6977, MCC = 0.4417, AUROC = 0.8045</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B61">61</xref>)</td>
<td valign="top" align="left">Liu et al. dataset</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">DF</td>
<td valign="top" align="left">AUROC = 0.808, Acc = 0.822, MCC = 0.655, Sn = 0.834, Sp = 0.810</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B58">58</xref>)</td>
<td valign="top" align="left">Liu et al. dataset (1, 2)</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">CNN &#x0002B; BiLSTM</td>
<td valign="top" align="left">1: Acc = 83.32, Sn = 84.18, Sp = 82.45, MCC = 0.6668 2: Acc = 83.30, Sn = 89.27, Sp = 77.33, MCC = 0.6736</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B65">65</xref>)</td>
<td valign="top" align="left">Liu et al. dataset (1, 2)</td>
<td valign="top" align="left">FastText</td>
<td valign="top" align="left">LSTM &#x0002B; CNN</td>
<td valign="top" align="left">Enhancer Prediction: Acc = 0.7525, MCC = 0.5051; Enhancer type Prediction: Acc = 0.6972, MCC = 0.3954;</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B59">59</xref>)</td>
<td valign="top" align="left">Liu et al. dataset (1, 2)</td>
<td valign="top" align="left">FastText</td>
<td valign="top" align="left">MLP</td>
<td valign="top" align="left">1: Sn = 85.81, Sp = 86.35, Acc = 86.07, MCC = 0.722; 2: Sn = 69.90, Sp = 69.32, Acc = 69.59, MCC = 0.392</td>
</tr>
<tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B66">66</xref>)</td>
<td valign="top" align="left">Liu et al. dataset (1, 2)</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">CNN</td>
<td valign="top" align="left">1: Acc = 0.784, Sn = 0.811, Sp = 0.758, MCC = 0.567; 2: Acc = 0.749, Sn = 0.961, Sp = 0.537, MCC = 0.505</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B60">60</xref>)</td>
<td valign="top" align="left">Liu et al. dataset</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">CNN</td>
<td valign="top" align="left">Sn = 80, Sp = 71.2, Acc = 75.6, MCC = 0.514</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B47">47</xref>)</td>
<td valign="top" align="left">Liu et al. dataset (1, 2)</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">CNN</td>
<td valign="top" align="left">1: Sn = 75.88, Sp = 88.88, Acc = 80.63, MCC = 0.6929, AUROC = 0.8957 2: Sn = 73.64, Sp = 76.80, Acc = 76.43, MCC = 0.4505, AUROC = 0.8109</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B64">64</xref>)</td>
<td valign="top" align="left">Liu et al. dataset (1, 2)</td>
<td valign="top" align="left">FastText</td>
<td valign="top" align="left">SVM</td>
<td valign="top" align="left">Cross-validation 1: Sn = 81.1, Sp = 83.5, Acc = 82.3, MCC = 0.65 2: Sn = 75.3, Sp = 60.8, Acc = 68.1, MCC = 0.37 independent 1: Sn = 82, Sp = 76, Acc = 79, MCC = 0.58 2: Sn = 74, Sp = 53, Acc = 63.5, MCC = 0.28</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Promoter identification</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B77"><bold>77</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Yang et al. dataset</bold></td>
<td valign="top" align="left"><bold>One-hot encoding</bold></td>
<td valign="top" align="left"><bold>Siamese network</bold></td>
<td valign="top" align="left"><bold>Acc = 96.80, Sn = 95.08, Sp = 98.56, MCC = 0.9367</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B251">251</xref>)</td>
<td valign="top" align="left">Xiao et al. dataset 1</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">MCC = 0.81, AUPRC = 0.98</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B331"><bold>331</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Xiao et al. dataset (1, 2)</bold></td>
<td valign="top" align="left"><bold><italic>C</italic></bold>2&#x0002B;<italic><bold>NCP</bold></italic></td>
<td valign="top" align="left"><bold>DenseNet</bold></td>
<td valign="top" align="left"><bold>1: Sn = 0.9389</bold> <bold>&#x000B1;</bold> <bold>0.0106, Sp = 0.9471</bold> <bold>&#x000B1;</bold> <bold>0.0117, Acc = 0.9429</bold> <bold>&#x000B1;</bold> <bold>0.0038, MCC = 0.8865</bold> <bold>&#x000B1;</bold> <bold>0.0076, AUROC = 0.9774</bold> <bold>&#x000B1;</bold> <bold>0.0017, F1-score = 0.9432</bold> <bold>&#x000B1;</bold> <bold>0.0040 2: Sn = 0.8711</bold> <bold>&#x000B1;</bold> <bold>0.0168, Sp = 0.9211</bold> <bold>&#x000B1;</bold> <bold>0.0088, Acc = 0.8967</bold> <bold>&#x000B1;</bold> <bold>0.0086, MCC = 0.7947</bold> <bold>&#x000B1;</bold> <bold>0.0165, AUROC = 0.9353</bold> <bold>&#x000B1;</bold> <bold>0.0042, F1-score = 0.8880</bold> <bold>&#x000B1;</bold> <bold>0.0117</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B236"><bold>236</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Wang et al., Datasets (1-8)</bold></td>
<td valign="top" align="left"><bold><italic>Node</italic>2<italic>Vec</italic>&#x0002B;</bold> <bold><italic>SocDim</italic>&#x0002B;</bold> <bold>GraRep</bold></td>
<td valign="top" align="left"><bold>CatBoost</bold></td>
<td valign="top" align="left"><bold>1</bold>. <italic><bold>H. Sapiens</bold></italic><bold>-I: Sn = 0.9565, Sp = 0.9782, Acc = 0.9673, MCC = 0.9350, Precision = 0.9772, F1-score = 0.9670, AUROC = 0.9952 2</bold>. <italic><bold>H. Sapiens</bold></italic><bold>-II: Sn = 0.8646, Sp = 0.9849, Acc = 0.9248, MCC = 0.8558, Precision = 0.9829, F1-score = 0.9199, AUROC = 0.9844 3</bold>. <italic><bold>R. Norvegicus</bold></italic><bold>-I: Sn = 0.9113, Sp = 0.9757, Acc = 0.9424, MCC = 0.8908, Precision = 0.9761, F1-score = 0.9425, AUROC = 0.9975 4</bold>. <italic><bold>R. Norvegicus</bold></italic><bold>-II: Sn = 0.8905, Sp = 0.9849, Acc = 0.9377, MCC = 0.8793, Precision = 0.9833, F1-score = 0.9346, AUROC = 0.9832 5</bold>. <italic><bold>D. melanogaster</bold></italic><bold>-I: Sn = 0.9350, Sp = 0.9670, Acc = 0.9560, MCC = 0.9123, Precision = 0.9662, F1-score = 0.9555, AUROC = 0.9923 6</bold>. <italic><bold>D. melanogaster</bold></italic><bold>-II: Sn = 0.9425, Sp = 0.7819, Acc = 0.8596, MCC = 0.7281, Precision = 0.8112, F1-score = 0.8697, AUROC = 0.9448 7</bold>. <italic><bold>Z. mays</bold></italic><bold>-I: Sn = 0.934, Sp = 0.9450, Acc = 0.9395, MCC = 0.8791, Precision = 0.9444, F1-score = 0.9392, AUROC = 0.9841 8</bold>. <italic><bold>Z. mays</bold></italic><bold>-II: Sn = 0.9516, Sp = 0.7806, Acc = 0.8661, MCC = 0.7433, Precision = 0.8127, F1-score = 0.8767, AUROC = 0.9485</bold></td>
</tr>
<tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B235">235</xref>)</td>
<td valign="top" align="left">Xiao et al. dataset 1</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">TCN</td>
<td valign="top" align="left">Acc = 91.86, Sn = 92.74, Sp = 91</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B348"><bold>348</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Ji et al. Dataset</bold></td>
<td valign="top" align="left"><bold>BERT</bold></td>
<td valign="top" align="left">_</td>
<td valign="top" align="left"><bold>Acc = 0.8613, AUROC = 0.9354, MCC = 0.7226, Precision = 0.8569, Recall = 0.8624</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B349"><bold>349</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Liu et al. dataset (1, 2)</bold></td>
<td valign="top" align="left"><bold>BERT</bold></td>
<td valign="top" align="left"><bold>XGBoost</bold></td>
<td valign="top" align="left"><bold>1. Promoter Identification: Sn = 84.34, Sp = 86.56, Acc = 85.45 2. Promoter strength classification: Sn = 70.85, Sp = 81.63, Acc = 76.92</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B89">89</xref>)</td>
<td valign="top" align="left">Human dataset</td>
<td valign="top" align="left">ELECTRA</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Acc = 0.862, AUROC = 0.935, F1-score = 0.862, MCC = 0.725, Precision = 0.863, Recall = 0.862</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B76"><bold>76</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>1. K562 2. GM12878 3. HeLa-S3 4. HUVEC</bold></td>
<td valign="top" align="left"><bold>Word2Vec</bold></td>
<td valign="top" align="left"><bold>RF &#x0002B; CNN</bold></td>
<td valign="top" align="left"><bold>1. AUROC = 0.9809, Acc = 0.9415, MCC = 0.8831 2. AUROC = 0.9783, Acc = 0.9334, MCC = 0.8668 3. AUROC = 0.9824, Acc = 0.9374, MCC = 0.8749 4. AUROC = 0.9847, Acc = 0.9481, MCC = 0.8963</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B48">48</xref>)</td>
<td valign="top" align="left">Xiao et al. dataset</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">LSTM</td>
<td valign="top" align="left">Acc = 90.59, MCC = 0.8114, Sn = 90.28, Sp = 90.94</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B249">249</xref>)</td>
<td valign="top" align="left">Yang et al. dataset</td>
<td valign="top" align="left">ALBERT</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">AUROC = 0.743</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B90">90</xref>)</td>
<td valign="top" align="left">Ji et al. dataset</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">CNN</td>
<td valign="top" align="left">Precision = 0.805, Recall = 0.803, Acc = 0.894</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B90">90</xref>)</td>
<td valign="top" align="left">Ji et al. dataset</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Precision = 0.805, Recall = 0.803, Acc = 0.894</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B34">34</xref>)</td>
<td valign="top" align="left">Yang et al. dataset</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Recall = 0.921, Precision = 0.940, F1-score = 0.933</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B93">93</xref>)</td>
<td valign="top" align="left">Human dataset</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">AUROC = 0.981, AUPRC = 0.982</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B79">79</xref>)</td>
<td valign="top" align="left">Xiao et al. dataset (1, 2)</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">CNN</td>
<td valign="top" align="left">1: Acc = 91.42 2: Acc = 82.42</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B78">78</xref>)</td>
<td valign="top" align="left">Xiao et al. dataset (1, 2)</td>
<td valign="top" align="left">FastText</td>
<td valign="top" align="left">CNN</td>
<td valign="top" align="left">1: Acc = 85.41 2: Acc = 73.1</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Transcription sites prediction</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B50"><bold>50</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Clauwaert et al. Dataeset</bold></td>
<td valign="top" align="left"><bold>Transformer-XL</bold></td>
<td valign="top" align="left"><bold>CNN</bold></td>
<td valign="top" align="left"><bold>AUROC = 0.977</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Transcription factor binding sites prediction</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B94"><bold>94</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>ChIP-Seq Dataset</bold></td>
<td valign="top" align="left"><bold>BERT</bold></td>
<td valign="top" align="left"><bold>CNN</bold></td>
<td valign="top" align="left"><bold>AUROC = 0.949, AUPRC = 0.326</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B89"><bold>89</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>690 ChIP-Seq dataset</bold></td>
<td valign="top" align="left"><bold>ELECTRA</bold></td>
<td valign="top" align="left">_</td>
<td valign="top" align="left"><bold>Acc = 0.856, AUROC = 0.935, F1-score = 0.851, MCC = 0.727, Precision = 0.859, Recall = 0.856</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B90"><bold>90</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>TF ChIP-Seq dataset</bold></td>
<td valign="top" align="left"><bold>BERT</bold></td>
<td valign="top" align="left">_</td>
<td valign="top" align="left"><bold>Precision = 0.937, Recall = 0.935, Acc = 0.989</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B91"><bold>91</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>TSSs dataset</bold></td>
<td valign="top" align="left"><bold>Transformer</bold></td>
<td valign="top" align="left"><bold>CNN</bold></td>
<td valign="top" align="left"><bold>AUROC = 0.981. AUPRC = 0.141</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B93"><bold>93</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>497 TF ChIP-Seq dataset</bold></td>
<td valign="top" align="left"><bold>BERT</bold></td>
<td valign="top" align="left">_</td>
<td valign="top" align="left"><bold>Mean Acc = 0.903, Mean AUROC = 0.954, Mean F1-score = 0.901, Mean MCC = 0.807, Mean Precision = 0.898, Mean Recall = 0.909</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B92">92</xref>)</td>
<td valign="top" align="left">1. A549 dataset 2. MCF-7 dataset 3. H1-HESC dataset 4. HUVEC dataset</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">BiGRU</td>
<td valign="top" align="left">HUVEC: Average Precision = 0.9618, Average AUROC = 0.9608 MCF7: Average Precision = 0.9653, Average AUROC = 0.9643 A549: Average Precision = 0.9608, Average AUROC = 0.9593 H1-HESC: Average Precision = 0.9528, Average AUROC = 0.9524</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">Splice sites prediction</td>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B97">97</xref>)</td>
<td valign="top" align="left">Splice-junction Gene sequence dataset</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">LSTM</td>
<td valign="top" align="left">Acc = 0.98, Precision = 0.98, Recall = 0.99, F1-score = 0.98</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B98"><bold>98</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Degroeve et al. Dataset</bold></td>
<td valign="top" align="left">_</td>
<td valign="top" align="left"><bold>CNN</bold></td>
<td valign="top" align="left"><bold>F1-score = 0.9187</bold> <bold>&#x000B1;</bold> <bold>0.0070, MCC = 0.9028</bold> <bold>&#x000B1;</bold> <bold>0.0085</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B99"><bold>99</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Liu et al. Dataset</bold></td>
<td valign="top" align="left"><bold>One-hot Encoding</bold></td>
<td valign="top" align="left"><bold>MLP</bold></td>
<td valign="top" align="left"><bold>Donor: Acc = 96.57 Acceptor: Acc = 95.82</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B100"><bold>100</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Wang et al. Dataset</bold></td>
<td valign="top" align="left"><bold>Transformer</bold></td>
<td valign="top" align="left">_</td>
<td valign="top" align="left"><bold>AUPRC = 0.984</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B93"><bold>93</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Ji et al. Dataset</bold></td>
<td valign="top" align="left"><bold>BERT</bold></td>
<td valign="top" align="left">_</td>
<td valign="top" align="left"><bold>Acc = 0.939, AUROC = 0.992, F1-score = 0.937, MCC = 0.903, Precision = 0.961, Recall = 0.919</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Translation initiation site prediction</td>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B101">101</xref>)</td>
<td valign="top" align="left">Kalkatawi et al. TIS dataset</td>
<td valign="top" align="left">k-mer Embedding</td>
<td valign="top" align="left">Bi-GRU</td>
<td valign="top" align="left">Acc = 96.40, AUROC = 96.40, AUPRC = 94.87</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B50"><bold>50</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>TIS dataset</bold></td>
<td valign="top" align="left"><bold>Transformer-XL</bold></td>
<td valign="top" align="left"><bold>CNN</bold></td>
<td valign="top" align="left"><bold>AUROC = 0.998</bold></td>
</tr>
<tr>
<td valign="top" align="left">Interaction</td>
<td valign="top" align="left">Enhancer-promoter interactions prediction</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B350"><bold>350</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Whalen et al. datasets (GM12878, HUVEC, HeLa-S3, IMR90, K562, NHEK)</bold></td>
<td valign="top" align="left"><bold>GCN, K-mer</bold></td>
<td valign="top" align="left"><bold>GCN</bold></td>
<td valign="top" align="left"><bold>F1-score: GM12878 = 0.8679, HUVEC = 0.8954, HeLa-S3 = 0.9175, IMR90 = 0.7949, NHEK = 0.6085</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B233">233</xref>)</td>
<td valign="top" align="left">Whalen et al. datasets (HMEC, IMR90, K562, NHEK)</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">Multi-Scale CNN</td>
<td valign="top" align="left">HMEC: AUROC = 0.9344, AUPRC = 0.6852, IMR90 AUROC = 0.8936, AUPRC = 0.5878, K562 AUROC = 0.8513, AUPRC = 0.2101, NHEK AUROC = 0.8243, AUPRC = 0.4760</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B332">332</xref>)</td>
<td valign="top" align="left">Zheng et al. Datasets (GM12878, HeLa)</td>
<td valign="top" align="left">Deeptools</td>
<td valign="top" align="left">RF</td>
<td valign="top" align="left">1. Sn = 0.578, Sp = 0.964, Precision = 0.799, Acc = 0.887, AUROC = 0.919, AUPRC = 0.773, 2. Sn = 0.363, Sp = 0.953, Precision = 0.660, Acc = 0.836, AUROC = 0.831, AUPRC = 0.601</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B177">177</xref>)</td>
<td valign="top" align="left">ESKAPE dataset</td>
<td valign="top" align="left">SDNE, Word2Vec</td>
<td valign="top" align="left">MLP</td>
<td valign="top" align="left">Acc = 86.65 &#x000B1; 1.55, Sn = 88.40 &#x000B1; 1.81, Sp = 84.91 &#x000B1; 1.96, Precision = 85.43 &#x000B1; 1.74, F1-score = 86.88 &#x000B1; 1.53, AUROC = 0.9208 &#x000B1; 0.0119</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B249"><bold>249</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Yang et al. Datasets (FoeT, Mon, nCD4, tB, tCD4, tCD8)</bold></td>
<td valign="top" align="left"><bold>Transformer</bold></td>
<td valign="top" align="left"><bold>MLP</bold></td>
<td valign="top" align="left"><bold>FoeT: AUPRC = 0.9447, Mon: AUPRC = 0.9414, nCD4: AUPRC = 0.9457, tB: AUPRC = 0.9474, tCD4: AUPRC = 0.9475, tCD8: AUPRC = 0.9387</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B82">82</xref>)</td>
<td valign="top" align="left">Whalen et al. Datasets (GM12878, HUVEC, HeLa-S3, IMR90, K562, NHEK)</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">CNN &#x0002B; BiGRU</td>
<td valign="top" align="left">GM12878: AUROC = 0.965, AUPRC = 0.819, HUVEC: AUROC = 0.950, AUPRC = 0.773, HeLa-S3: AUROC = 0.960, AUPRC = 0.820, IMR90: AUROC = 0.962, AUPRC = 0.801, K562: AUROC = 0.959, AUPRC = 0.814, NHEK: AUROC = 0.985, AUPRC = 0.899</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B49">49</xref>)</td>
<td valign="top" align="left">Whalen et al. Datasets (GM12878, HUVEC, HeLa-S3, IMR90, K562, NHEK)</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">CNN&#x0002B;BiGRU</td>
<td valign="top" align="left">GM12878: AUROC = 0.949, AUPRC = 0.819, F1-score = 0.766, HUVEC: AUROC = 0.948, AUPRC = 0.720, F1-score = 0.649, HeLa-S3: AUROC = 0.952, AUPRC = 0.824, F1-score = 0.78, IMR90: AUROC = 0.948, AUPRC = 0.818, F1-score = 0.778, K562: AUROC = 0.955, AUPRC = 0.826, F1-score = 0.795, NHEK: AUROC = 0.977, AUPRC = 0.893, F1-score = 0.861</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">Protein-DNA binding sites prediction</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B53"><bold>53</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Patiyal et al. dataset (dataset 1) Xia et al. dataset (dataset 2)</bold></td>
<td valign="top" align="left"><bold>BERT</bold></td>
<td valign="top" align="left"><bold>CNN</bold></td>
<td valign="top" align="left"><bold>dataset 1: TE46 Sp = 0.835, Recall = 0.747, Precision = 0.306, F1-score = 0.434, MCC = 0.401, AUROC = 0.871 dataset 2: TE129 Sp = 0.955, Recall = 0.464, Precision = 0.396, F1-score = 0.427, MCC = 0.389, AUROC = 0.881</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B96">96</xref>)</td>
<td valign="top" align="left"><bold>690 ChIP-Seq Dataset</bold></td>
<td valign="top" align="left"><bold>BERT</bold></td>
<td valign="top" align="left">_</td>
<td valign="top" align="left"><bold>AUROC = 0.947</bold> <bold>&#x000B1;</bold> <bold>0.041, Acc = 0.880</bold> <bold>&#x000B1;</bold> <bold>0.062, Precision = 0.882</bold> <bold>&#x000B1;</bold> <bold>0.061, Recall = 0.880</bold> <bold>&#x000B1;</bold> <bold>0.062, F1-score = 0.880</bold> <bold>&#x000B1;</bold> <bold>0.062, MCC = 0.762</bold> <bold>&#x000B1;</bold> <bold>0.122</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B95"><bold>95</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Liu et al., Dataset, Tian et al., Dataset</bold></td>
<td valign="top" align="left"><bold>BERT</bold></td>
<td valign="top" align="left"><bold>RF</bold></td>
<td valign="top" align="left"><bold>dataset 1: Sp = 0.529, Precision = 0.106, Recall = 0.574, F1-score = 0.179, AUROC = 0.551, MCC = 0.025 dataset 2: Sp = 0.724, Precision = 0.119, Recall = 0.536, F1-score = 0.194, AUROC = 0.630, MCC = 0.067</bold></td>
</tr>
<tr>
<td valign="top" align="left">Regression</td>
<td valign="top" align="left">Transcription factor binding affinity prediction</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B52"><bold>52</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>1. Weirauch et al. dataset, 2. Jolma et al. dataset</bold></td>
<td valign="top" align="left"><bold>BERT</bold></td>
<td valign="top" align="left">_</td>
<td valign="top" align="left"><bold>1: Average PCC = 0.649 2: Average R2 = 0.977</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Bold values highlight top performers on unique datasets related to distinct tasks.</p>
</table-wrap-foot>
</table-wrap>
<p>For this goal, most commonly used representation learning approach is BERT. BERT is used with five different classifiers across six unique tasks. BERT with self-classifier is evaluated across all six unique tasks, whereas BERT with other four classifiers is evaluated on some of these six tasks. Specifically, BERT with CNN classifier is evaluated on four common tasks, namely, enhancer identification, promoter identification, protein-DNA binding site prediction, and transcription factor binding site prediction. BERT with DF, RF, and XGBoost is evaluated on one common task each including enhancer identification, protein-DNA binding site prediction, and promoter identification. Among all BERT-based predictive pipelines, BERT achieves the best performance with CNN classifier on two tasks, namely, transcription factor binding site prediction and protein-DNA binding site prediction. Second most common representation learning approach for this goal is Word2vec that is explored with seven different classifiers for four different tasks. Specifically, Word2Vec with CNN and CNN&#x0002B;BiLSTM is used for one task, with LSTM, TCN, and RF&#x0002B;CNN for one task, with BiGRU for one task, and with CNN&#x0002B;BiGRU for one task. Among all Word2vec-based predictive pipelines, Word2Vec achieves the best performance with CNN&#x0002B;BiGRU classifier on enhancer-promoter interaction prediction task. From two most common approaches, BERT with CNN and self-classifiers manages to yield top performance values as compared to Word2vec-based predictive pipelines. Beyond BERT and Word2vec, transformer-XL is used with CNN for two different tasks, ALBERT and ELECTRA are used with self-classifiers for a single task, and ULMFiT is used with CNN for a single task. In addition, potential of transformer is explored with CNN for one task and with a self-classifier for two tasks. FastText-based representation learning is used with MLP and SVM classifiers for a single task and potential of three graph embedding, namely, Node2Vec, SocDim, and GraRep, is explored with CatBoost classifier for a single task. Overall, among all approaches, ULMFiT manages to achieve best performance with CNN classifier on enhancer identification task. From all nine tasks, protein-DNA binding site prediction and protein-DNA binding affinity prediction have some room for improvement. Considering the promising performance trends for this goal, Word2vec potential can be explored with CNN&#x0002B;BiGRU classifier and ULMFiT potential can be explored with standalone CNN as well as ensemble CNN&#x0002B;BiGRU classifier to enhance the performance of under-performing tasks.</p>
<p>In addition, <xref ref-type="table" rid="T9">Table 9</xref> summarizes predictive models developed for seven different DNA sequence analysis tasks classified under the hood of gene analysis. For gene analysis goal, 12 unique representation learning methods are used that include Gapped K-mer Encoding, ESM-2, Flux Sampling, Node2Vec, FastText, OPA2Vec, Transformer, Laplacian eigenmaps &#x0002B; Locally linear Embedding &#x0002B; DeepWalk &#x0002B; Node2Vec, and GPT. Overall, eight unique classifiers, namely, GCNN, GNN, GAT, kNN, RF, SVM, CNN, and MLP, are used in different predictive pipelines. Most commonly used representation learning scheme for this goal is Node2vec followed by FastText. Node2vec is used with GCNN classifier for three different tasks and with GAT and MLP classifiers for two different tasks, whereas FastText is used with an ensemble and MLP classifiers for two different tasks. From most commonly used approaches, Node2Vec performs better in the majority of tasks and achieve top performance with GAT classifier as compared to FastText approach. Apart from Node2Vec and FastText, potential of ESM-2 along with a self-classifier and flux sampling with GNN is explored for a single task, transformer with CNN is evaluated for two different tasks, and GPT is used with a self-classifier for two tasks. In addition, OPA2Vec is used with RF for one task, Gapped K-mer encoding with GCN classifier is used for one task, and potential of Laplacian eigenmaps, Locally linear Embedding, DeepWalk, and Node2Vec is explored with RF for one task. In the holistic view, among all approaches, Gapped K-mer Encoding approach with GCNN classifier achieves the best performance for essential gene identification task. Among all seven distinct tasks, multi-label classification task, namely, gene function prediction provides a lot of room for improvement as the performance of its respective predictive pipeline based on Node2vec and GCNN classifier fall below 60%. Considering the promising performance of Gapped K-mer Encoding method, Gapped K-mer Encoding method and GCNN-based predictive pipeline can prove fruitful for various low performance tasks such as gene function prediction.</p>
<table-wrap position="float" id="T9">
<label>Table 9</label>
<caption><p>Gene analysis related 15 distinct DNA sequence analysis task predictive pipeline performance.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Task type</bold></th>
<th valign="top" align="left"><bold>Task name</bold></th>
<th valign="top" align="left"><bold>References</bold></th>
<th valign="top" align="left"><bold>dataset</bold></th>
<th valign="top" align="left"><bold>Representation learning</bold></th>
<th valign="top" align="left"><bold>Classifier</bold></th>
<th valign="top" align="left"><bold>Performance evaluation</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">Essential genes identification</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B333"><bold>333</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Hu et al. dataset (D. melanogaster, M. maripaludis</bold>, <italic><bold>H. sapiens, C. elegans)</bold></italic></td>
<td valign="top" align="left"><bold>Gapped K-mer encoding</bold></td>
<td valign="top" align="left"><bold>GCNN</bold></td>
<td valign="top" align="left"><bold>1. Sn = 0.8333, Sp = 0.9939, Acc = 0.9847, MCC = 0.8545, AUROC = 0.8283 2. Sn = 0.9052, Sp = 0.9304, Acc = 0.9221, MCC = 0.8265, AUROC = 0.8422 3. Sn = 0.9048, Sp = 0.9566, Acc = 0.9501, MCC = 0.7961, AUROC = 0.8655 4. Sn = 0.8362, Sp = 0.9368, Acc = 0.9242, MCC = 0.6983, AUROC = 0.7834</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B109">109</xref>)</td>
<td valign="top" align="left">Ma et al. dataset (<italic>S.cerevisiae, E.coli, H.sapiens, D.melanogaster</italic>)</td>
<td valign="top" align="left">ESM-2</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B104"><bold>104</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Campos et al. dataset</bold></td>
<td valign="top" align="left"><bold>Flux sampling</bold></td>
<td valign="top" align="left"><bold>GNN</bold></td>
<td valign="top" align="left"><bold>Acc = 0.871</bold> <bold>&#x000B1;</bold> <bold>0.012, Precision = 0.769</bold> <bold>&#x000B1;</bold> <bold>0.030, Recall = 0.718</bold> <bold>&#x000B1;</bold> <bold>0.037, F1-score = 0.743</bold> <bold>&#x000B1;</bold> <bold>0.023</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B108"><bold>108</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Ma et al. dataset (</bold><italic><bold>S. cerevisiae, E. coli</bold></italic>, <italic><bold>H. sapiens</bold></italic>, <italic><bold>D. melanogaster</bold></italic><bold>)</bold></td>
<td valign="top" align="left"><bold>Node2Vec</bold></td>
<td valign="top" align="left"><bold>GAT</bold></td>
<td valign="top" align="left"><bold>AUROC S. cerevisiae: (DIP: 76.57</bold> <bold>&#x000B1;</bold> <bold>0.74, BioGrid: 87.66</bold> <bold>&#x000B1;</bold> <bold>2.58, STRING: 90.13</bold> <bold>&#x000B1;</bold> <bold>1.08); E.coli: (DIP: 79.96</bold> <bold>&#x000B1;</bold> <bold>2.20, BioGrid: 92.35</bold> <bold>&#x000B1;</bold> <bold>1.15, STRING: 97.02</bold> <bold>&#x000B1;</bold> <bold>0.50); H.sapiens: (DIP: 75.61</bold> <bold>&#x000B1;</bold> <bold>0.90, BioGrid: 88.39</bold> <bold>&#x000B1;</bold> <bold>0.52, STRING: 90.95</bold> <bold>&#x000B1;</bold> <bold>0.54); D.melanogaster: (DIP: 32.90</bold> <bold>&#x000B1;</bold> <bold>5.34, BioGrid: 78.78</bold> <bold>&#x000B1;</bold> <bold>4.24, STRING: 77.82</bold> <bold>&#x000B1;</bold> <bold>1.95)</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B105">105</xref>)</td>
<td valign="top" align="left">Hu et al. dataset</td>
<td valign="top" align="left">FastText</td>
<td valign="top" align="left">kNN &#x0002B; RF &#x0002B; SVM &#x0002B; CNN</td>
<td valign="top" align="left">Sn = 60.2, Sp = 84.6, Acc = 76.3, MCC = 0.449, AUROC = 0.814</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B106">106</xref>)</td>
<td valign="top" align="left">Zhang et al. dataset</td>
<td valign="top" align="left">Node2Vec</td>
<td valign="top" align="left">MLP</td>
<td valign="top" align="left">AUROC = 94.15, Sp = 94.75, Average Precision = 90.64, Acc = 90.88</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B107">107</xref>)</td>
<td valign="top" align="left">Xiao et al. dataset</td>
<td valign="top" align="left">Node2Vec</td>
<td valign="top" align="left">MLP</td>
<td valign="top" align="left">Average AUROC = 95.17, AUPRC = 92.21, Acc = 91.59, F1-score = 78.71</td>
</tr>
<tr>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">Target gene classification</td>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B113">113</xref>)</td>
<td valign="top" align="left">Argoty et al. dataset</td>
<td valign="top" align="left">FastText</td>
<td valign="top" align="left">MLP</td>
<td valign="top" align="left">Precision = 0.99, Recall = 0.99</td>
</tr>
<tr>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">Disease genes prediction</td>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B110">110</xref>)</td>
<td valign="top" align="left">Nunes et al. dataset</td>
<td valign="top" align="left">OPA2Vec</td>
<td valign="top" align="left">RF</td>
<td valign="top" align="left">Median WAF score = 0.768</td>
</tr>
<tr>
<td valign="top" align="left">Multi-label classification</td>
<td valign="top" align="left">Gene function prediction</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B112"><bold>112</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>(1). 1,000 human gene set from gene ontology, (2). 100 omic gene set</bold></td>
<td valign="top" align="left"><bold>GPT</bold></td>
<td valign="top" align="left"><bold>_</bold></td>
<td valign="top" align="left"><bold>Semantic similarity = 0.50, Gene covered = 30</bold></td>
</tr>
<tr>
<td valign="top" align="left">Multi-class classification</td>
<td valign="top" align="left">Gene expression prediction</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B102"><bold>102</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Reddy et al. dataset (Jurkat, K-562, THP-1)</bold></td>
<td valign="top" align="left"><bold>Transformer</bold></td>
<td valign="top" align="left"><bold>CNN</bold></td>
<td valign="top" align="left"><bold>(1). Jurkat PCC = 0.6389</bold> <bold>&#x000B1;</bold> <bold>0.0036, SRCC = 0.5996</bold> <bold>&#x000B1;</bold> <bold>0.0093 (2). K-562 PCC = 0.6152</bold> <bold>&#x000B1;</bold> <bold>0.0082, SRCC = 0.6043</bold> <bold>&#x000B1;</bold> <bold>0.0045 (3). THP-1 PCC = 0.5672</bold> <bold>&#x000B1;</bold> <bold>0.0131, SRCC = 0.4742</bold> <bold>&#x000B1;</bold> <bold>0.0136</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B103">103</xref>)</td>
<td valign="top" align="left">Al Taweraqi et al. dataset</td>
<td valign="top" align="left">Laplacian eigenmaps &#x0002B; Locally linear Embedding &#x0002B; DeepWalk &#x0002B; Node2Vec</td>
<td valign="top" align="left">RF</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Multi-class classification</td>
<td valign="top" align="left">Pseudo-gene function prediction</td>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B111">111</xref>)</td>
<td valign="top" align="left">Fan et al. dataset (CC, MF, BP)</td>
<td valign="top" align="left">Node2Vec</td>
<td valign="top" align="left">GCN</td>
<td valign="top" align="left">CC: (AUPRC = 0.587 &#x000B1; 0.02, F1-score = 0.380 &#x000B1; 0.01) MF: (AUPRC = 0.463 &#x000B1; 0.02, F1-score = 0.319 &#x000B1; 0.01) BP: (AUPRC = 0.362 &#x000B1; 0.01, F1-score = 0.193 &#x000B1; 0.01)</td>
</tr>
<tr>
<td valign="top" align="left">Multi-class classification</td>
<td valign="top" align="left">Candidate gene prioritization &#x00026; Selection</td>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B114">114</xref>)</td>
<td valign="top" align="left">Toufiq et al. dataset</td>
<td valign="top" align="left">GPT</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Bold values highlight top performers on unique datasets related to distinct tasks.</p>
</table-wrap-foot>
</table-wrap>
<p>Moreover, <xref ref-type="table" rid="T10">Table 10</xref> summarizes eight DNA sequence analysis tasks classified across three unique biological goals, namely, DNA modification prediction, environmental and microbial genomics, and gene network analysis. For DNA modification prediction goal, across 4 DNA sequence analysis tasks, 27 predictors are developed. In the predictive pipelines, overall 10 unique representation learning methods are used which include PSeKNC, BERT, nucleotide physico-chemical properties and occurrence frequency based encoder, Transformer-XL, Word2Vec, One-hot encoding, FastText, ULMFIT, and BERT&#x0002B; ALBERT&#x0002B;XLNet&#x0002B; ELECTRA. Similarly, nine unique classifiers, namely, Structural Sparse Regularized Random Vector Functional Link Network, CatBoost, KNN, CNN, BiLSTM, CNN&#x0002B;BiLSTM, SVM, XGBoost, and FGM, are employed in different predictive pipelines. For DNA modification prediction goal, most commonly used representation learning approach is BERT. BERT is used with five different classifiers for all four tasks. Specifically, BERT with a self-classifier is evaluated for three tasks, namely, 4mC-methyl cytosine, 5mC-methyl cytosine, and DNA methylation modification prediction. BERT is also used with two other classifiers, namely, CatBoost, and FGM, for two common tasks, namely, 4mC-methyl cytosine modification prediction, and DNA methylation modification prediction. In addition, BERT is used with CNN and CNN&#x0002B;BiLSTM classifier for one task, namely, 6mA-methyl adenine modification prediction. Among all BERT-based predictive pipelines, BERT with a self-classifier showed top performance values for two tasks, namely, 5mC-methyl cytosine modification prediction, and DNA methylation modification prediction.</p>
<table-wrap position="float" id="T10">
<label>Table 10</label>
<caption><p>Distinct predictive pipeline performance related to DNA modification, environmental and microbial genomics tasks, and gene network analysis.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Task type</bold></th>
<th valign="top" align="left"><bold>Task name</bold></th>
<th valign="top" align="left"><bold>References</bold></th>
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="left"><bold>Representation learning</bold></th>
<th valign="top" align="left"><bold>Classifier</bold></th>
<th valign="top" align="left"><bold>Performance evaluation</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="7"><bold>Goal: DNA modification prediction</bold></td>
</tr>
<tr>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">4-Methylcytosine (4mc) modification prediction</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B134"><bold>134</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Chen et al. datasets (1-6)</bold></td>
<td valign="top" align="left"><bold>PseKNC</bold></td>
<td valign="top" align="left"><bold>Structural Sparse Regularized Random Vector Functional Link Network</bold></td>
<td valign="top" align="left"><bold>1. Acc = 0.8761, Sn = 0.8630, Sp = 0.8893, MCC = 0.7530 2. Acc = 0.8753, Sn = 0.8739, Sp = 0.8768, MCC = 0.7512 3. Acc = 0.8278, Sn = 0.8256, Sp = 0.8301, MCC = 0.6566 4. Acc = 0.9601, Sn = 0.8641, Sp = 0.9562, MCC = 0.9210 5. Acc = 0.9011, Sn = 0.8895, Sp = 0.9127, MCC = 0.8031 6. Acc = 0.9139, Sn = 0.9087, Sp = 0.9190, MCC = 0.8289</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B143"><bold>143</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Yang et al. dataset (</bold><italic><bold>A. thaliana, C.elegans, D. melanogaster, E. coli, G. pickeringii, G. subterraneous</bold></italic><bold>)</bold></td>
<td valign="top" align="left"><bold>BERT</bold></td>
<td valign="top" align="left"><bold>CatBoost</bold></td>
<td valign="top" align="left"><bold>A. thaliana: MCC = 0.6954, F1-score = 0.8521 C. elegans: MCC = 0.8326, F1-score = 0.9183 D. melanogaster: MCC = 0.7924, F1-score = 0.8986 E. coli: MCC = 0.9356, F1-score = 0.9679 G. pickeringii: MCC = 0.8904, F1-score = 0.9431 G. subterraneous: MCC = 0.8796, F1-score = 0.9363</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B351">351</xref>)</td>
<td valign="top" align="left">Chen et al. dataset</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">AUROC = 0.897, AUPRC = 0.907</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B135"><bold>135</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Khanal et al. dataset (</bold><italic><bold>C. elegans, D. melanogaster, A. thaliana, E. coli, G. subterraneus, G. pickeringi, F. vesca, R. chinensis</bold></italic><bold>)</bold></td>
<td valign="top" align="left"><bold>Number of codons, Number of occurrences of each codon, Proportion of each codon, Number of Nucleotides, Average number of Nucleotides per codon, Percentage of GC, Percentage of purines AG, Percentage of pyrimidines CT, Percentage of AT, Molecular weight of the sequence, Melting temperature, Proportion of Nucleotide DNA sequence, Protein sequence from DNA sequence, Number of amino acids, Percentage of amino acids, Aromaticity, Instability index, Isoelectric point, Molecular weight of Portion, Gravy</bold></td>
<td valign="top" align="left"><bold>KNN</bold></td>
<td valign="top" align="left"><italic><bold>C.elegans</bold></italic><bold>: Acc = 92.20, AUROC = 91.99, Precision = 89.47, Recall = 95.67</bold> <italic><bold>D. melanogaster</bold></italic><bold>: Acc = 92.79, AUROC = 92.80, Precision = 88.54, Recall = 98.30</bold> <italic><bold>A. thaliana</bold></italic><bold>: Acc = 90.27, AUROC = 90.28, Precision = 87.52, Recall = 93.93</bold> <italic><bold>E. coli</bold></italic><bold>: Acc = 91.02, AUROC = 91.03, Precision = 86.36, Recall = 97.43</bold> <italic><bold>G. subterraneus</bold></italic><bold>: Acc = 93.09, AUROC = 93.09, Precision = 91.48, Recall = 95.02</bold> <italic><bold>G. pickeringi</bold></italic><bold>: Acc = 90.78, AUROC = 90.79, Precision = 87.20, Recall = 95.61 F. vesca: Acc = 90.67, AUROC = 90.68, Precision = 85.31, Recall = 98.26</bold> <italic><bold>R. chinensis</bold></italic><bold>: Acc = 91.87, AUROC = 91.88, Precision = 87.35, Recall = 97.93</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B136"><bold>136</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Zulifiqar et al. dataset</bold></td>
<td valign="top" align="left"><bold>Word2Vec</bold></td>
<td valign="top" align="left"><bold>CNN</bold></td>
<td valign="top" align="left"><bold>Acc = 0.946, Sn = 0.938, Sp = 0.881, MCC = 0.778, AUROC = 0.989</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B50"><bold>50</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Clauwaert et al. dataset</bold></td>
<td valign="top" align="left"><bold>Transformer-XL</bold></td>
<td valign="top" align="left"><bold>CNN</bold></td>
<td valign="top" align="left"><bold>AUROC = 0.985</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B138">138</xref>)</td>
<td valign="top" align="left">Khanal et al. dataset (<italic>F. vesca, R. chinensis</italic>)</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">CNN</td>
<td valign="top" align="left">F. vesca: Sn = 0.8976, Sp = 0.8417, Acc = 0.8697, MCC = 0.7407, AUROC = 0.9400 R. chinensis: Sn = 0.8219, Sp = 0.8854, Acc = 0.8541, MCC = 0.7093, AUROC = 0.9370</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B137"><bold>137</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Zeng et al. dataset</bold></td>
<td valign="top" align="left"><bold>Word2Vec</bold></td>
<td valign="top" align="left"><bold>CNN</bold></td>
<td valign="top" align="left"><bold>Acc = 0.9321, MCC = 0.8559, Sn = 0.9508, Sp = 0.9161, AUROC = 0.9712</bold></td>
</tr>
<tr>
<td/>
<td valign="top" align="left">Methyladenine (6ma) modification Prediction</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B149"><bold>149</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>1</bold>. <italic><bold>A. thaliana</bold></italic> <bold>2</bold>. <italic><bold>D. melanogaster</bold></italic> <bold>3. 6mA-rice-Chen 4. 6mA-rice-Lv 5. Rosaceae</bold></td>
<td valign="top" align="left"><bold>One-hot Encoding</bold></td>
<td valign="top" align="left"><bold>BiLSTM</bold></td>
<td valign="top" align="left"><bold>1. Sn = 0.896, Sp = 0.935, Acc = 0.915, MCC = 0.831, AUROC = 0.967 2. Sn = 0.903, Sp = 0.952, Acc = 0.927, MCC = 0.855, AUROC = 0.963 3. Sn = 0.850, Sp = 0.917, Acc = 0.882, MCC = 0.763, AUROC = 0.947 4. Sn = 0.947, Sp = 0.930, Acc = 0.938, MCC = 0.877, AUROC = 0.976 5. Sn = 0.962, Sp = 0.961, Acc = 0.962, MCC = 0.924, AUROC = 0.990</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B150"><bold>150</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>1. Homo sapiens dataset (Train, Independent) 2. Mus musculus dataset</bold></td>
<td valign="top" align="left"><bold>Transformer</bold></td>
<td valign="top" align="left"><bold>CNN</bold></td>
<td valign="top" align="left"><bold>Homo sapiens (Train): Acc = 96.5 Homo sapiens (Independent): Acc = 93.75 Mus musculus: Acc = 96.86</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B144">144</xref>)</td>
<td valign="top" align="left">Lv et al. dataset (<italic>A. thaliana, C. elegans, C. equisetispolia, D. melanogaster, F. vesva, H. sapiens, R. chinensis, S. cerevisiae, T. thermophilus</italic>, Ts. SUP5-1, Xoc. BLS256)</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">CNN &#x0002B; BiLSTM</td>
<td valign="top" align="left"><italic>A. thaliana</italic>: AUROC = 0.927 C. elegans: AUROC = 0.962 <italic>C. equisetifpolia</italic>: AUROC = 0.800 <italic>D. melanogaster</italic>: AUROC = 0.967 <italic>F. vesca</italic>: AUROC = 0.976 <italic>H. sapiens</italic>: AUROC = 0.963 <italic>R. chinensis</italic>: AUROC = 0.876 <italic>S. cerevisiae</italic>: AUROC = 0.892 <italic>T. thermophilus</italic>: AUROC = 0.938 Ts. SUP5-1: AUROC = 0.829 Xoc. BLS256: AUROC = 0.937</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B281"><bold>281</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>DNA 6 mA dataset</bold></td>
<td valign="top" align="left"><bold>BERT</bold></td>
<td valign="top" align="left"><bold>CNN</bold></td>
<td valign="top" align="left"><bold>Cross-Validation Sn = 86.4, Sp = 68.8, Acc = 77.6, MCC = 0.651 Independent Sn = 84.3, Sp = 73.1, Acc = 79.3, MCC = 0.580</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B145">145</xref>)</td>
<td valign="top" align="left"><italic>A. thaliana</italic> dataset</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Acc = 0.9633, Sn = 0.9655, Sp = 0.9611, MCC = 0.9266</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B151">151</xref>)</td>
<td valign="top" align="left">1. Rice dataset 2. Mus musculus dataset</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">BiLSTM</td>
<td valign="top" align="left">Rice dataset: Sn = 95.66, Sp = 92.38, Acc = 94.02, MCC = 0.88, AUROC = 0.981 Mus musculus: Sn = 93.28, Sp = 100, Acc = 96.73, MCC = 0.93</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B234">234</xref>)</td>
<td valign="top" align="left">Chen et al. dataset</td>
<td valign="top" align="left">FastText</td>
<td valign="top" align="left">SVM</td>
<td valign="top" align="left">Sn = 86.48, Sp = 89.09, Acc = 87.78, MCC = 0.756</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">5-methylcytosine (5mc) modification prediction</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B282"><bold>282</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Wang et al. dataset</bold></td>
<td valign="top" align="left"><bold>BERT</bold></td>
<td valign="top" align="left">_</td>
<td valign="top" align="left"><bold>Acc = 0.932, MCC = 0.653, AUROC = 0.966</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B153">153</xref>)</td>
<td valign="top" align="left">Wang et al. dataset</td>
<td valign="top" align="left">FastText</td>
<td valign="top" align="left">XGBoost</td>
<td valign="top" align="left">Acc = 91.8, Sp = 92.0, Sn = 89.9, MCC = 0.626, AUROC = 0.962</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B152"><bold>152</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Stanojevic et al. datasets (GM24385, NA12878, NA19240, HIESc, K562, HX1)</bold></td>
<td valign="top" align="left"><bold>Transformer</bold></td>
<td valign="top" align="left">_</td>
<td valign="top" align="left"><bold>GM24385: Acc = 0.9988, Precision = 0.9990, Recall = 0.9988, FPR = 0.0011, F1-score = 0.9989 NA12878: Acc = 0.9936, Precision = 0.9921, Recall = 0.9953, FPR = 0.0081, F1-score = 0.9937 NA19240: Acc = 0.9872, Precision = 0.9678, Recall = 0.9765, FPR = 0.0096, F1-score = 0.9721 H1ESc: Acc = 0.9938, Precision = 0.9995, Recall = 0.9935, FPR = 0.0040, F1-score = 0.9965 K562: Acc = 0.9972, Precision = 0.9512, Recall = 0.9964, FPR = 0.0028, F1-score = 0.9733 HX1: Acc = 0.9950, Precision = 0.9993, Recall = 0.9951, FPR = 0.0057, F1-score = 0.9972</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Methylation modification prediction</td>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B159">159</xref>)</td>
<td valign="top" align="left">Jeong et al. dataset</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Precision = 0.98</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B351">351</xref>)</td>
<td valign="top" align="left">Yu et al. dataset</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">AUROC = 0.897, AUPRC = 0.907</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B155">155</xref>)</td>
<td valign="top" align="left">Lv et al. datasets (5hmc: <italic>M. musculus, H. sapiens</italic>, 4mc: <italic>C. equisetifolia, F. vesca, S. cerevisiae, Tolypocladium</italic>, 6mA: <italic>A. thaliana, C. elegans, C. equisetifolia, D. melanogaster, F. vesca, H. sapiens, R. chinensis, S. cerevisiae, T. thermophile, Tolypocladium</italic>, XocBLS256)</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">5hmC_<italic>M. sapiens</italic>: Acc = 0.949, AUROC = 0.967, MCC = 0.900 5hmC_<italic>M. musculus</italic>: Acc = 0.968, AUROC = 0.981, MCC = 0.936 4mC_<italic>C. equisetifolia</italic>: Acc = 0.853, AUROC = 0.896, MCC = 0.706 4mC_<italic>F. vesca</italic>: Acc = 0.853, AUROC = 0.928, MCC = 0.706 4mC_<italic>S. cerevisiae</italic>: Acc = 0.710, AUROC = 0.776, MCC = 0.423 4mC_Tolypocladium: Acc = 0.743, AUROC = 0.819, MCC = 0.487 6mA_<italic>A. thaliana</italic>: Acc = 0.861, AUROC = 0.934, MCC = 0.722 6mA_<italic>C. elegans</italic>: Acc = 0.909, AUROC = 0.966, MCC = 0.818 6mA_<italic>C. equisetifolia</italic>: Acc = 0.745, AUROC = 0.816, MCC = 0.494 6mA_<italic>D. melanogaster</italic>: Acc = 0.923, AUROC = 0.971, MCC = 0.846 6mA_<italic>F. vesca</italic>: Acc = 0.939, AUROC = 0.981, MCC = 0.878 6mA_<italic>H. sapiens</italic>: Acc = 0.907, AUROC = 0.969, MCC = 0.815 6mA_<italic>R. chinensis</italic>: Acc = 0.818, AUROC = 0.881, MCC = 0.635 6mA_<italic>S. cerevisiae</italic>: Acc = 0.827, AUROC = 0.905, MCC = 0.654 6mA_<italic>T. thermophile</italic>: Acc = 0.882, AUROC = 0.944, MCC = 0.772 6mA_<italic>Tolypocladium</italic>: Acc = 0.768, AUROC = 0.845, MCC = 0.538 6mA_XocBLS256: Acc = 0.877, AUROC = 0.949, MCC = 0.756</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B158">158</xref>)</td>
<td valign="top" align="left">CCLE dataset</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Sn = 0.831, Sp = 0.991, Acc = 0.978, MCC = 0.871, AUROC = 0.989</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B156"><bold>156</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Lv et al. datasets (5hmc:</bold> <italic><bold>M. musculus</bold></italic>, <italic><bold>H. sapiens</bold></italic><bold>, 4mc:</bold> <italic><bold>C. equisetifolia, F. vesca, S. cerevisiae, Tolypocladium</bold></italic><bold>, 6mA:</bold> <italic><bold>A. thaliana, C. elegans, C. equisetifolia, D. melanogaster, F. vesca</bold></italic>, <italic><bold>H. sapiens</bold></italic>, <italic><bold>R. chinensis, S. cerevisiae</bold>,</italic> <italic><bold>T. thermophile</bold></italic>, Tolypocladium, XocBLS256</td>
<td valign="top" align="left"><bold>BERT &#x0002B; ALBERT &#x0002B; XLNet &#x0002B; ELECTRA</bold></td>
<td valign="top" align="left">_</td>
<td valign="top" align="left"><bold>(6mA)</bold> <italic><bold>T. thermophile</bold></italic><bold>: AUROC = 0.9467, Acc = 0.8840, F1-score = 0.8923, Recall = 0.9611, AUPRC = 0.9321 A. thaliana: AUROC = 0.9378, Acc = 0.8649, F1-score = 0.8615, Recall = 0.8401, AUPRC = 0.9423</bold> <italic><bold>H. sapiens</bold></italic><bold>: AUROC = 0.9687, Acc = 0.9077, F1-score = 0.9068, Recall = 0.8975</bold>, <bold>AUPRC = 0.9721 Xoc. BLS256: AUROC = 0.9446, Acc = 0.8742, F1-score = 0.8712, Recall = 0.8511, AUPRC = 0.9421</bold> <italic><bold>D. melanogaster</bold></italic><bold>: AUROC = 0.9730, Acc = 0.9276, F1-score = 0.9275, Recall = 0.9258, AUPRC = 0.9761 C. elegans: AUROC = 0.9684, Acc = 0.9131, F1-score = 0.9138, Recall = 0.9219, AUPRC = 0.9674 C. equisetifolia: AUROC = 0.8350, Acc = 0.7590, F1-score = 0.7481, Recall = 0.7158, AUPRC = 0.8492 S. cerevisiae: AUROC = 0.9082, Acc = 0.8325, F1-score = 0.8233, Recall = 0.7802, AUPRC = 0.9198 Tolypocladium: AUROC = 0.8669, Acc = 0.7895, F1-score = 0.7824, Recall = 0.7567, AUPRC = 0.8730 F. vesca: AUROC = 0.9821, Acc = 0.9407, F1-score = 0.9403, Recall = 0.9336, AUPRC = 0.9831 R. chinensis: AUROC = 0.9654, Acc = 0.9164, F1-score = 0.9167, Recall = 0.9197, AUPRC = 0.9691 (4mC) C. equisetifolia: AUROC = 0.9108, Acc = 0.8333, F1-score = 0.8272, Recall = 0.7978, AUPRC = 0.9221 F. vesca: AUROC = 0.9256, Acc = 0.8522, F1-score = 0.8554, Recall = 0.8739, AUPRC = 0.9144 S. cerevisiae: AUROC = 0.8064, Acc = 0.7376, F1-score = 0.7253, Recall = 0.6926, AUPRC = 0.8215 Tolypocladium: AUROC = 0.8149, Acc = 0.7380, F1-score = 0.7285, Recall = 0.7031, AUPRC = 0.80889 (5hmC)</bold> <italic><bold>M. musculus</bold></italic><bold>: AUROC = 0.9817, Acc = 0.9649, F1-score = 0.9651, Recall = 0.9685, AUPRC = 0.9782</bold> <italic><bold>H. sapiens</bold></italic><bold>: AUROC = 0.9680, Acc = 0.9484, F1-score = 0.9500, Recall = 0.9787, AUPRC = 0.9485</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B157">157</xref>)</td>
<td valign="top" align="left">Lv et al. datasets (5hmc: <italic>M. musculus, H. sapiens</italic>, 4mc: <italic>C. equisetifolia, F. vesca, S. cerevisiae, Tolypocladium</italic>, 6mA: <italic>A. thaliana, C. elegans, C. equisetifolia, D. melanogaster, F. vesca, H. sapiens</italic>, R. chinensis, S. cerevisiae, T. thermophile, Tolypocladium, XocBLS256)</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">FGM</td>
<td valign="top" align="left">5hmC_<italic>H. sapiens</italic>: Acc = 0.9501, Sn = 0.9838, Sp = 0.9164, AUROC = 0.9501, MCC = 0.9022 5hmC_<italic>M. musculus</italic>: Acc = 0.9679, Sn = 0.969, Sp = 0.9668, AUROC = 0.9679, MCC = 0.9358 4mC_C. equisetifolia: Acc = 0.8579, Sn = 0.8743, Sp = 0.8415, AUROC = 0.8579, MCC = 0.7162 4mC_F. vesca: Acc = 0.8524, Sn = 0.8535, Sp = 0.8512, AUROC = 0.8524, MCC = 0.7047 4mC_S. cerevisiae: Acc = 0.723, Sn = 0.6876, Sp = 0.7583, AUROC = 0.723, MCC = 0.447 4mC_Tolypocladium: Acc = 0.7434, Sn = 0.7385, Sp = 0.7483, AUROC = 0.7434, MCC = 0.4868 6mA_A. thaliana: Acc = 0.8603, Sn = 0.8264, Sp = 0.8942, AUROC = 0.8603, MCC = 0.7223 6mA_C. elegans: Acc = 0.9138, Sn = 0.9256, Sp = 0.902, AUROC = 0.9138, MCC = 0.8279 6mA_C. equisetifolia: Acc = 0.7399, Sn = 0.6713, Sp = 0.8084, AUROC = 0.7399, MCC = 0.4843 6mA_<italic>D. melanogaster</italic>: Acc = 0.9228, Sn = 0.9301, Sp = 0.9155, AUROC = 0.9228, MCC = 0.8457 6mA_F. vesca: Acc = 0.9413, Sn = 0.9452, Sp = 0.9375, AUROC = 0.9413, MCC = 0.8827 6mA_R. chinensis: Acc = 0.8629, Sn = 0.8328, Sp = 0.893, AUROC = 0.8629, MCC = 0.7271 6mA_S. cerevisiae: Acc = 0.8278, Sn = 0.7966, Sp = 0.859, AUROC = 0.8278, MCC = 0.6569 6mA_T. thermophile: Acc = 0.8804, Sn = 0.9442, Sp = 0.8167, AUROC = 0.8804, MCC = 0.7671 6mA_Tolypocladium: Acc = 0.7771, Sn = 0.7649, Sp = 0.7892, AUROC = 0.7771, MCC = 0.5543 6mA_Xoc BLS256: Acc = 0.8817, Sn = 0.8808, Sp = 0.8827, AUROC = 0.8817, MCC = 0.7634</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B250">250</xref>)</td>
<td valign="top" align="left">DNAm dataset (Brain, Blood, Buccal, Saliva)</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">SRCC: (Brain: Mean = 0.82, SD = 0.004) (Blood: Mean = 0.78, SD = 0.005) (Buccal: Mean = 0.79, SD = 0.007) (Saliva: Mean = 0.79, SD = 0.010) Mean squared error: (Brain: Mean = 0.030, SD = 0.0026) (Blood: Mean = 0.043, SD = 0.0023) (Buccal: Mean = 0.049, SD = 0.0080) (Saliva: Mean = 0.040, SD = 0.055)</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B146">146</xref>)</td>
<td valign="top" align="left">1. Maize 5mC dataset 2. Nipponbare 5mC dataset 3. 6mA Chen dataset 4. 6mA Lv dataset</td>
<td valign="top" align="left">ULMFiT</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">1. Maize 5mC Acc = 0.9524, AUNP = 0.97, AUNU = 0.95, Macro precision = 0.9378, Min precision = 0.9524, Macro Sn = 0.9188, Micro Sn = 0.9524, Macro F1-score = 0.9269, Min F1-score = 0.9524 2. Nipponbare 5mC Acc = 0.8106, AUNP = 0.88, AUNU = 0.88, Macro precision = 80.74, Min precision = 81.93, Macro Sn = 81.93, Micro Sn = 81.93, Macro F1-score = 80.32, Min F1-score = 81.93 3. 6mA Chen Sn = 0.9303, Sp = 0.9225, Acc = 0.9265, MCC = 0.85, AUROC = 0.93 4. 6mA Lv Sn = 0.9605, Sp = 0.9248, Acc = 0.9426, MCC = 0.89, AUROC = 0.94</td>
</tr>
<tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B154">154</xref>)</td>
<td valign="top" align="left">Lv et al. datasets (5hmc: <italic>M. musculus, H. sapiens</italic>, 4mc: <italic>C. equisetifolia, F. vesca, S. cerevisiae, Tolypocladium</italic>, 6mA: <italic>A. thaliana, C. elegans, C. equisetifolia, D. melanogaster, F. vesca, H. sapiens, R. chinensis, S. cerevisiae, T. thermophile, Tolypocladium</italic>, XocBLS256)</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">5hmC_<italic>H. sapiens</italic>: Acc = 94.92, Sn = 98.63, Sp = 91.21, MCC = 90.09, AUROC = 95.53, F1-score = 95.1 5hmC_<italic>M. musculus</italic>: Acc = 96.85, Sn = 97.06, Sp = 96.63, MCC = 93.69, AUROC = 97.57, F1-score = 96.85 4mC_<italic>C. equisetifolia</italic>: Acc = 82.51, Sn = 79.23, Sp = 85.79, MCC = 65.17, AUROC = 85.55, F1-score = 81.92 4mC_F. vesca: Acc = 84.2, Sn = 85.2, Sp = 83.21, MCC = 68.42, AUROC = 90.7, F1-score = 84.36 4mC_<italic>S. cerevisiae</italic>: Acc = 70.27, Sn = 66.94, Sp = 73.61, MCC = 40.64, AUROC = 75.37, F1-score = 69.25 4mC_<italic>Tolypocladium</italic>: Acc = 73.83, Sn = 72.16, Sp = 75.49, MCC = 47.68, AUROC = 80.57, F1-score = 73.39 6mA_A. thaliana: Acc = 85.38, Sn = 82.33, Sp = 88.42, MCC = 70.88, AUROC = 91.84, F1-score = 84.92 6mA_<italic>C. elegans</italic>: Acc = 89.03, Sn = 88.17, Sp = 89.9, MCC = 78.08, AUROC = 94.33, F1-score = 88.94 6mA_<italic>C. equisetifolia</italic>: Acc = 73.28, Sn = 68.91, Sp = 77.65, MCC = 46.73, AUROC = 79.02, F1-score = 72.06 6mA_<italic>D. melanogaster</italic>: Acc = 91.22, Sn = 90.38, Sp = 92.05, MCC = 82.44, AUROC = 95.44, F1-score = 91.14 6mA_<italic>F. vesca</italic>: Acc = 92.68, Sn = 92.33, Sp = 93.04, MCC = 82.44, AUROC = 95.44, F1-score = 92.66 6mA_<italic>H. sapiens</italic>: Acc = 89.8, Sn = 89.4, Sp = 90.2, MCC = 79.6, AUROC = 95.1, F1-score = 89.76 6mA_<italic>R. chinensis</italic>: Acc = 82.61, Sn = 80.94, Sp = 84.28, MCC = 65.25, AUROC = 87.89, F1-score = 82.31 6mA_<italic>S. cerevisiae</italic>: Acc = 80.11, Sn = 72. 37, Sp = 87.85, MCC = 60.96, AUROC = 87.09, F1-score = 78.44 6mA_<italic>T. thermophile</italic>: Acc = 87.4, Sn = 93.34, Sp = 81.54, MCC = 75.4, AUROC = 93.1, F1-score = 88.14 6mA_<italic>Tolypocladium</italic>: Acc = 77.38, Sn = 71.76, Sp = 83.01, MCC = 55.12, AUROC = 83.61, F1-score = 76.04 6mA_Xoc BLS256: Acc = 86.94, Sn = 88.9, Sp = 84.92, MCC = 73.94, AUROC = 92.61, F1-score = 87.2</td>
</tr>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="7"><bold>Goal: environmental and microbial genomics tasks</bold></td>
</tr>
<tr>
<td valign="top" align="left">Multi-class classification</td>
<td valign="top" align="left">Nitrogen cycle prediction</td>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B27">27</xref>)</td>
<td valign="top" align="left">NCycDB</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Macro F1-score = 99.5, Weighted F1-score = 99.2</td>
</tr>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="7"><bold>Goal: gene network analysis</bold></td>
</tr>
<tr>
<td valign="top" align="left">Multi-class classification</td>
<td valign="top" align="left">Gene taxonomy classification</td>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B121">121</xref>)</td>
<td valign="top" align="left">Verma et al. dataset</td>
<td valign="top" align="left">FastText</td>
<td valign="top" align="left">MLP</td>
<td valign="top" align="left">Macro F1-score = 0.92 &#x000B1; 0.0054</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B123"><bold>123</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Mock et al. dataset (Superkingdom, Phylum)</bold></td>
<td valign="top" align="left"><bold>BERT</bold></td>
<td valign="top" align="left">_</td>
<td valign="top" align="left"><bold>Superkingdom: Acc = 94.78 Phylum: Acc = 85.55</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B122">122</xref>)</td>
<td valign="top" align="left">ActinoMock dataset (LSH, Decimal, FNV)</td>
<td valign="top" align="left">LSH&#x0002B;FastText</td>
<td valign="top" align="left">MLP</td>
<td valign="top" align="left">LSH: Acc = 1.00, Precision = 1.00, Recall = 1.00, F1-score = 0.99 Decimal: Acc = 0.93, Precision = 0.94, Recall = 0.93, F1-score = 0.91 FNV: Acc = 0.937, Precision = 0.94, Recall = 0.94, F1-score = 0.92</td>
</tr>
<tr>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">Gene network reconstruction</td>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B127">127</xref>)</td>
<td valign="top" align="left">DREAM4 10, DREAM4 100, E. coli cold</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">LightGBM</td>
<td valign="top" align="left">DREAM4 10: AUROC = 0.956, AUPRC = 0.891 DREAM4 100: AUROC = 0.909, AUPRC = 0.445 E. coli cold: AUROC = 0.602, AUPRC = 0.030</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B126">126</xref>)</td>
<td valign="top" align="left">Pio et al. dataset</td>
<td valign="top" align="left">Metabolic feature encoding</td>
<td valign="top" align="left">Clustering</td>
<td valign="top" align="left">_</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B128">128</xref>)</td>
<td valign="top" align="left">Ceci et al. datasets</td>
<td valign="top" align="left">Node2Vec</td>
<td valign="top" align="left">PCT</td>
<td valign="top" align="left">_</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Bold values highlight top performers on unique datasets related to distinct tasks.</p>
</table-wrap-foot>
</table-wrap>
<p>Second most commonly used representation learning approach in this goal is Word2vec which is explored with two unique classifiers for two different tasks. Specifically, Word2vec with CNN classifier is used for one task, namely, 4mC-cytosine modification prediction, and with BiLSTM for one task, namely, 6mA-methyl adenine modification prediction. From most common approaches, BERT manages to achieve best performance with a self-classifier as compared to Word2vec-based predictive pipeline. Apart from BERT and Word2vec, Transformer with a self-classifier is used for three tasks, namely, 5mC-methycyctosine modification prediction, 6mA-methyl adenine modification prediction, and DNA methylation modification prediction. Transformer is also used with CNN classifier on one of the common task, namely, 6mA-methyl adenine modification prediction. Furthermore, potential of transformer-XL is explored with CNN classifier for one task, ULMFiT and hybrid encoding scheme (BERT&#x0002B;ALBERT&#x0002B;XLNet&#x0002B;ELECTRA) with self-classifier for one task, FastText with SVM classifier for one task and FastText with XGBoost classifier for one task. Overall, among all approaches, PseKNC encoding approach with structural sparse regularized random vector functional link network classifier manages to achieve best predictive performance on 4mC-methylcytosine modification prediction. Among all four tasks, DNA methylation modification prediction and 5mC-methyl cytosine modification prediction have some room for improvement. Building on the performance trends of predictive pipelines developed for different tasks of this goal, potential of BERT or PseKNC representation learning approach with structural sparse regularized random vector functional link network classifier can be explored to enhance the performance of under-performing tasks.</p>
<p>For environmental and microbial genomics goal, only potential of BERT representation learning is explored with a self-classifier. However, the potential of neural word embeddings and domain specific encoders based predictive pipelines remains unexplored.</p>
<p>For gene network analysis goal, across two different tasks, four unique representation learning approaches, namely, FastText, Node2vec, BERT, and metabolic encoding, along with four unique classifiers, namely, MLP, LightGBM, Clustering algorithm, and PCT, are used by six predictors. FastText representation learning is most commonly used among all approaches. Specifically, FastText along with MLP classifier is used for gene taxonomy classification task. Second most common representation learning approach is BERT that is used with a self-classifier for same gene taxonomy classification task. Apart from this, Node2Vec representation is explored with PCT classifier and metabolic feature encoding is explored with clustering algorithm for gene network reconstruction task. Among all approaches, FastText and MLP classifier-based predictive pipeline manages to achieve best performance for gene taxonomy classification task. Among all tasks of this goal, gene taxonomy classification offers some room for improvement. Building on promising performance achieved by contemporary language models for different sequence analysis tasks, hierarchical graph transformer and sophisticated machine or deep learning-based ensemble classifier can further raise the predictive performance on gene taxonomy classification task. Furthermore, advanced graph-based representation learning methods, GraRep, HOPE, and LINE with deep classifiers can also potentially raise the predictive performance on gene taxonomy classification task.</p>
<p>In addition, <xref ref-type="table" rid="T11">Table 11</xref> provides an overview of six DNA sequence analysis tasks classified under the goal of DNA functional analysis. For this goal, four unique representation learning methods, namely, Word2Vec, Transformer, BERT, and FastText, are used in conjunction with three different predictors, namely, LogR, SVM, and cosine similarity. Among all four representation learning methods, Transformer is most commonly used followed by Word2vec and BERT. Transformer is used in three different tasks with self-classifier, Word2vec, and BERT are used with LogR and self-classifier in two different tasks. In addition, potential of FastText is explored with SVM for 1 task. Among all representation learning methods, Transformer with self-classifier manages to achieve top performance for tumor type prediction. Among all six tasks, disease risk estimation task offers a room for improvement as its respective BERT and self-classifier-based predictive pipeline accuracy falls approximately 56%. Hybrid approaches combining the powers of Transformer, BERT, and Word2vec with sophisticated machine learning classifier such as deep forest or deep learning classifiers such as CNN and CNN&#x0002B;BiGRU can potentially enhance the performance on under-performing tasks. Furthermore, except for two tasks, namely, species classification and functional prioritization, of non-coding variants, all other four tasks are evaluated on a single benchmark dataset. Considering deep learning models require huge amount of data to achieve promising performance, development, and utilization of more datasets in model building and validation can also prove fruitful for enhancing the predictive performance.</p>
<table-wrap position="float" id="T11">
<label>Table 11</label>
<caption><p>DNA functional analysis task predictive pipeline performance.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Task type</bold></th>
<th valign="top" align="left"><bold>Task name</bold></th>
<th valign="top" align="left"><bold>References</bold></th>
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="left"><bold>Representation learning</bold></th>
<th valign="top" align="left"><bold>Classifier</bold></th>
<th valign="top" align="left"><bold>Performance evaluation</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">Conserved non-coding elements classification</td>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B163">163</xref>)</td>
<td valign="top" align="left">Polychronopoulos et al. dataset (D1, D2, D3)</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">LogR</td>
<td valign="top" align="left">D1: F1-score = 83.0, D2: F1-score = 85.5, D3: F1-score = 77.4</td>
</tr>
<tr>
<td valign="top" align="left">Multi-class classification</td>
<td valign="top" align="left">Functional prioritization of non-coding variant</td>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B34">34</xref>)</td>
<td valign="top" align="left">Yang et al. dataset (LOGO-919, LOGO-2002, LOGO-3357)</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">_</td>
</tr>
<tr>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">Exon and intron region classification</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B164"><bold>164</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Akalin et al. dataset</bold></td>
<td valign="top" align="left"><bold>BERT</bold></td>
<td valign="top" align="left">_</td>
<td valign="top" align="left"><bold>Precision = 100, Sn = 75, Sp = 100, Acc = 88.88, F1-score = 85.71</bold></td>
</tr>
<tr>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">Recombination spots identification</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B165"><bold>165</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Liu et al. dataset</bold></td>
<td valign="top" align="left"><bold>FastText</bold></td>
<td valign="top" align="left"><bold>SVM</bold></td>
<td valign="top" align="left"><bold>Sn = 90, Sp = 94.76, Acc = 92.6, MCC = 0.851</bold></td>
</tr>
<tr>
<td valign="top" align="left">Multi-class classification</td>
<td valign="top" align="left">Species classification</td>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B44">44</xref>)</td>
<td valign="top" align="left">1. Mouse enhancers dataset 2. Coding vs. intergenomic dataset 3. Human vs. worm dataset 4. Human enhancers cohn dataset 5. Human enhancers ensembl dataset 6. Human regulatory dataset 7. Human nontata promoter dataset 8. Human OCR ensembl dataset</td>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">Mouse enhancers: Acc = 85.1 Coding vs. Intergenomic: Acc = 91.3 human vs. worm: Acc = 96.6 human enhancers cohn: Acc = 74.2 human enhancers ensembl: Acc = 89.2 human regulatory: Acc = 93.8 human nontata promoter: Acc = 96.6 human OCR ensembl: Acc = 80.9</td>
</tr>
<tr>
<td valign="top" align="left">Interaction</td>
<td valign="top" align="left">Prediction of context-specific functional impact of genetic variants</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B36"><bold>36</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>eQTLs dataset</bold></td>
<td valign="top" align="left"><bold>Transformer</bold></td>
<td valign="top" align="left">_</td>
<td valign="top" align="left"><bold>AUPRC = 0.922</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Bold values highlight top performers on unique datasets related to distinct tasks.</p>
</table-wrap-foot>
</table-wrap>
<p>Finally, <xref ref-type="table" rid="T12">Table 12</xref> summarizes predictive models developed for seven unique DNA sequence analysis tasks categorized under the goal of disease analysis. For this goal, seven unique representation learning methods, namely, Node2Vec, Graph2Vec, BERT, Graph Embedding, SDNE, Word2Vec, and Transformer, and five predictors, namely, MLP, RF, cosine similarity, clustering, and BERT self-classifier, are used in different tasks. Among all representation learning approaches, BERT and Word2vec are most commonly used. BERT is used with self-classifier on two different tasks, and word2vec is used with cosine similarity for a multi-class classification task, namely, mutation susceptibility analysis and with clustering algorithm for an only clustering task, namely, phylogenetic analysis. Apart from BERT and Word2vec, other representation learning methods Node2vec&#x0002B;Graph2vec, Graph Embedding, SDNE&#x0002B;Word2vec, and Transformer are used on one classification task each with MLP and self-classifiers. Overall, among all approaches, Transformer with self-classifier-based predictive pipelines manages to achieve best performance on tumor type prediction task. Among all seven tasks, disease risk estimation task offers a lot of room for improvement as its respective BERT with self-classifier-based predictive pipeline performance falls approximately 56%. Taking the transformer performance trends into account, latest sophisticated language models such as hierarchical graph transformer, ELECTRA, and GPT-4 along with ensemble machine or deep learning predictors can achieve significance performance rise in under-performing classification and clustering tasks.</p>
<table-wrap position="float" id="T12">
<label>Table 12</label>
<caption><p>Disease analysis task predictive pipeline performance.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Task type</bold></th>
<th valign="top" align="left"><bold>Task name</bold></th>
<th valign="top" align="left"><bold>References</bold></th>
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="left"><bold>Representation learning</bold></th>
<th valign="top" align="left"><bold>Classifier</bold></th>
<th valign="top" align="left"><bold>Performance evaluation</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Binary classification</td>
<td valign="top" align="left">Pathogen signatures identification</td>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B171">171</xref>)</td>
<td valign="top" align="left">DS500 dataset, DS5000 dataset</td>
<td valign="top" align="left">Node2Vec&#x0002B;Graph2Vec</td>
<td valign="top" align="left">MLP</td>
<td valign="top" align="left">1. Acc = 73.49, 2. Acc = 89.7</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Disease risks estimation</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B90"><bold>90</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>HSCR-RET, HSCR-RET-Long</bold></td>
<td valign="top" align="left"><bold>BERT</bold></td>
<td valign="top" align="left">_</td>
<td valign="top" align="left"><bold>HSCR-RET: Precision = 0.770, Recall = 0.519, Acc = 0.562; HSCR-RET-Long: Precision = 0.768, Recall = 0.513, Acc = 0.541</bold></td>
</tr>
<tr>
<td valign="top" align="left">Interaction</td>
<td valign="top" align="left">Phage-host interactions prediction</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B175"><bold>175</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Qiu et al. dataset</bold></td>
<td valign="top" align="left">_</td>
<td valign="top" align="left"><bold>RF</bold></td>
<td valign="top" align="left"><bold>Acc = 0.801, Recall = 0.801, Sp = 0.801, Precision = 0.803, F1-score = 0.801, AUROC = 0.801</bold></td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B176">176</xref>)</td>
<td valign="top" align="left">Wang et al. dataset</td>
<td valign="top" align="left">Graph Embedding</td>
<td valign="top" align="left">MLP</td>
<td valign="top" align="left">AUROC = 0.88317</td>
</tr>
 <tr>
<td/>
<td/>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B177">177</xref>)</td>
<td valign="top" align="left">ESKAPE dataset</td>
<td valign="top" align="left">SDNE&#x0002B;Word2Vec</td>
<td valign="top" align="left"><bold>MLP</bold></td>
<td valign="top" align="left">Acc = 86.65 &#x000B1; 1.55, Sn = 88.40 &#x000B1; 1.81, Sp = 84.91 &#x000B1; 1.96, Precision = 85.43 &#x000B1; 1.74, F1-score = 86.88 &#x000B1; 1.53, AUROC = 0.9208 &#x000B1; 0.0119</td>
</tr>
<tr>
<td valign="top" align="left">Multi-class classification</td>
<td valign="top" align="left">Mutation susceptibility analysis</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B173"><bold>173</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>Yilmaz et al. dataset (Human, Mouse)</bold></td>
<td valign="top" align="left"><bold>Word2Vec</bold></td>
<td valign="top" align="left"><bold>Cosine similarity</bold></td>
<td valign="top" align="left"><bold>Human data: Acc = 0.7974, mouse data: Acc = 0.8322</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Tumor type prediction</td>
<td valign="top" align="left"><bold>(</bold><xref ref-type="bibr" rid="B180"><bold>180</bold></xref><bold>)</bold></td>
<td valign="top" align="left"><bold>TCGA pan-cancer dataset</bold></td>
<td valign="top" align="left"><bold>Transformer</bold></td>
<td valign="top" align="left">_</td>
<td valign="top" align="left"><bold>Acc = 98.4, Precision = 98.50, Recall = 98.4, F1-score = 98.37</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Pathogenicity potential assessment</td>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B27">27</xref>)</td>
<td valign="top" align="left">E-K12, CARD-A, CARD-D, CARD-R, VFDB, ENZYME, PATRIC, NCycDB</td>
<td valign="top" align="left">BERT</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">E-K12: Macro F1-score = 61.8, Weighted F1-score = 65.4; CARD-A AMR: Macro F1-score = 78.6, Weighted F1-score = 90.1; CARD-D: Macro F1-score = 57.4, Weighted F1-score = 85.2; CARD-R: Macro F1-score = 69.4, Weighted F1-score = 91.4; VFDB: Macro F1-score = 75.7, Weighted F1-score = 90.2; ENZYME: Macro F1-score = 99.1, Weighted F1-score = 98.8; PATRIC: Macro F1-score = 99.3, Weighted F1-score = 99.0; NCycDB: Macro F1-score = 99.5, Weighted F1-score = 99.2</td>
</tr>
<tr>
<td valign="top" align="left">Clustering</td>
<td valign="top" align="left">Phylogenetic analysis</td>
<td valign="top" align="left">(<xref ref-type="bibr" rid="B21">21</xref>)</td>
<td valign="top" align="left">Ren et al. dataset</td>
<td valign="top" align="left">Word2Vec</td>
<td valign="top" align="left">Clustering</td>
<td valign="top" align="left">Acc = 0.84</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Bold values highlight top performers on unique datasets related to distinct tasks.</p>
</table-wrap-foot>
</table-wrap>
<p>In a nutshell, a comprehensive analysis of state-of-the-art predictive pipelines developed using word embeddings, language models, and nucleotide compositional and positional information-based encoders reveals interesting trends. From 44 DNA sequence analysis tasks classified under the hood of 8 major biological goals, 24 tasks belong to binary classification, 4 belong to interaction prediction, 11 belong to multi-class classification, only 3 belong to multi-label classification, 1 belong to regression, and 1 belong to clustering. Overall, 25 unique representation learning methods and 28 predictors are explored for developing robust predictive pipelines for 44 DNA sequence analysis tasks classified under the hood of 8 major biological goals. Across all eight goals, language model-based representation learning approaches and deep learning classifiers are achieving better performance across majority of the tasks. Researchers can explore the performance potential of latest transformer-based language models such as Hierarchical graph transformer, GPT-4, and hybrid representation learning methods along with sophisticated ensemble machine learning or deep learning predictors for different classification, regression, and clustering tasks.</p>
</sec>
<sec id="s11">
<title>11 Publisher and journal-wise distribution of research articles</title>
<p>This section provides an overview of 44 distinct DNA sequence analysis task-related articles distribution across conferences, journals, and publishers. Before paper submission, identification of relevant journals for a study publication in the interdisciplinary field of AI applications in DNA sequence analysis is an important task. There are three types of journals in AI and DNA sequence analysis fields: (1) Journals focusing on core AI algorithms, (2) Journals dedicated to core biological findings, (3) Hybrid journals that publish research integrating both AI algorithms and biological data. Researchers often face desk rejections when submitting to core AI or biology journals. Instead, they should target hybrid journals. While many tools exist to find suitable journals, this comprehensive guide provides detailed information to help researchers to identify journals where applications using word embeddings and large language models for DNA sequence analysis are published.</p>
<p><xref ref-type="fig" rid="F7">Figure 7</xref> graphically depicts distribution of 127 studies across 53 journals, 1 transactions, 3 conferences, and 2 pre-print repositories. Among all journals, more studies are published in Briefings in Bioinformatics followed by Bioinformatics, Computational Biology, and Chemistry, and International Journal of Molecular Sciences. Similarly, among all conferences, more studies are published in the International Conference on Bioinformatics and Biomedicine (BIBM) followed by the 11<sup><italic>th</italic></sup> Hellenic Conference on Artificial Intelligence, Proceedings of the 12th and 13<sup><italic>th</italic></sup> ACM International Conference on Bioinformatics, Computational Biology, and Health Informatics. Moreover, 5 studies are published in ACM transaction of computational biology. In the light of rapid development in research findings, researchers have also published 24 studies in bioRxiv, ArXiV, and MedXiv platforms. However, researchers generally prefer journal publications for their sustained impact.</p>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>Publication distribution of DNA sequence analysis literature across diverse journals and conferences from 2018 to 2024.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1503229-g0007.tif"/>
</fig>
<p>Furthermore, <xref ref-type="fig" rid="F8">Figure 8</xref> illustrate that 127 DNA sequence analysis studies are published by 17 different publishers, namely, Springer,<xref ref-type="fn" rid="fn0009"><sup>9</sup></xref> Elsevier (see text footnote <xref ref-type="fn" rid="fn0003"><sup>3</sup></xref>), Oxford University Press,<xref ref-type="fn" rid="fn0010"><sup>10</sup></xref> Cold Spring Harbor Laboratory,<xref ref-type="fn" rid="fn0011"><sup>11</sup></xref> IEEE,<xref ref-type="fn" rid="fn0012"><sup>12</sup></xref> Ozer UYGUN,<xref ref-type="fn" rid="fn0013"><sup>13</sup></xref> ACS Publication,<xref ref-type="fn" rid="fn0014"><sup>14</sup></xref> Frontier Media SA,<xref ref-type="fn" rid="fn0015"><sup>15</sup></xref> Gazi University,<xref ref-type="fn" rid="fn0016"><sup>16</sup></xref> Marry Ann Liebert,<xref ref-type="fn" rid="fn0017"><sup>17</sup></xref> MDPI,<xref ref-type="fn" rid="fn0018"><sup>18</sup></xref> National Acad Sciences,<xref ref-type="fn" rid="fn0019"><sup>19</sup></xref> Nature Publishing Group UK London,<xref ref-type="fn" rid="fn0020"><sup>20</sup></xref> PeerJ Inc.,<xref ref-type="fn" rid="fn0021"><sup>21</sup></xref> Public Library of science,<xref ref-type="fn" rid="fn0022"><sup>22</sup></xref> ACM (see text footnote <xref ref-type="fn" rid="fn0002"><sup>2</sup></xref>), and pre-prints.<xref ref-type="fn" rid="fn0023"><sup>23</sup></xref> Notably, approximately 60 out of 127 DNA sequence analysis studies are published by Oxford University Press, Elsevier, and Cold Spring Harbor Laboratory. In addition, IEEE, Springer, and MDPI have contributed 30 relevant papers. Furthermore, 32 DNA sequence analysis research articles are published by ACS Publications, Frontiers Media SA, Mary Ann Liebert, Inc., National Acad Sciences, Nature Publishing Group UK London, Public Library of Science, PeerJ Inc, and others. Collectively, 96 are journal publications, 6 are conference papers, 1 is transaction articles, and 24 are pre-prints out of 127 DNA sequence analysis studies published by 21 different publishers. This comprehensive analysis across various journals, conferences, transactions, and pre-print repositories highlights diverse and extensive research landscape in DNA sequence analysis.</p>
<fig id="F8" position="float">
<label>Figure 8</label>
<caption><p>Distribution of publishers involved in the publication of DNA sequence analysis literature from 2018 to 2024.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1503229-g0008.tif"/>
</fig>
</sec>
<sec sec-type="discussion" id="s12">
<title>12 Discussion</title>
<p>We acknowledge that &#x0201C;DNA sequence analysis&#x0201D; encompasses a much broader range of bioinformatics applications than covered in this review. Our focus is specifically on AI-based approaches that analyze raw DNA sequence data to predict biological functions and features. Other important areas of bioinformatics such as genome assembly, comprehensive variant analysis, phylogenomics, and many aspects of population genetics utilize different computational approaches and would benefit from separate dedicated reviews. A comprehensive review of existing literature on AI-driven DNA sequence analysis tasks reveals a significant inconsistency in the evaluation of predictive pipelines across similar datasets. Researchers have developed numerous datasets tailored to specific tasks, and most of the researchers have evaluated their proposed predictors solely on their own datasets.</p>
<p>The creation of new datasets is essential because public databases are frequently updated with new sequence information. These new datasets can incorporate the most recent sequence data alongside existing information. Moreover, existing datasets tend to be smaller, while deep learning models perform better with larger datasets. To address performance comparison inconsistency, there is an urgent need to standardize dataset utilization. One potential solution is to benchmark existing predictors on newly developed datasets and compare the performance of proposed predictors against these benchmarks. This approach would provide a more objective evaluation of proposed predictor performance. However, in this context, a significant challenge is the limited availability of source codes for existing predictors. Many studies make their source code private, hindering the reproducibility of results and direct comparison with other methods. To streamline the integration of innovative methods and ensure methodological advancement, it is crucial to analyze task-specific datasets and create standardized datasets with detailed descriptions. By benchmarking the performance of existing predictors on these standardized datasets, researchers can establish a common ground for comparison and facilitate more accurate evaluations of models. This approach would enhance transparency and reproducibility in DNA sequence analysis studies.</p>
<p>The development of AI-driven predictive pipelines for DNA sequence analysis relies heavily on effective sequence representation learning methods and appropriate machine or deep learning models. Machine and deep learning models inherently depend on statistical vectors and cannot process raw DNA sequences directly. Therefore, the role of representation learning methods in these pipelines is crucial. These methods are responsible for transforming raw DNA sequences into statistical vectors by capturing and encoding the most informative nucleotide patterns.</p>
<p>In the current landscape of AI-driven DNA sequence analysis, researchers have employed a variety of representation learning methods, including 12 distinct word embedding techniques and 8 language models. However, when it comes to other genetic molecules such as RNA and proteins, researchers have explored an additional set of 17 word embedding methods and 13 language models that have not yet been applied to DNA sequence analysis. These unexplored word embedding methods include DANE (<xref ref-type="bibr" rid="B285">285</xref>), ELMo (<xref ref-type="bibr" rid="B286">286</xref>&#x02013;<xref ref-type="bibr" rid="B288">288</xref>), GATNE (<xref ref-type="bibr" rid="B289">289</xref>), GEMSEC (<xref ref-type="bibr" rid="B290">290</xref>), MetaGraph2Vec (<xref ref-type="bibr" rid="B291">291</xref>), HAKE (<xref ref-type="bibr" rid="B292">292</xref>), HIN2Vec (<xref ref-type="bibr" rid="B293">293</xref>), HOPE (<xref ref-type="bibr" rid="B294">294</xref>, <xref ref-type="bibr" rid="B295">295</xref>), LINE (<xref ref-type="bibr" rid="B296">296</xref>&#x02013;<xref ref-type="bibr" rid="B298">298</xref>), Mashup (<xref ref-type="bibr" rid="B299">299</xref>, <xref ref-type="bibr" rid="B300">300</xref>), Random Watcher-Walker (RW2) (<xref ref-type="bibr" rid="B301">301</xref>), RotatE (<xref ref-type="bibr" rid="B292">292</xref>, <xref ref-type="bibr" rid="B302">302</xref>, <xref ref-type="bibr" rid="B303">303</xref>), RWR (<xref ref-type="bibr" rid="B304">304</xref>), Struc2Vec (<xref ref-type="bibr" rid="B305">305</xref>, <xref ref-type="bibr" rid="B306">306</xref>), SVD (<xref ref-type="bibr" rid="B307">307</xref>, <xref ref-type="bibr" rid="B308">308</xref>), Topo2Vec (<xref ref-type="bibr" rid="B309">309</xref>), and TransE (<xref ref-type="bibr" rid="B310">310</xref>), while the unexplored language models include AlphaFold (<xref ref-type="bibr" rid="B311">311</xref>&#x02013;<xref ref-type="bibr" rid="B315">315</xref>), AlphaFold2 (<xref ref-type="bibr" rid="B316">316</xref>, <xref ref-type="bibr" rid="B317">317</xref>), BigBird (<xref ref-type="bibr" rid="B318">318</xref>), ESM-1 (<xref ref-type="bibr" rid="B315">315</xref>, <xref ref-type="bibr" rid="B319">319</xref>, <xref ref-type="bibr" rid="B320">320</xref>), ESM-2 (<xref ref-type="bibr" rid="B109">109</xref>, <xref ref-type="bibr" rid="B286">286</xref>, <xref ref-type="bibr" rid="B316">316</xref>, <xref ref-type="bibr" rid="B320">320</xref>), Graph Transformer Network (<xref ref-type="bibr" rid="B321">321</xref>), Heterogeneous Graph Transformer (<xref ref-type="bibr" rid="B322">322</xref>), IgFold (<xref ref-type="bibr" rid="B323">323</xref>), LongFormer (<xref ref-type="bibr" rid="B318">318</xref>), RoBERTa (<xref ref-type="bibr" rid="B324">324</xref>, <xref ref-type="bibr" rid="B325">325</xref>), T5 (<xref ref-type="bibr" rid="B320">320</xref>, <xref ref-type="bibr" rid="B326">326</xref>&#x02013;<xref ref-type="bibr" rid="B328">328</xref>), and Vision Transformer (<xref ref-type="bibr" rid="B288">288</xref>). Integrating these advanced word embedding techniques and large language models into AI-driven DNA sequence analysis pipelines could potentially enhance their performance and robustness.</p>
<p>Within 127 AI-driven DNA sequence analysis predictive pipelines, researchers have utilized 18 machine and deep learning algorithms at the predictor level. In some cases, they have developed meta-predictors by combining multiple machine learning and deep learning algorithms to enhance predictive performance. However, similar to the representation learning stage, there are 24 distinct methods at the predictor level that have not yet been explored, representing untapped potential for improving the accuracy and robustness of these AI-driven pipelines.</p>
<p>Our categorization of 44 distinct tasks into 8 biological goals provides a structured framework that serves as a valuable starting taxonomy for both computer scientists and life scientists. This organization is informed by both computational and biological literature and creates a common reference point that bridges these disciplines while facilitating interdisciplinary communication. We recognize the inherent complexity of biological systems and the interconnected nature of many of these tasks. For example, enhancer identification categorized under gene expression regulation shares biological connections with chromatin accessibility prediction categorized under genome structure and stability. Nevertheless, this framework offers a practical organizing principle that will naturally evolve and be refined over time. The taxonomy presented here lays groundwork that future collaborative efforts between AI researchers and domain specialists in genomics can build upon. We anticipate gradual development into a more nuanced framework that maintains practical utility while better reflecting biological realities. Similar to many scientific classification systems, we expect this taxonomy to mature through iterative refinement as the field advances.</p>
</sec>
<sec sec-type="conclusions" id="s13">
<title>13 Conclusion</title>
<p>This review serves as a comprehensive resource for researchers working at the intersection of AI and DNA sequence analysis. It provides a structured foundation for future innovations in the rapidly evolving field of computational genomics. It bridges the critical gap between molecular biology and artificial intelligence by systematically analyzing 44 different DNA sequence analysis tasks, their associated databases, datasets, and AI methodologies. It identifies 36 biological databases and 140 benchmark datasets that provide a robust foundation for developing and evaluating AI predictors. Furthermore, our examination of existing predictive pipelines demonstrates the successful application of 39 word embeddings and 67 language models across various DNA sequence analysis tasks. Our analysis reveals that while significant progress has been made in developing AI-driven predictive pipelines for DNA sequence analysis, several challenges and opportunities remain unexplored. Several promising directions emerge for the advancement of this field. First, the integration of 17 unexplored word embedding methods and 13 language models (currently utilized only for RNA and protein analysis) could significantly enhance DNA sequence analysis capabilities. Second, the development of standardized benchmark datasets and evaluation protocols would facilitate fair comparisons between different predictive models and accelerate progress in the field. Third, the adoption of 24 untapped machine learning and deep learning algorithms at the predictor level presents an opportunity to improve prediction accuracy and robustness.</p>
<p>Future research should focus on developing multi-task learning frameworks that can simultaneously handle multiple DNA sequence analysis tasks, thereby improving computational efficiency and leveraging shared biological features. Furthermore, ensuring public accessibility of source codes and detailed documentation of predictive pipelines would foster reproducibility and collaborative advancement in the field. The establishment of standardized performance metrics and evaluation protocols across different DNA sequence analysis tasks would enable more meaningful comparisons between various approaches and guide future developments. As DNA sequence data continue to grow exponentially, the integration of more sophisticated AI architectures, particularly those capable of handling large-scale genomic data efficiently, will become increasingly important. Our categorization of 44 tasks reflects common AI applications in DNA sequence analysis literature and provides a starting point for interdisciplinary discourse. We recognize that deeper collaboration between AI researchers and life scientists would further strengthen the biological relevance of this framework. A greater amount of input from geneticists and bioinformaticians would be essential to develop a more comprehensive and biologically relevant tasks taxonomy.</p>
</sec>
</body>
<back>
<sec sec-type="author-contributions" id="s14">
<title>Author contributions</title>
<p>MA: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. MI: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. AZ: Data curation, Formal analysis, Writing &#x02013; original draft. AD: Supervision, Writing &#x02013; review &#x00026; editing.</p>
</sec>
<sec sec-type="funding-information" id="s15">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research and/or publication of this article.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>MA, MI, AZ, and AD were employed by German Research Center for Artificial Intelligence GmbH. MA and AD are shareholders of company Intelligentx GmbH.</p>
</sec>
<sec sec-type="ai-statement" id="s16">
<title>Generative AI statement</title>
<p>The author(s) declare Gen AI was used in the creation of this manuscript. During the preparation of this study, the authors used Grammarly tool to fix language and grammar issues and ChatGpt for outlining, better understanding different studies, and expansion of concepts. After using these tools, the authors reviewed and edited the content as needed and take full responsibility for the content of the publication.</p>
</sec>
<sec sec-type="disclaimer" id="s17">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<fn-group>
<fn id="fn0001"><p><sup>1</sup><ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/">https://scholar.google.com/</ext-link></p></fn>
<fn id="fn0002"><p><sup>2</sup><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/">https://dl.acm.org/</ext-link></p></fn>
<fn id="fn0003"><p><sup>3</sup><ext-link ext-link-type="uri" xlink:href="https://www.elsevier.com/">https://www.elsevier.com/</ext-link></p></fn>
<fn id="fn0004"><p><sup>4</sup><ext-link ext-link-type="uri" xlink:href="https://ieeexplore.ieee.org/">https://ieeexplore.ieee.org/</ext-link></p></fn>
<fn id="fn0005"><p><sup>5</sup><ext-link ext-link-type="uri" xlink:href="https://www.wiley.com/en-us">https://www.wiley.com/en-us</ext-link></p></fn>
<fn id="fn0006"><p><sup>6</sup><ext-link ext-link-type="uri" xlink:href="https://www.springer.com/gp">https://www.springer.com/gp</ext-link></p></fn>
<fn id="fn0007"><p><sup>7</sup><ext-link ext-link-type="uri" xlink:href="https://www.sciencedirect.com/">https://www.sciencedirect.com/</ext-link></p></fn>
<fn id="fn0008"><p><sup>8</sup><ext-link ext-link-type="uri" xlink:href="https://research.google/blog/more-efficient-nlp-model-pre-training-with-electra/">https://research.google/blog/more-efficient-nlp-model-pre-training-with-electra/</ext-link></p></fn>
<fn id="fn0009"><p><sup>9</sup><ext-link ext-link-type="uri" xlink:href="https://www.springer.com/in">https://www.springer.com/in</ext-link></p></fn>
<fn id="fn0010"><p><sup>10</sup><ext-link ext-link-type="uri" xlink:href="https://global.oup.com/academic/">https://global.oup.com/academic/</ext-link></p></fn>
<fn id="fn0011"><p><sup>11</sup><ext-link ext-link-type="uri" xlink:href="https://www.cshlpress.com/">https://www.cshlpress.com/</ext-link></p></fn>
<fn id="fn0012"><p><sup>12</sup><ext-link ext-link-type="uri" xlink:href="https://www.ieee.org/">https://www.ieee.org/</ext-link></p></fn>
<fn id="fn0013"><p><sup>13</sup><ext-link ext-link-type="uri" xlink:href="https://dergipark.org.tr/en/pub/&#x00040;ozeruygun">https://dergipark.org.tr/en/pub/&#x00040;ozeruygun</ext-link></p></fn>
<fn id="fn0014"><p><sup>14</sup><ext-link ext-link-type="uri" xlink:href="https://pubs.acs.org/">https://pubs.acs.org/</ext-link></p></fn>
<fn id="fn0015"><p><sup>15</sup><ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/">https://www.frontiersin.org/</ext-link></p></fn>
<fn id="fn0016"><p><sup>16</sup><ext-link ext-link-type="uri" xlink:href="https://gazi.edu.tr/">https://gazi.edu.tr/</ext-link></p></fn>
<fn id="fn0017"><p><sup>17</sup><ext-link ext-link-type="uri" xlink:href="https://www.liebertpub.com/">https://www.liebertpub.com/</ext-link></p></fn>
<fn id="fn0018"><p><sup>18</sup><ext-link ext-link-type="uri" xlink:href="https://www.mdpi.com/">https://www.mdpi.com/</ext-link></p></fn>
<fn id="fn0019"><p><sup>19</sup><ext-link ext-link-type="uri" xlink:href="https://www.nasonline.org/">https://www.nasonline.org/</ext-link></p></fn>
<fn id="fn0020"><p><sup>20</sup><ext-link ext-link-type="uri" xlink:href="https://www.nature.com/">https://www.nature.com/</ext-link></p></fn>
<fn id="fn0021"><p><sup>21</sup><ext-link ext-link-type="uri" xlink:href="https://peerj.com/">https://peerj.com/</ext-link></p></fn>
<fn id="fn0022"><p><sup>22</sup><ext-link ext-link-type="uri" xlink:href="https://plos.org/">https://plos.org/</ext-link></p></fn>
<fn id="fn0023"><p><sup>23</sup><ext-link ext-link-type="uri" xlink:href="https://arxiv.org/">https://arxiv.org/</ext-link></p></fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Allison</surname> <given-names>LA</given-names></name></person-group>. <source>Fundamental Molecular Biology</source>. New York: John Wiley &#x00026; Sons. (<year>2021</year>).</citation>
</ref>
<ref id="B2">
<label>2.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gartner</surname> <given-names>A</given-names></name> <name><surname>Engebrecht</surname> <given-names>J</given-names></name></person-group>. <article-title>DNA repair, recombination, and damage signaling</article-title>. <source>Genetics</source>. (<year>2022</year>) <volume>220</volume>:<fpage>iyab178</fpage>. <pub-id pub-id-type="doi">10.1093/genetics/iyab178</pub-id><pub-id pub-id-type="pmid">35137093</pub-id></citation></ref>
<ref id="B3">
<label>3.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>J</given-names></name> <name><surname>Potlapalli</surname> <given-names>R</given-names></name> <name><surname>Quan</surname> <given-names>H</given-names></name> <name><surname>Chen</surname> <given-names>L</given-names></name> <name><surname>Xie</surname> <given-names>Y</given-names></name> <name><surname>Pouriyeh</surname> <given-names>S</given-names></name> <etal/></person-group>. <article-title>Exploring DNA damage and repair mechanisms: a review with computational insights</article-title>. <source>BioTech</source>. (<year>2024</year>) <volume>13</volume>:<fpage>3</fpage>. <pub-id pub-id-type="doi">10.3390/biotech13010003</pub-id><pub-id pub-id-type="pmid">38247733</pub-id></citation></ref>
<ref id="B4">
<label>4.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Aseervatham</surname> <given-names>GSB</given-names></name> <name><surname>Sivasudha</surname> <given-names>T</given-names></name> <name><surname>Jeyadevi</surname> <given-names>R</given-names></name> <name><surname>Arul Ananth</surname> <given-names>D</given-names></name></person-group>. <article-title>Environmental factors and unhealthy lifestyle influence oxidative stress in humans&#x02013;an overview</article-title>. <source>Environ Sci Pollut Res</source>. (<year>2013</year>) <volume>20</volume>:<fpage>4356</fpage>&#x02013;<lpage>69</lpage>. <pub-id pub-id-type="doi">10.1007/s11356-013-1748-0</pub-id><pub-id pub-id-type="pmid">23636598</pub-id></citation></ref>
<ref id="B5">
<label>5.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liao</surname> <given-names>X</given-names></name> <name><surname>Zhu</surname> <given-names>W</given-names></name> <name><surname>Zhou</surname> <given-names>J</given-names></name> <name><surname>Li</surname> <given-names>H</given-names></name> <name><surname>Xu</surname> <given-names>X</given-names></name> <name><surname>Zhang</surname> <given-names>B</given-names></name> <etal/></person-group>. <article-title>Repetitive DNA sequence detection and its role in the human genome</article-title>. <source>Commun Biol</source>. (<year>2023</year>) <volume>6</volume>:<fpage>954</fpage>. <pub-id pub-id-type="doi">10.1038/s42003-023-05322-y</pub-id><pub-id pub-id-type="pmid">37726397</pub-id></citation></ref>
<ref id="B6">
<label>6.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Joiret</surname> <given-names>M</given-names></name> <name><surname>Leclercq</surname> <given-names>M</given-names></name> <name><surname>Lambrechts</surname> <given-names>G</given-names></name> <name><surname>Rapino</surname> <given-names>F</given-names></name> <name><surname>Close</surname> <given-names>P</given-names></name> <name><surname>Louppe</surname> <given-names>G</given-names></name> <etal/></person-group>. <article-title>Cracking the genetic code with neural networks</article-title>. <source>Front Artif Intell</source>. (<year>2023</year>) <volume>6</volume>:<fpage>1128153</fpage>. <pub-id pub-id-type="doi">10.3389/frai.2023.1128153</pub-id><pub-id pub-id-type="pmid">37091301</pub-id></citation></ref>
<ref id="B7">
<label>7.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Laub</surname> <given-names>V</given-names></name> <name><surname>Devraj</surname> <given-names>K</given-names></name> <name><surname>Elias</surname> <given-names>L</given-names></name> <name><surname>Schulte</surname> <given-names>D</given-names></name></person-group>. <article-title>Bioinformatics for wet-lab scientists: practical application in sequencing analysis</article-title>. <source>BMC Genomics</source>. (<year>2023</year>) <volume>24</volume>:<fpage>382</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-023-09454-7</pub-id><pub-id pub-id-type="pmid">37420172</pub-id></citation></ref>
<ref id="B8">
<label>8.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Elshafei</surname> <given-names>A</given-names></name> <name><surname>Al-Toubat</surname> <given-names>M</given-names></name> <name><surname>Feibus</surname> <given-names>AH</given-names></name> <name><surname>Koul</surname> <given-names>K</given-names></name> <name><surname>Jazayeri</surname> <given-names>SB</given-names></name> <name><surname>Lelani</surname> <given-names>N</given-names></name> <etal/></person-group>. <article-title>Genetic mutations in smoking-associated prostate cancer</article-title>. <source>Prostate</source>. (<year>2023</year>) <volume>83</volume>:<fpage>1229</fpage>&#x02013;<lpage>37</lpage>. <pub-id pub-id-type="doi">10.1002/pros.24554</pub-id><pub-id pub-id-type="pmid">37455402</pub-id></citation></ref>
<ref id="B9">
<label>9.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>Y</given-names></name> <name><surname>Liu</surname> <given-names>M</given-names></name></person-group>. <article-title>Application of machine learning based genome sequence analysis in pathogen identification</article-title>. <source>Front Microbiol</source>. (<year>2024</year>) <volume>15</volume>:<fpage>1474078</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2024.1474078</pub-id><pub-id pub-id-type="pmid">39417073</pub-id></citation></ref>
<ref id="B10">
<label>10.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bi</surname> <given-names>Z</given-names></name> <name><surname>Dip</surname> <given-names>SA</given-names></name> <name><surname>Hajialigol</surname> <given-names>D</given-names></name> <name><surname>Kommu</surname> <given-names>S</given-names></name> <name><surname>Liu</surname> <given-names>H</given-names></name> <name><surname>Lu</surname> <given-names>M</given-names></name> <etal/></person-group>. <article-title>AI for biomedicine in the era of large language models</article-title>. <source>arXiv preprint arXiv:240315673</source>. (<year>2024</year>).</citation>
</ref>
<ref id="B11">
<label>11.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Q</given-names></name> <name><surname>Ding</surname> <given-names>K</given-names></name> <name><surname>Lyv</surname> <given-names>T</given-names></name> <name><surname>Wang</surname> <given-names>X</given-names></name> <name><surname>Yin</surname> <given-names>Q</given-names></name> <name><surname>Zhang</surname> <given-names>Y</given-names></name> <etal/></person-group>. <article-title>Scientific large language models: a survey on biological &#x00026; chemical domains</article-title>. <source>arXiv preprint arXiv:240114656</source>. (<year>2024</year>).</citation>
</ref>
<ref id="B12">
<label>12.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Asim</surname> <given-names>MN</given-names></name></person-group>. <source>An efficient automated machine learning framework for genomics and proteomics sequence analysis</source>. Rheinland-Pf&#x000E4;lzische Technische Universit&#x000E4;t Kaiserslautern-Landau. (<year>2023</year>).</citation>
</ref>
<ref id="B13">
<label>13.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>O&#x00027;Leary</surname> <given-names>NA</given-names></name> <name><surname>Cox</surname> <given-names>E</given-names></name> <name><surname>Holmes</surname> <given-names>JB</given-names></name> <name><surname>Anderson</surname> <given-names>WR</given-names></name> <name><surname>Falk</surname> <given-names>R</given-names></name> <name><surname>Hem</surname> <given-names>V</given-names></name> <etal/></person-group>. <article-title>Exploring and retrieving sequence and metadata for species across the tree of life with NCBI datasets</article-title>. <source>Sci Data</source>. (<year>2024</year>) <volume>11</volume>:<fpage>732</fpage>. <pub-id pub-id-type="doi">10.1038/s41597-024-03571-y</pub-id><pub-id pub-id-type="pmid">38969627</pub-id></citation></ref>
<ref id="B14">
<label>14.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Abbasi</surname> <given-names>AF</given-names></name> <name><surname>Asim</surname> <given-names>MN</given-names></name> <name><surname>Ahmed</surname> <given-names>S</given-names></name> <name><surname>Dengel</surname> <given-names>A</given-names></name></person-group>. <article-title>Long extrachromosomal circular DNA identification by fusing sequence-derived features of physicochemical properties and nucleotide distribution patterns</article-title>. <source>Sci Rep</source>. (<year>2024</year>) <volume>14</volume>:<fpage>9466</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-024-57457-5</pub-id><pub-id pub-id-type="pmid">38658614</pub-id></citation></ref>
<ref id="B15">
<label>15.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jin</surname> <given-names>YT</given-names></name> <name><surname>Tan</surname> <given-names>Y</given-names></name> <name><surname>Gan</surname> <given-names>ZH</given-names></name> <name><surname>Hao</surname> <given-names>YD</given-names></name> <name><surname>Wang</surname> <given-names>TY</given-names></name> <name><surname>Lin</surname> <given-names>H</given-names></name> <etal/></person-group>. <article-title>Identification of DNase I hypersensitive sites in the human genome by multiple sequence descriptors</article-title>. <source>Methods</source>. (<year>2024</year>) <volume>229</volume>:<fpage>125</fpage>&#x02013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1016/j.ymeth.2024.06.012</pub-id><pub-id pub-id-type="pmid">38964595</pub-id></citation></ref>
<ref id="B16">
<label>16.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Y</given-names></name> <name><surname>Wei</surname> <given-names>X</given-names></name> <name><surname>Yang</surname> <given-names>Q</given-names></name> <name><surname>Xiong</surname> <given-names>A</given-names></name> <name><surname>Li</surname> <given-names>X</given-names></name> <name><surname>Zou</surname> <given-names>Q</given-names></name> <etal/></person-group>. <article-title>msBERT-Promoter: a multi-scale ensemble predictor based on BERT pre-trained model for the two-stage prediction of DNA promoters and their strengths</article-title>. <source>BMC Biol</source>. (<year>2024</year>) <volume>22</volume>:<fpage>126</fpage>. <pub-id pub-id-type="doi">10.1186/s12915-024-01923-z</pub-id><pub-id pub-id-type="pmid">38816885</pub-id></citation></ref>
<ref id="B17">
<label>17.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>B</given-names></name> <name><surname>Guneri</surname> <given-names>D</given-names></name> <name><surname>Yu</surname> <given-names>H</given-names></name> <name><surname>Wright</surname> <given-names>EP</given-names></name> <name><surname>Chen</surname> <given-names>W</given-names></name> <name><surname>Waller</surname> <given-names>ZA</given-names></name> <etal/></person-group>. <article-title>Prediction of DNA i-motifs via machine learning</article-title>. <source>Nucleic Acids Res</source>. (<year>2024</year>) <volume>52</volume>:<fpage>2188</fpage>&#x02013;<lpage>97</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkae092</pub-id><pub-id pub-id-type="pmid">38364855</pub-id></citation></ref>
<ref id="B18">
<label>18.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>X</given-names></name> <name><surname>Qiao</surname> <given-names>L</given-names></name> <name><surname>Qu</surname> <given-names>P</given-names></name> <name><surname>Yang</surname> <given-names>Q</given-names></name></person-group>. <article-title>TBCA: prediction of transcription factor binding sites using a deep neural network with lightweight attention mechanism</article-title>. <source>IEEE J Biomed Health Inf</source>. (<year>2024</year>) <volume>28</volume>:<fpage>2397</fpage>&#x02013;<lpage>2407</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2024.3355758</pub-id><pub-id pub-id-type="pmid">38236675</pub-id></citation></ref>
<ref id="B19">
<label>19.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>X</given-names></name> <name><surname>Du</surname> <given-names>Q</given-names></name> <name><surname>Wang</surname> <given-names>R</given-names></name></person-group>. <article-title>Mus4mCPred: accurate identification of DNA N4-methylcytosine sites in mouse genome using multi-view feature learning and deep hybrid network</article-title>. <source>Processes</source>. (<year>2024</year>) <volume>12</volume>:<fpage>1129</fpage>. <pub-id pub-id-type="doi">10.3390/pr12061129</pub-id></citation>
</ref>
<ref id="B20">
<label>20.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Z</given-names></name> <name><surname>Zhao</surname> <given-names>P</given-names></name> <name><surname>Li</surname> <given-names>C</given-names></name> <name><surname>Li</surname> <given-names>F</given-names></name> <name><surname>Xiang</surname> <given-names>D</given-names></name> <name><surname>Chen</surname> <given-names>YZ</given-names></name> <etal/></person-group>. <article-title>iLearnPlus: a comprehensive and automated machine-learning platform for nucleic acid and protein sequence analysis, prediction and visualization</article-title>. <source>Nucleic Acids Res</source>. (<year>2021</year>) <volume>49</volume>:<fpage>e60</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkab122</pub-id><pub-id pub-id-type="pmid">33660783</pub-id></citation></ref>
<ref id="B21">
<label>21.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ren</surname> <given-names>R</given-names></name> <name><surname>Yin</surname> <given-names>C</given-names></name> <name><surname>S-T Yau</surname> <given-names>S</given-names></name></person-group>. <article-title>kmer2vec: a novel method for comparing DNA sequences by word2vec embedding</article-title>. <source>J Comput Biol</source>. (<year>2022</year>) <volume>29</volume>:<fpage>1001</fpage>&#x02013;<lpage>21</lpage>. <pub-id pub-id-type="doi">10.1089/cmb.2021.0536</pub-id><pub-id pub-id-type="pmid">35593919</pub-id></citation></ref>
<ref id="B22">
<label>22.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>Z</given-names></name> <name><surname>Wu</surname> <given-names>W</given-names></name> <name><surname>Ho</surname> <given-names>H</given-names></name> <name><surname>Wang</surname> <given-names>J</given-names></name> <name><surname>Shi</surname> <given-names>L</given-names></name> <name><surname>Davuluri</surname> <given-names>RV</given-names></name> <etal/></person-group>. <article-title>Dnabert-s: Learning species-aware DNA embedding with genome foundation models</article-title>. <source>arXiv:2402.08777</source>. (<year>2024</year>).<pub-id pub-id-type="pmid">38410647</pub-id></citation></ref>
<ref id="B23">
<label>23.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>HL</given-names></name> <name><surname>Pang</surname> <given-names>YH</given-names></name> <name><surname>Liu</surname> <given-names>B</given-names></name></person-group>. <article-title>BioSeq-BLM: a platform for analyzing DNA, RNA and protein sequences based on biological language models</article-title>. <source>Nucleic Acids Res</source>. (<year>2021</year>) <volume>49</volume>:<fpage>e129</fpage>&#x02013;<lpage>e129</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkab829</pub-id><pub-id pub-id-type="pmid">34581805</pub-id></citation></ref>
<ref id="B24">
<label>24.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Airlangga</surname> <given-names>G</given-names></name></person-group>. <article-title>Comparative analysis of deep learning architectures for DNA sequence classification: performance evaluation and model insights</article-title>. <source>J Comput Syst Inf</source>. (<year>2024</year>) <volume>5</volume>:<fpage>709</fpage>&#x02013;<lpage>18</lpage>.<pub-id pub-id-type="pmid">34550967</pub-id></citation></ref>
<ref id="B25">
<label>25.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>C</given-names></name> <name><surname>He</surname> <given-names>Z</given-names></name> <name><surname>Jia</surname> <given-names>R</given-names></name> <name><surname>Pan</surname> <given-names>S</given-names></name> <name><surname>Coin</surname> <given-names>LJ</given-names></name> <name><surname>Song</surname> <given-names>J</given-names></name> <etal/></person-group>. <article-title>PLANNER: a multi-scale deep language model for the origins of replication site prediction</article-title>. <source>IEEE J Biomed Health Inf</source>. (<year>2024</year>) <volume>28</volume>:<fpage>2445</fpage>&#x02013;<lpage>2454</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2024.3349584</pub-id><pub-id pub-id-type="pmid">38190667</pub-id></citation></ref>
<ref id="B26">
<label>26.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yin</surname> <given-names>ZN</given-names></name> <name><surname>Lai</surname> <given-names>FL</given-names></name> <name><surname>Gao</surname> <given-names>F</given-names></name></person-group>. <article-title>Unveiling human origins of replication using deep learning: accurate prediction and comprehensive analysis</article-title>. <source>Brief Bioinf</source>. (<year>2024</year>) <volume>25</volume>:<fpage>bbad432</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbad432</pub-id><pub-id pub-id-type="pmid">38008420</pub-id></citation></ref>
<ref id="B27">
<label>27.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Duan</surname> <given-names>C</given-names></name> <name><surname>Zang</surname> <given-names>Z</given-names></name> <name><surname>Xu</surname> <given-names>Y</given-names></name> <name><surname>He</surname> <given-names>H</given-names></name> <name><surname>Liu</surname> <given-names>Z</given-names></name> <name><surname>Song</surname> <given-names>Z</given-names></name> <etal/></person-group>. <article-title>FGBERT: function-driven pre-trained gene language model for metagenomics</article-title>. <source>arXiv preprint arXiv:240216901</source>. (<year>2024</year>).</citation>
</ref>
<ref id="B28">
<label>28.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>J</given-names></name> <name><surname>Chiu</surname> <given-names>TP</given-names></name> <name><surname>Rohs</surname> <given-names>R</given-names></name></person-group>. <article-title>Predicting DNA structure using a deep learning method</article-title>. <source>Nat Commun</source>. (<year>2024</year>) <volume>15</volume>:<fpage>1243</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-024-45191-5</pub-id><pub-id pub-id-type="pmid">38336958</pub-id></citation></ref>
<ref id="B29">
<label>29.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fazeel</surname> <given-names>A</given-names></name> <name><surname>Agha</surname> <given-names>A</given-names></name> <name><surname>Dengel</surname> <given-names>A</given-names></name> <name><surname>Ahmed</surname> <given-names>S</given-names></name></person-group>. <article-title>NP-BERT: a two-staged BERT based nucleosome positioning prediction architecture for multiple species</article-title>. In: <source>BIOINFORMATICS</source>. (<year>2023</year>). p. <fpage>175</fpage>&#x02013;<lpage>187</lpage>. <pub-id pub-id-type="doi">10.5220/0011679200003414</pub-id></citation>
</ref>
<ref id="B30">
<label>30.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Han GS Li</surname> <given-names>Q</given-names></name> <name><surname>Li</surname> <given-names>Y</given-names></name></person-group>. <article-title>Comparative analysis and prediction of nucleosome positioning using integrative feature representation and machine learning algorithms</article-title>. <source>BMC Bioinform</source>. (<year>2021</year>) <volume>22</volume>:<fpage>129</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-021-04006-w</pub-id><pub-id pub-id-type="pmid">34078256</pub-id></citation></ref>
<ref id="B31">
<label>31.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Y</given-names></name> <name><surname>Chu</surname> <given-names>X</given-names></name> <name><surname>Jiang</surname> <given-names>Y</given-names></name> <name><surname>Wu</surname> <given-names>H</given-names></name> <name><surname>Quan</surname> <given-names>L</given-names></name></person-group>. <article-title>SemanticCAP: chromatin accessibility prediction enhanced by features learning from a language model</article-title>. <source>Genes</source>. (<year>2022</year>) <volume>13</volume>:<fpage>568</fpage>. <pub-id pub-id-type="doi">10.3390/genes13040568</pub-id><pub-id pub-id-type="pmid">35456374</pub-id></citation></ref>
<ref id="B32">
<label>32.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fishman</surname> <given-names>V</given-names></name> <name><surname>Kuratov</surname> <given-names>Y</given-names></name> <name><surname>Petrov</surname> <given-names>M</given-names></name> <name><surname>Shmelev</surname> <given-names>A</given-names></name> <name><surname>Shepelin</surname> <given-names>D</given-names></name> <name><surname>Chekanov</surname> <given-names>N</given-names></name> <etal/></person-group>. <article-title>GENA-LM: a family of open-source foundational DNA language models for long sequences</article-title>. <source>bioRxiv</source>. (<year>2023</year>). p. <fpage>2023</fpage>&#x02013;<lpage>06</lpage>. <pub-id pub-id-type="doi">10.1101/2023.06.12.544594</pub-id><pub-id pub-id-type="pmid">39817513</pub-id></citation></ref>
<ref id="B33">
<label>33.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Guo</surname> <given-names>Y</given-names></name> <name><surname>Zhou</surname> <given-names>D</given-names></name> <name><surname>Nie</surname> <given-names>R</given-names></name> <name><surname>Ruan</surname> <given-names>X</given-names></name> <name><surname>Li</surname> <given-names>W</given-names></name></person-group>. <article-title>DeepANF: A deep attentive neural framework with distributed representation for chromatin accessibility prediction</article-title>. <source>Neurocomputing</source>. (<year>2020</year>) <volume>379</volume>:<fpage>305</fpage>&#x02013;<lpage>18</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2019.10.091</pub-id></citation>
</ref>
<ref id="B34">
<label>34.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>M</given-names></name> <name><surname>Huang</surname> <given-names>H</given-names></name> <name><surname>Huang</surname> <given-names>L</given-names></name> <name><surname>Zhang</surname> <given-names>N</given-names></name> <name><surname>Wu</surname> <given-names>J</given-names></name> <name><surname>Yang</surname> <given-names>H</given-names></name> <etal/></person-group>. <article-title>LOGO, a contextualized pre-trained language model of human genome flexibly adapts to various downstream tasks by fine-tuning</article-title>. (<year>2021</year>). <pub-id pub-id-type="doi">10.21203/rs.3.rs-448927/v1</pub-id></citation>
</ref>
<ref id="B35">
<label>35.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tao</surname> <given-names>H</given-names></name> <name><surname>Li</surname> <given-names>H</given-names></name> <name><surname>Xu</surname> <given-names>K</given-names></name> <name><surname>Hong</surname> <given-names>H</given-names></name> <name><surname>Jiang</surname> <given-names>S</given-names></name> <name><surname>Du</surname> <given-names>G</given-names></name> <etal/></person-group>. <article-title>Computational methods for the prediction of chromatin interaction and organization using sequence and epigenomic profiles</article-title>. <source>Brief Bioinform</source>. (<year>2021</year>) <volume>22</volume>:<fpage>bbaa405</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbaa405</pub-id><pub-id pub-id-type="pmid">33454752</pub-id></citation></ref>
<ref id="B36">
<label>36.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>Z</given-names></name> <name><surname>Liu</surname> <given-names>Q</given-names></name> <name><surname>Zeng</surname> <given-names>W</given-names></name> <name><surname>Jiang</surname> <given-names>R</given-names></name> <name><surname>Wong</surname> <given-names>WH</given-names></name></person-group>. <article-title>EpiGePT: a pretrained transformer model for epigenomics</article-title>. <source>bioRxiv</source>. (<year>2023</year>). <pub-id pub-id-type="doi">10.1101/2023.07.15.549134</pub-id><pub-id pub-id-type="pmid">37502861</pub-id></citation></ref>
<ref id="B37">
<label>37.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vanhaeren</surname> <given-names>T</given-names></name> <name><surname>Divina</surname> <given-names>F</given-names></name> <name><surname>Garc&#x000ED;a-Torres</surname> <given-names>M</given-names></name> <name><surname>G&#x000F3;mez-Vela</surname> <given-names>F</given-names></name> <name><surname>Vanhoof</surname> <given-names>W</given-names></name> <name><surname>Mart&#x000ED;nez-Garc&#x000ED;a</surname> <given-names>PM</given-names></name></person-group>. <article-title>A comparative study of supervised machine learning algorithms for the prediction of long-range chromatin interactions</article-title>. <source>Genes</source>. (<year>2020</year>) <volume>11</volume>:<fpage>985</fpage>. <pub-id pub-id-type="doi">10.3390/genes11090985</pub-id><pub-id pub-id-type="pmid">32847102</pub-id></citation></ref>
<ref id="B38">
<label>38.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Z</given-names></name> <name><surname>Li</surname> <given-names>F</given-names></name> <name><surname>Zhao</surname> <given-names>J</given-names></name> <name><surname>Zheng</surname> <given-names>C</given-names></name></person-group>. <article-title>CapsNetYY1: identifying YY1-mediated chromatin loops based on a capsule network architecture</article-title>. <source>BMC Genomics</source>. (<year>2023</year>) <volume>24</volume>:<fpage>448</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-023-09217-4</pub-id><pub-id pub-id-type="pmid">37559017</pub-id></citation></ref>
<ref id="B39">
<label>39.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Abbasi</surname> <given-names>AF</given-names></name> <name><surname>Asim</surname> <given-names>MN</given-names></name> <name><surname>Trygg</surname> <given-names>J</given-names></name> <name><surname>Dengel</surname> <given-names>A</given-names></name> <name><surname>Ahmed</surname> <given-names>S</given-names></name></person-group>. <article-title>Deep learning architectures for the prediction of YY1-mediated chromatin loops</article-title>. In: <source>International Symposium on Bioinformatics Research and Applications</source>. <publisher-loc>Springer</publisher-loc> (<year>2023</year>). p. <fpage>72</fpage>&#x02013;<lpage>84</lpage>. <pub-id pub-id-type="doi">10.1007/978-981-99-7074-2_6</pub-id></citation>
</ref>
<ref id="B40">
<label>40.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>W</given-names></name> <name><surname>Feng</surname> <given-names>P</given-names></name> <name><surname>Lin</surname> <given-names>H</given-names></name></person-group>. <article-title>Prediction of replication origins by calculating DNA structural properties</article-title>. <source>FEBS Lett</source>. (<year>2012</year>) <volume>586</volume>:<fpage>934</fpage>&#x02013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1016/j.febslet.2012.02.034</pub-id><pub-id pub-id-type="pmid">22449982</pub-id></citation></ref>
<ref id="B41">
<label>41.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nieduszynski</surname> <given-names>C</given-names></name> <name><surname>Hiraga</surname> <given-names>S</given-names></name> <name><surname>Ak</surname> <given-names>P</given-names></name> <name><surname>Benham</surname> <given-names>C</given-names></name> <name><surname>Donaldson</surname> <given-names>A</given-names></name></person-group>. <article-title>Oridb: a DNA replication origin database</article-title>. <source>Nucleic Acids Res</source>. (<year>2006</year>) <volume>35</volume>:<fpage>D40</fpage>&#x02013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkl758</pub-id><pub-id pub-id-type="pmid">17065467</pub-id></citation></ref>
<ref id="B42">
<label>42.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jordan</surname> <given-names>K</given-names></name> <name><surname>He</surname> <given-names>F</given-names></name> <name><surname>Soto</surname> <given-names>M</given-names></name> <name><surname>Akhunova</surname> <given-names>A</given-names></name> <name><surname>Akhunov</surname> <given-names>E</given-names></name></person-group>. <article-title>Differential chromatin accessibility landscape reveals structural and functional features of the allopolyploid wheat chromosomes</article-title>. <source>Genome Biol</source>. (<year>2020</year>) <volume>21</volume>:<fpage>176</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-020-02093-1</pub-id><pub-id pub-id-type="pmid">32684157</pub-id></citation></ref>
<ref id="B43">
<label>43.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Minnoye</surname> <given-names>L</given-names></name> <name><surname>Marinov</surname> <given-names>G</given-names></name> <name><surname>Krausgruber</surname> <given-names>T</given-names></name> <name><surname>Pan</surname> <given-names>L</given-names></name> <name><surname>Marand</surname> <given-names>A</given-names></name> <name><surname>Secchia</surname> <given-names>S</given-names></name> <etal/></person-group>. <article-title>Chromatin accessibility profiling methods</article-title>. <source>Nat Rev Methods Primers</source>. (<year>2021</year>) <volume>1</volume>:<fpage>10</fpage>. <pub-id pub-id-type="doi">10.1038/s43586-020-00008-9</pub-id><pub-id pub-id-type="pmid">38410680</pub-id></citation></ref>
<ref id="B44">
<label>44.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nguyen</surname> <given-names>E</given-names></name> <name><surname>Poli</surname> <given-names>M</given-names></name> <name><surname>Faizi</surname> <given-names>M</given-names></name> <name><surname>Thomas</surname> <given-names>A</given-names></name> <name><surname>Wornow</surname> <given-names>M</given-names></name> <name><surname>Birch-Sykes</surname> <given-names>C</given-names></name> <etal/></person-group>. <article-title>HyenaDNA: long-range genomic sequence modeling at single nucleotide resolution</article-title>. In: <source>Advances in Neural Information Processing Systems</source>. (<year>2024</year>). p. 36.</citation>
</ref>
<ref id="B45">
<label>45.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dao</surname> <given-names>FY</given-names></name> <name><surname>Lv</surname> <given-names>H</given-names></name> <name><surname>Zhang</surname> <given-names>D</given-names></name> <name><surname>Zhang</surname> <given-names>ZM</given-names></name> <name><surname>Liu</surname> <given-names>L</given-names></name> <name><surname>Lin</surname> <given-names>H</given-names></name></person-group>. <article-title>DeepYY1: a deep learning approach to identify YY1-mediated chromatin loops</article-title>. <source>Brief Bioinform</source>. (<year>2021</year>) <volume>22</volume>:<fpage>bbaa356</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbaa356</pub-id><pub-id pub-id-type="pmid">33279983</pub-id></citation></ref>
<ref id="B46">
<label>46.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Z</given-names></name> <name><surname>Gao</surname> <given-names>E</given-names></name> <name><surname>Zhou</surname> <given-names>J</given-names></name> <name><surname>Han</surname> <given-names>W</given-names></name> <name><surname>Xu</surname> <given-names>X</given-names></name> <name><surname>Gao</surname> <given-names>X</given-names></name></person-group>. <article-title>Applications of deep learning in understanding gene regulation</article-title>. <source>Cell Rep Methods</source>. (<year>2023</year>) <volume>3</volume>:<fpage>100384</fpage>. <pub-id pub-id-type="doi">10.1016/j.crmeth.2022.100384</pub-id><pub-id pub-id-type="pmid">36814848</pub-id></citation></ref>
<ref id="B47">
<label>47.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Khanal</surname> <given-names>J</given-names></name> <name><surname>Tayara</surname> <given-names>H</given-names></name> <name><surname>Chong</surname> <given-names>KT</given-names></name></person-group>. <article-title>Identifying enhancers and their strength by the integration of word embedding and convolution neural network</article-title>. <source>IEEE Access</source>. (<year>2020</year>) <volume>8</volume>:<fpage>58369</fpage>&#x02013;<lpage>76</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2020.2982666</pub-id></citation>
</ref>
<ref id="B48">
<label>48.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Q</given-names></name> <name><surname>Zhang</surname> <given-names>L</given-names></name> <name><surname>Xu</surname> <given-names>L</given-names></name> <name><surname>Zou</surname> <given-names>Q</given-names></name> <name><surname>Wu</surname> <given-names>J</given-names></name> <name><surname>Li</surname> <given-names>Q</given-names></name></person-group>. <article-title>Identification and classification of promoters using the attention mechanism based on long short-term memory</article-title>. <source>Front Comput Sci</source>. (<year>2022</year>) <volume>16</volume>:<fpage>164348</fpage>. <pub-id pub-id-type="doi">10.1007/s11704-021-0548-9</pub-id><pub-id pub-id-type="pmid">36116322</pub-id></citation></ref>
<ref id="B49">
<label>49.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Min</surname> <given-names>X</given-names></name> <name><surname>Ye</surname> <given-names>C</given-names></name> <name><surname>Liu</surname> <given-names>X</given-names></name> <name><surname>Zeng</surname> <given-names>X</given-names></name></person-group>. <article-title>Predicting enhancer-promoter interactions by deep learning and matching heuristic</article-title>. <source>Brief Bioinform</source>. (<year>2021</year>) <volume>22</volume>:<fpage>bbaa254</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbaa254</pub-id><pub-id pub-id-type="pmid">33096548</pub-id></citation></ref>
<ref id="B50">
<label>50.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Clauwaert</surname> <given-names>J</given-names></name> <name><surname>Waegeman</surname> <given-names>W</given-names></name></person-group>. <article-title>Novel transformer networks for improved sequence labeling in genomics</article-title>. <source>IEEE/ACM Trans Computat Biol Bioinform</source>. (<year>2020</year>) <volume>19</volume>:<fpage>97</fpage>&#x02013;<lpage>106</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2020.3035021</pub-id><pub-id pub-id-type="pmid">33125335</pub-id></citation></ref>
<ref id="B51">
<label>51.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ghosh</surname> <given-names>N</given-names></name> <name><surname>Santoni</surname> <given-names>D</given-names></name> <name><surname>Saha</surname> <given-names>I</given-names></name> <name><surname>Felici</surname> <given-names>G</given-names></name></person-group>. <article-title>Predicting transcription factor binding sites with deep learning</article-title>. <source>Int J Mol Sci</source>. (<year>2024</year>) <volume>25</volume>:<fpage>4990</fpage>. <pub-id pub-id-type="doi">10.3390/ijms25094990</pub-id><pub-id pub-id-type="pmid">38732207</pub-id></citation></ref>
<ref id="B52">
<label>52.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>Y</given-names></name> <name><surname>Wang</surname> <given-names>C</given-names></name> <name><surname>Xu</surname> <given-names>K</given-names></name> <name><surname>Ding</surname> <given-names>Y</given-names></name> <name><surname>Lyu</surname> <given-names>A</given-names></name> <name><surname>Zhang</surname> <given-names>L</given-names></name></person-group>. <article-title>TRAFICA: improving transcription factor binding affinity prediction using deep language model on ATAC-seq data</article-title>. <source>bioRxiv</source>. (<year>2023</year>). <pub-id pub-id-type="doi">10.1101/2023.11.02.565416</pub-id></citation>
</ref>
<ref id="B53">
<label>53.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Y</given-names></name> <name><surname>Tian</surname> <given-names>B</given-names></name></person-group>. <article-title>Protein-DNA binding sites prediction based on pre-trained protein language model and contrastive learning</article-title>. <source>Brief Bioinform</source>. (<year>2024</year>) <volume>25</volume>:<fpage>bbad488</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbad488</pub-id><pub-id pub-id-type="pmid">38171929</pub-id></citation></ref>
<ref id="B54">
<label>54.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>S</given-names></name> <name><surname>Hu</surname> <given-names>H</given-names></name> <name><surname>Jiang</surname> <given-names>T</given-names></name> <name><surname>Zhang</surname> <given-names>L</given-names></name> <name><surname>Zeng</surname> <given-names>J</given-names></name></person-group>. <article-title>TITER predicting translation initiation sites by deep learning</article-title>. <source>Bioinformatics</source>. (<year>2017</year>) <volume>33</volume>:<fpage>i234</fpage>&#x02013;<lpage>42</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btx247</pub-id><pub-id pub-id-type="pmid">28881981</pub-id></citation></ref>
<ref id="B55">
<label>55.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>J</given-names></name> <name><surname>Sun</surname> <given-names>W</given-names></name> <name><surname>Li</surname> <given-names>K</given-names></name> <name><surname>Zhang</surname> <given-names>W</given-names></name> <name><surname>Zhang</surname> <given-names>W</given-names></name> <name><surname>Zeng</surname> <given-names>Y</given-names></name> <etal/></person-group>. <article-title>MNESEDA: a prior-guided subgraph representation learning framework for predicting disease-related enhancers</article-title>. <source>Knowl-Based Syst</source>. (<year>2024</year>) <volume>294</volume>:<fpage>111734</fpage>. <pub-id pub-id-type="doi">10.1016/j.knosys.2024.111734</pub-id></citation>
</ref>
<ref id="B56">
<label>56.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Y</given-names></name></person-group>. <article-title>EnhancerBD identifing sequence feature</article-title>. <source>bioRxiv</source>. (<year>2024</year>). <pub-id pub-id-type="doi">10.1101/2024.03.05.583459</pub-id></citation>
</ref>
<ref id="B57">
<label>57.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mehmood</surname> <given-names>F</given-names></name> <name><surname>Arshad</surname> <given-names>S</given-names></name> <name><surname>Shoaib</surname> <given-names>M</given-names></name></person-group>. <article-title>ADH-Enhancer: an attention-based deep hybrid framework for enhancer identification and strength prediction</article-title>. <source>Brief Bioinform</source>. (<year>2024</year>) <volume>25</volume>:<fpage>bbae030</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbae030</pub-id><pub-id pub-id-type="pmid">38385876</pub-id></citation></ref>
<ref id="B58">
<label>58.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liao</surname> <given-names>M</given-names></name> <name><surname>Zhao</surname> <given-names>JP</given-names></name> <name><surname>Tian</surname> <given-names>J</given-names></name> <name><surname>Zheng</surname> <given-names>CH</given-names></name></person-group>. <article-title>iEnhancer-DCLA: using the original sequence to identify enhancers and their strength based on a deep learning framework</article-title>. <source>BMC Bioinform</source>. (<year>2022</year>) <volume>23</volume>:<fpage>480</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-022-05033-x</pub-id><pub-id pub-id-type="pmid">36376800</pub-id></citation></ref>
<ref id="B59">
<label>59.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Inayat</surname> <given-names>N</given-names></name> <name><surname>Khan</surname> <given-names>M</given-names></name> <name><surname>Iqbal</surname> <given-names>N</given-names></name> <name><surname>Khan</surname> <given-names>S</given-names></name> <name><surname>Raza</surname> <given-names>M</given-names></name> <name><surname>Khan</surname> <given-names>DM</given-names></name> <etal/></person-group>. <article-title>iEnhancer-DHF: identification of enhancers and their strengths using optimize deep neural network with multiple features extraction methods</article-title>. <source>IEEE Access</source>. (<year>2021</year>) <volume>9</volume>:<fpage>40783</fpage>&#x02013;<lpage>96</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2021.3062291</pub-id></citation>
</ref>
<ref id="B60">
<label>60.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Le</surname> <given-names>NQK</given-names></name> <name><surname>Ho</surname> <given-names>QT</given-names></name> <name><surname>Nguyen</surname> <given-names>TTD</given-names></name> <name><surname>Ou</surname> <given-names>YY</given-names></name></person-group>. <article-title>A transformer architecture based on BERT and 2D convolutional neural network to identify DNA enhancers from sequence information</article-title>. <source>Brief Bioinform</source>. (<year>2021</year>) <volume>22</volume>:<fpage>bbab005</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbab005</pub-id><pub-id pub-id-type="pmid">33539511</pub-id></citation></ref>
<ref id="B61">
<label>61.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Y</given-names></name> <name><surname>Hou</surname> <given-names>Z</given-names></name> <name><surname>Yang</surname> <given-names>Y</given-names></name> <name><surname>Wong</surname> <given-names>KC</given-names></name> <name><surname>Li</surname> <given-names>X</given-names></name></person-group>. <article-title>Genome-wide identification and characterization of DNA enhancers with a stacked multivariate fusion framework</article-title>. <source>PLoS Comput Biol</source>. (<year>2022</year>) <volume>18</volume>:<fpage>e1010779</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1010779</pub-id><pub-id pub-id-type="pmid">36520922</pub-id></citation></ref>
<ref id="B62">
<label>62.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>D</given-names></name> <name><surname>Karchin</surname> <given-names>R</given-names></name> <name><surname>Beer</surname> <given-names>MA</given-names></name></person-group>. <article-title>Discriminative prediction of mammalian enhancers from DNA sequence</article-title>. <source>Genome Res</source>. (<year>2011</year>) <volume>21</volume>:<fpage>2167</fpage>&#x02013;<lpage>80</lpage>. <pub-id pub-id-type="doi">10.1101/gr.121905.111</pub-id><pub-id pub-id-type="pmid">21875935</pub-id></citation></ref>
<ref id="B63">
<label>63.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Min</surname> <given-names>X</given-names></name> <name><surname>Zeng</surname> <given-names>W</given-names></name> <name><surname>Chen</surname> <given-names>S</given-names></name> <name><surname>Chen</surname> <given-names>N</given-names></name> <name><surname>Chen</surname> <given-names>T</given-names></name> <name><surname>Jiang</surname> <given-names>R</given-names></name></person-group>. <article-title>Predicting enhancers with deep convolutional neural networks</article-title>. <source>BMC Bioinform</source>. (<year>2017</year>) <volume>18</volume>:<fpage>478</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-017-1878-3</pub-id><pub-id pub-id-type="pmid">29219068</pub-id></citation></ref>
<ref id="B64">
<label>64.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Le</surname> <given-names>NQK</given-names></name> <name><surname>Yapp</surname> <given-names>EKY</given-names></name> <name><surname>Ho</surname> <given-names>QT</given-names></name> <name><surname>Nagasundaram</surname> <given-names>N</given-names></name> <name><surname>Ou</surname> <given-names>YY</given-names></name> <name><surname>Yeh</surname> <given-names>HY</given-names></name></person-group>. <article-title>iEnhancer-5Step: identifying enhancers using hidden information of DNA sequences via Chou&#x00027;s 5-step rule and word embedding</article-title>. <source>Anal Biochem</source>. (<year>2019</year>) <volume>571</volume>:<fpage>53</fpage>&#x02013;<lpage>61</lpage>. <pub-id pub-id-type="doi">10.1016/j.ab.2019.02.017</pub-id><pub-id pub-id-type="pmid">30822398</pub-id></citation></ref>
<ref id="B65">
<label>65.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Geng</surname> <given-names>Q</given-names></name> <name><surname>Yang</surname> <given-names>R</given-names></name> <name><surname>Zhang</surname> <given-names>L</given-names></name> <name><surname>A</surname></name></person-group>. <article-title>deep learning framework for enhancer prediction using word embedding and sequence generation</article-title>. <source>Biophys Chem</source>. (<year>2022</year>) <volume>286</volume>:<fpage>106822</fpage>. <pub-id pub-id-type="doi">10.1016/j.bpc.2022.106822</pub-id><pub-id pub-id-type="pmid">35605495</pub-id></citation></ref>
<ref id="B66">
<label>66.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>R</given-names></name> <name><surname>Wu</surname> <given-names>F</given-names></name> <name><surname>Zhang</surname> <given-names>C</given-names></name> <name><surname>Zhang</surname> <given-names>L</given-names></name></person-group>. <article-title>iEnhancer-GAN: a deep learning framework in combination with word embedding and sequence generative adversarial net to identify enhancers and their strength</article-title>. <source>Int J Mol Sci</source>. (<year>2021</year>) <volume>22</volume>:<fpage>3589</fpage>. <pub-id pub-id-type="doi">10.3390/ijms22073589</pub-id><pub-id pub-id-type="pmid">33808317</pub-id></citation></ref>
<ref id="B67">
<label>67.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>J</given-names></name> <name><surname>Wu</surname> <given-names>Z</given-names></name> <name><surname>Lin</surname> <given-names>W</given-names></name> <name><surname>Luo</surname> <given-names>J</given-names></name> <name><surname>Zhang</surname> <given-names>J</given-names></name> <name><surname>Chen</surname> <given-names>Q</given-names></name> <etal/></person-group>. <article-title>iEnhancer-ELM: improve enhancer identification by extracting position-related multiscale contextual information based on enhancer language models</article-title>. <source>Bioinform Adv</source>. (<year>2023</year>) <volume>3</volume>:<fpage>vbad043</fpage>. <pub-id pub-id-type="doi">10.1093/bioadv/vbad043</pub-id><pub-id pub-id-type="pmid">37113248</pub-id></citation></ref>
<ref id="B68">
<label>68.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Omar</surname> <given-names>N</given-names></name> <name><surname>Wong YS Li</surname> <given-names>X</given-names></name> <name><surname>Chong</surname> <given-names>YL</given-names></name> <name><surname>Abdullah</surname> <given-names>MT</given-names></name> <name><surname>Lee</surname> <given-names>NK</given-names></name></person-group>. <article-title>Enhancer prediction in proboscis monkey genome: a comparative study</article-title>. <source>J Telecommun Electr Comput Eng</source>. (<year>2017</year>) <volume>9</volume>:<fpage>175</fpage>&#x02013;<lpage>9</lpage>.</citation>
</ref>
<ref id="B69">
<label>69.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>B</given-names></name> <name><surname>Fang</surname> <given-names>L</given-names></name> <name><surname>Long</surname> <given-names>R</given-names></name> <name><surname>Lan</surname> <given-names>X</given-names></name> <name><surname>Chou</surname> <given-names>KC</given-names></name></person-group>. <source>iEnhancer-2L: A Two-Layer Predictor for Identifying Enhancers and Their Strength by Pseudo k-Tuple Nucleotide Composition</source>. <publisher-loc>Oxford</publisher-loc>: <publisher-name>Oxford University Press.</publisher-name> (<year>2016</year>). <pub-id pub-id-type="doi">10.1093/bioinformatics/btv604</pub-id><pub-id pub-id-type="pmid">26476782</pub-id></citation></ref>
<ref id="B70">
<label>70.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jia</surname> <given-names>C</given-names></name> <name><surname>He</surname> <given-names>W</given-names></name></person-group>. <article-title>EnhancerPred: a predictor for discovering enhancers based on the combination and selection of multiple features</article-title>. <source>Sci Rep</source>. (<year>2016</year>) <volume>6</volume>:<fpage>38741</fpage>. <pub-id pub-id-type="doi">10.1038/srep38741</pub-id><pub-id pub-id-type="pmid">27941893</pub-id></citation></ref>
<ref id="B71">
<label>71.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>B</given-names></name> <name><surname>Li</surname> <given-names>K</given-names></name> <name><surname>Huang</surname> <given-names>DS</given-names></name> <name><surname>Chou</surname> <given-names>KC</given-names></name></person-group>. <article-title>iEnhancer-EL: identifying enhancers and their strength with ensemble learning approach</article-title>. <source>Bioinformatics</source>. (<year>2018</year>) <volume>34</volume>:<fpage>3835</fpage>&#x02013;<lpage>42</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty458</pub-id><pub-id pub-id-type="pmid">29878118</pub-id></citation></ref>
<ref id="B72">
<label>72.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>B</given-names></name></person-group>. <article-title>iEnhancer-PsedeKNC: Identification of enhancers and &#x00040;articlebgroups based on Pseudo degenerate kMER nucleotide composition</article-title>. <source>Neurocomputing</source>. (<year>2016</year>) <volume>217</volume>:<fpage>46</fpage>&#x02013;<lpage>52</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2015.12.138</pub-id></citation>
</ref>
<ref id="B73">
<label>73.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Asim</surname> <given-names>MN</given-names></name> <name><surname>Ibrahim</surname> <given-names>MA</given-names></name> <name><surname>Malik</surname> <given-names>MI</given-names></name> <name><surname>Dengel</surname> <given-names>A</given-names></name> <name><surname>Ahmed</surname> <given-names>S</given-names></name></person-group>. <article-title>Enhancer-DSNet: a supervisedly prepared enriched sequence representation for the identification of enhancers and their strength</article-title>. In: <source>International Conference on Neural Information Processing</source>. <publisher-loc>Springer</publisher-loc> (<year>2020</year>). p. <fpage>38</fpage>&#x02013;<lpage>48</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-030-63836-8_4</pub-id></citation>
</ref>
<ref id="B74">
<label>74.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>W</given-names></name> <name><surname>Jia</surname> <given-names>C</given-names></name></person-group>. <article-title>EnhancerPred2</article-title>.0: predicting enhancers and their strength based on position-specific trinucleotide propensity and electron-ion interaction potential feature selection. <source>Molec Biosyst</source>. (<year>2017</year>) <volume>13</volume>:<fpage>767</fpage>&#x02013;<lpage>74</lpage>. <pub-id pub-id-type="doi">10.1039/C7MB00054E</pub-id><pub-id pub-id-type="pmid">28239713</pub-id></citation></ref>
<ref id="B75">
<label>75.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Q</given-names></name> <name><surname>Chen</surname> <given-names>P</given-names></name> <name><surname>Wang</surname> <given-names>B</given-names></name> <name><surname>Zhang</surname> <given-names>J</given-names></name> <name><surname>Li</surname> <given-names>J</given-names></name></person-group>. <article-title>Hot spot prediction in protein-protein interactions by an ensemble system</article-title>. <source>BMC Syst Biol</source>. (<year>2018</year>) <volume>12</volume>:<fpage>89</fpage>&#x02013;<lpage>99</lpage>. <pub-id pub-id-type="doi">10.1186/s12918-018-0665-8</pub-id><pub-id pub-id-type="pmid">30598091</pub-id></citation></ref>
<ref id="B76">
<label>76.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>P</given-names></name> <name><surname>Zhang</surname> <given-names>H</given-names></name> <name><surname>Wu</surname> <given-names>H</given-names></name></person-group>. <article-title>Ipro-wael: a comprehensive and robust framework for identifying promoters in multiple species</article-title>. <source>Nucleic Acids Res</source>. (<year>2022</year>) <volume>50</volume>:<fpage>10278</fpage>&#x02013;<lpage>89</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkac824</pub-id><pub-id pub-id-type="pmid">36161334</pub-id></citation></ref>
<ref id="B77">
<label>77.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>G</given-names></name> <name><surname>Li</surname> <given-names>J</given-names></name> <name><surname>Hu</surname> <given-names>J</given-names></name> <name><surname>Shi</surname> <given-names>JY</given-names></name></person-group>. <article-title>Recognition of cyanobacteria promoters via Siamese network-based contrastive learning under novel non-promoter generation</article-title>. <source>Brief Bioinform</source>. (<year>2024</year>) <volume>25</volume>:<fpage>bbae193</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbae193</pub-id><pub-id pub-id-type="pmid">38701419</pub-id></citation></ref>
<ref id="B78">
<label>78.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Le</surname> <given-names>NQK</given-names></name> <name><surname>Yapp</surname> <given-names>EKY</given-names></name> <name><surname>Nagasundaram</surname> <given-names>N</given-names></name> <name><surname>Yeh</surname> <given-names>HY</given-names></name></person-group>. <article-title>Classifying promoters by interpreting the hidden information of DNA sequences via deep learning and combination of continuous fasttext N-grams</article-title>. <source>Front Bioeng Biotechnol</source>. (<year>2019</year>) <volume>7</volume>:<fpage>305</fpage>. <pub-id pub-id-type="doi">10.3389/fbioe.2019.00305</pub-id><pub-id pub-id-type="pmid">31750297</pub-id></citation></ref>
<ref id="B79">
<label>79.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tahir</surname> <given-names>M</given-names></name> <name><surname>Hayat</surname> <given-names>M</given-names></name> <name><surname>Gul</surname> <given-names>S</given-names></name> <name><surname>Chong</surname> <given-names>KT</given-names></name></person-group>. <article-title>An intelligent computational model for prediction of promoters and their strength via natural language processing</article-title>. <source>Chemometr Intell Lab Syst</source>. (<year>2020</year>) <volume>202</volume>:<fpage>104034</fpage>. <pub-id pub-id-type="doi">10.1016/j.chemolab.2020.104034</pub-id></citation>
</ref>
<ref id="B80">
<label>80.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>F</given-names></name> <name><surname>Chen</surname> <given-names>J</given-names></name> <name><surname>Ge</surname> <given-names>Z</given-names></name> <name><surname>Wen</surname> <given-names>Y</given-names></name> <name><surname>Yue</surname> <given-names>Y</given-names></name> <name><surname>Hayashida</surname> <given-names>M</given-names></name> <etal/></person-group>. <article-title>Computational prediction and interpretation of both general and specific types of promoters in Escherichia coli by exploiting a stacked ensemble-learning framework</article-title>. <source>Brief Bioinform</source>. (<year>2021</year>) <volume>22</volume>:<fpage>2126</fpage>&#x02013;<lpage>40</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbaa049</pub-id><pub-id pub-id-type="pmid">32363397</pub-id></citation></ref>
<ref id="B81">
<label>81.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>B</given-names></name> <name><surname>Yang</surname> <given-names>F</given-names></name> <name><surname>Huang</surname> <given-names>DS</given-names></name> <name><surname>Chou</surname> <given-names>KC</given-names></name></person-group>. <article-title>iPromoter-2L: a two-layer predictor for identifying promoters and their types by multi-window-based PseKNC</article-title>. <source>Bioinformatics</source>. (<year>2018</year>) <volume>34</volume>:<fpage>33</fpage>&#x02013;<lpage>40</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btx579</pub-id><pub-id pub-id-type="pmid">28968797</pub-id></citation></ref>
<ref id="B82">
<label>82.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hong</surname> <given-names>Z</given-names></name> <name><surname>Zeng</surname> <given-names>X</given-names></name> <name><surname>Wei</surname> <given-names>L</given-names></name> <name><surname>Liu</surname> <given-names>X</given-names></name></person-group>. <article-title>Identifying enhancer-promoter interactions with neural network based on pre-trained DNA vectors and attention mechanism</article-title>. <source>Bioinformatics</source>. (<year>2020</year>) <volume>36</volume>:<fpage>1037</fpage>&#x02013;<lpage>43</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz694</pub-id><pub-id pub-id-type="pmid">31588505</pub-id></citation></ref>
<ref id="B83">
<label>83.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ni</surname> <given-names>Y</given-names></name> <name><surname>Fan</surname> <given-names>L</given-names></name> <name><surname>Wang</surname> <given-names>M</given-names></name> <name><surname>Zhang</surname> <given-names>N</given-names></name> <name><surname>Zuo</surname> <given-names>Y</given-names></name> <name><surname>Liao</surname> <given-names>M</given-names></name></person-group>. <article-title>EPI-mind: identifying enhancer-promoter interactions based on transformer mechanism</article-title>. <source>Interdiscipl Sci</source>. (<year>2022</year>) <volume>14</volume>:<fpage>786</fpage>&#x02013;<lpage>94</lpage>. <pub-id pub-id-type="doi">10.1007/s12539-022-00525-z</pub-id><pub-id pub-id-type="pmid">35633468</pub-id></citation></ref>
<ref id="B84">
<label>84.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Talukder</surname> <given-names>A</given-names></name> <name><surname>Saadat</surname> <given-names>S</given-names></name> <name><surname>Li</surname> <given-names>X</given-names></name> <name><surname>Hu</surname> <given-names>H</given-names></name></person-group>. <article-title>EPIP a novel approach for condition-specific enhancer-promoter interaction prediction</article-title>. <source>Bioinformatics</source>. (<year>2019</year>) <volume>35</volume>:<fpage>3877</fpage>&#x02013;<lpage>83</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz641</pub-id><pub-id pub-id-type="pmid">31410461</pub-id></citation></ref>
<ref id="B85">
<label>85.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhuang</surname> <given-names>Z</given-names></name> <name><surname>Shen</surname> <given-names>X</given-names></name> <name><surname>Pan</surname> <given-names>W</given-names></name></person-group>. <article-title>A simple convolutional neural network for prediction of enhancer-promoter interactions with DNA sequence data</article-title>. <source>Bioinformatics</source>. (<year>2019</year>) <volume>35</volume>:<fpage>2899</fpage>&#x02013;<lpage>906</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty1050</pub-id><pub-id pub-id-type="pmid">30649185</pub-id></citation></ref>
<ref id="B86">
<label>86.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Singh</surname> <given-names>S</given-names></name> <name><surname>Yang</surname> <given-names>Y</given-names></name> <name><surname>P&#x000F3;czos</surname> <given-names>B</given-names></name> <name><surname>Ma</surname> <given-names>J</given-names></name></person-group>. <article-title>Predicting enhancer-promoter interaction from genomic sequence with deep neural networks</article-title>. <source>Quant Biol</source>. (<year>2019</year>) <volume>7</volume>:<fpage>122</fpage>&#x02013;<lpage>37</lpage>. <pub-id pub-id-type="doi">10.1007/s40484-019-0154-0</pub-id><pub-id pub-id-type="pmid">34113473</pub-id></citation></ref>
<ref id="B87">
<label>87.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>S</given-names></name> <name><surname>Xu</surname> <given-names>X</given-names></name> <name><surname>Yang</surname> <given-names>Z</given-names></name> <name><surname>Zhao</surname> <given-names>X</given-names></name> <name><surname>Liu</surname> <given-names>S</given-names></name> <name><surname>Zhang</surname> <given-names>W</given-names></name></person-group>. <article-title>EPIHC: improving enhancer-promoter interaction prediction by using hybrid features and communicative learning</article-title>. In: <source>IEEE/ACM Transactions on Computational Biology and Bioinformatics</source>. (<year>2021</year>). p. <fpage>1</fpage>&#x02013;<lpage>1</lpage>.<pub-id pub-id-type="pmid">34473626</pub-id></citation></ref>
<ref id="B88">
<label>88.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>M</given-names></name> <name><surname>Hu</surname> <given-names>Y</given-names></name> <name><surname>Zhu</surname> <given-names>M</given-names></name></person-group>. <article-title>Epishilbert: prediction of enhancer-promoter interactions via hilbert curve encoding and transfer learning</article-title>. <source>Genes</source>. (<year>2021</year>) <volume>12</volume>:<fpage>1385</fpage>. <pub-id pub-id-type="doi">10.3390/genes12091385</pub-id><pub-id pub-id-type="pmid">34573367</pub-id></citation></ref>
<ref id="B89">
<label>89.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>An</surname> <given-names>W</given-names></name> <name><surname>Guo</surname> <given-names>Y</given-names></name> <name><surname>Bian</surname> <given-names>Y</given-names></name> <name><surname>Ma</surname> <given-names>H</given-names></name> <name><surname>Yang</surname> <given-names>J</given-names></name> <name><surname>Li</surname> <given-names>C</given-names></name> <etal/></person-group>. <article-title>MoDNA: motif-oriented pre-training for DNA language model</article-title>. In: <source>Proceedings of the 13th ACM International Conference on Bioinformatics, Computational Biology and Health Informatics</source>. (<year>2022</year>). p. <fpage>1</fpage>&#x02013;<lpage>5</lpage>. <pub-id pub-id-type="doi">10.1145/3535508.3545512</pub-id></citation>
</ref>
<ref id="B90">
<label>90.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mo</surname> <given-names>S</given-names></name> <name><surname>Fu</surname> <given-names>X</given-names></name> <name><surname>Hong</surname> <given-names>C</given-names></name> <name><surname>Chen</surname> <given-names>Y</given-names></name> <name><surname>Zheng</surname> <given-names>Y</given-names></name> <name><surname>Tang</surname> <given-names>X</given-names></name> <etal/></person-group>. <article-title>Multi-modal self-supervised pre-training for regulatory genome across cell types</article-title>. <source>arXiv preprint arXiv:211005231</source>. (<year>2021</year>).</citation>
</ref>
<ref id="B91">
<label>91.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Clauwaert</surname> <given-names>J</given-names></name> <name><surname>Menschaert</surname> <given-names>G</given-names></name> <name><surname>Waegeman</surname> <given-names>W</given-names></name></person-group>. <article-title>Explainability in transformer models for functional genomics</article-title>. <source>Brief Bioinform</source>. (<year>2021</year>) <volume>22</volume>:<fpage>bbab060</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbab060</pub-id><pub-id pub-id-type="pmid">33834200</pub-id></citation></ref>
<ref id="B92">
<label>92.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shen</surname> <given-names>Z</given-names></name> <name><surname>Bao</surname> <given-names>W</given-names></name> <name><surname>Huang</surname> <given-names>DS</given-names></name></person-group>. <article-title>Recurrent neural network for predicting transcription factor binding sites</article-title>. <source>Sci Rep</source>. (<year>2018</year>) <volume>8</volume>:<fpage>15270</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-018-33321-1</pub-id><pub-id pub-id-type="pmid">30323198</pub-id></citation></ref>
<ref id="B93">
<label>93.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ji</surname> <given-names>Y</given-names></name> <name><surname>Zhou</surname> <given-names>Z</given-names></name> <name><surname>Liu</surname> <given-names>H</given-names></name> <name><surname>Davuluri</surname> <given-names>RV</given-names></name></person-group>. <article-title>DNABERT pre-trained bidirectional encoder representations from transformers model for DNA-language in genome</article-title>. <source>Bioinformatics</source>. (<year>2021</year>) <volume>37</volume>:<fpage>2112</fpage>&#x02013;<lpage>20</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btab083</pub-id><pub-id pub-id-type="pmid">33538820</pub-id></citation></ref>
<ref id="B94">
<label>94.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kabir</surname> <given-names>A</given-names></name> <name><surname>Bhattarai</surname> <given-names>M</given-names></name> <name><surname>Rasmussen</surname> <given-names>K&#x000D8;</given-names></name> <name><surname>Shehu</surname> <given-names>A</given-names></name> <name><surname>Bishop</surname> <given-names>AR</given-names></name> <name><surname>Alexandrov</surname> <given-names>B</given-names></name> <etal/></person-group>. <article-title>Advancing transcription factor binding site prediction using DNA breathing dynamics and sequence transformers via cross attention</article-title>. <source>bioRxiv</source>. (<year>2024</year>). p. <fpage>2024</fpage>&#x02013;<lpage>01</lpage>. <pub-id pub-id-type="doi">10.1101/2024.01.16.575935</pub-id><pub-id pub-id-type="pmid">38293094</pub-id></citation></ref>
<ref id="B95">
<label>95.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Murad</surname> <given-names>T</given-names></name> <name><surname>Ali</surname> <given-names>S</given-names></name> <name><surname>Chourasia</surname> <given-names>P</given-names></name> <name><surname>Patterson</surname> <given-names>M</given-names></name></person-group>. <article-title>Advancing protein-DNA binding site prediction: integrating sequence models and machine learning classifiers</article-title>. <source>bioRxiv</source>. (<year>2023</year>). p. <fpage>2023</fpage>&#x02013;<lpage>08</lpage>. <pub-id pub-id-type="doi">10.1101/2023.08.23.554389</pub-id><pub-id pub-id-type="pmid">39985437</pub-id></citation></ref>
<ref id="B96">
<label>96.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Luo</surname> <given-names>H</given-names></name> <name><surname>Shan</surname> <given-names>W</given-names></name> <name><surname>Chen</surname> <given-names>C</given-names></name> <name><surname>Ding</surname> <given-names>P</given-names></name> <name><surname>Luo</surname> <given-names>L</given-names></name></person-group>. <article-title>Improving language model of human genome for DNA-protein binding prediction based on task-specific pre-training</article-title>. <source>Interdiscipl Sci</source>. (<year>2023</year>) <volume>15</volume>:<fpage>32</fpage>&#x02013;<lpage>43</lpage>. <pub-id pub-id-type="doi">10.1007/s12539-022-00537-9</pub-id><pub-id pub-id-type="pmid">36136096</pub-id></citation></ref>
<ref id="B97">
<label>97.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sha</surname> <given-names>M</given-names></name> <name><surname>Rahamathulla</surname> <given-names>MP</given-names></name></person-group>. <article-title>Splice site recognition-deciphering Exon-Intron transitions for genetic insights using Enhanced integrated Block-Level gated LSTM model</article-title>. <source>Gene</source>. (<year>2024</year>) <volume>915</volume>:<fpage>148429</fpage>. <pub-id pub-id-type="doi">10.1016/j.gene.2024.148429</pub-id><pub-id pub-id-type="pmid">38575098</pub-id></citation></ref>
<ref id="B98">
<label>98.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kabanga</surname> <given-names>E</given-names></name> <name><surname>Yun</surname> <given-names>S</given-names></name> <name><surname>Van Messem</surname> <given-names>A</given-names></name> <name><surname>De Neve</surname> <given-names>W</given-names></name></person-group>. <article-title>Impact of U2-type introns on splice site prediction in Arabidopsis thaliana using deep learning</article-title>. <source>bioRxiv</source>. (<year>2024</year>). p. <fpage>2024</fpage>&#x02013;<lpage>05</lpage>. <pub-id pub-id-type="doi">10.1101/2024.05.13.593811</pub-id></citation>
</ref>
<ref id="B99">
<label>99.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>X</given-names></name> <name><surname>Zhang</surname> <given-names>H</given-names></name> <name><surname>Zeng</surname> <given-names>Y</given-names></name> <name><surname>Zhu</surname> <given-names>X</given-names></name> <name><surname>Zhu</surname> <given-names>L</given-names></name> <name><surname>Fu</surname> <given-names>J</given-names></name></person-group>. <article-title>DRANetSplicer: a splice site prediction model based on deep residual attention networks</article-title>. <source>Genes</source>. (<year>2024</year>) <volume>15</volume>:<fpage>404</fpage>. <pub-id pub-id-type="doi">10.3390/genes15040404</pub-id><pub-id pub-id-type="pmid">38674339</pub-id></citation></ref>
<ref id="B100">
<label>100.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dalla-Torre</surname> <given-names>H</given-names></name> <name><surname>Gonzalez</surname> <given-names>L</given-names></name> <name><surname>Mendoza-Revilla</surname> <given-names>J</given-names></name> <name><surname>Carranza</surname> <given-names>NL</given-names></name> <name><surname>Grzywaczewski</surname> <given-names>AH</given-names></name> <name><surname>Oteri</surname> <given-names>F</given-names></name> <etal/></person-group>. <article-title>The nucleotide transformer: Building and evaluating robust foundation models for human genomics</article-title>. <source>bioRxiv</source>. (<year>2023</year>). p. <fpage>2023</fpage>&#x02013;<lpage>01</lpage>. <pub-id pub-id-type="doi">10.1101/2023.01.11.523679</pub-id><pub-id pub-id-type="pmid">39609566</pub-id></citation></ref>
<ref id="B101">
<label>101.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>W</given-names></name> <name><surname>Guo</surname> <given-names>Y</given-names></name> <name><surname>Wang</surname> <given-names>B</given-names></name> <name><surname>Yang</surname> <given-names>B</given-names></name></person-group>. <article-title>Learning spatiotemporal embedding with gated convolutional recurrent networks for translation initiation site prediction</article-title>. <source>Pattern Recognit</source>. (<year>2023</year>) <volume>136</volume>:<fpage>109234</fpage>. <pub-id pub-id-type="doi">10.1016/j.patcog.2022.109234</pub-id></citation>
</ref>
<ref id="B102">
<label>102.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Reddy</surname> <given-names>AJ</given-names></name> <name><surname>Herschl</surname> <given-names>MH</given-names></name> <name><surname>Kolli</surname> <given-names>S</given-names></name> <name><surname>Lu</surname> <given-names>AX</given-names></name> <name><surname>Geng</surname> <given-names>X</given-names></name> <name><surname>Kumar</surname> <given-names>A</given-names></name> <etal/></person-group>. <article-title>Strategies for effectively modelling promoter-driven gene expression using transfer learning</article-title>. <source>bioRxiv</source>. (<year>2023</year>). <pub-id pub-id-type="doi">10.1101/2023.02.24.529941</pub-id><pub-id pub-id-type="pmid">36909524</pub-id></citation></ref>
<ref id="B103">
<label>103.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Al Taweraqi</surname> <given-names>N</given-names></name> <name><surname>King</surname> <given-names>RD</given-names></name></person-group>. <article-title>Improved prediction of gene expression through integrating cell signalling models with machine learning</article-title>. <source>BMC Bioinform</source>. (<year>2022</year>) <volume>23</volume>:<fpage>323</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-022-04787-8</pub-id><pub-id pub-id-type="pmid">35933367</pub-id></citation></ref>
<ref id="B104">
<label>104.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sharma</surname> <given-names>K</given-names></name> <name><surname>Marucci</surname> <given-names>L</given-names></name> <name><surname>Abdallah</surname> <given-names>ZS</given-names></name></person-group>. <article-title>FluxGAT: integrating flux sampling with graph neural networks for unbiased gene essentiality classification</article-title>. <source>arXiv preprint arXiv:240318666</source>. (<year>2024</year>).</citation>
</ref>
<ref id="B105">
<label>105.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Le</surname> <given-names>NQK</given-names></name> <name><surname>Do</surname> <given-names>DT</given-names></name> <name><surname>Hung</surname> <given-names>TNK</given-names></name> <name><surname>Lam</surname> <given-names>LHT</given-names></name> <name><surname>Huynh</surname> <given-names>TT</given-names></name> <name><surname>Nguyen</surname> <given-names>NTK</given-names></name> <etal/></person-group>. <article-title>computational framework based on ensemble deep neural networks for essential genes identification</article-title>. <source>Int J Mol Sci</source>. (<year>2020</year>) <volume>21</volume>:<fpage>9070</fpage>. <pub-id pub-id-type="doi">10.3390/ijms21239070</pub-id><pub-id pub-id-type="pmid">33260643</pub-id></citation></ref>
<ref id="B106">
<label>106.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>X</given-names></name> <name><surname>Xiao</surname> <given-names>W</given-names></name> <name><surname>Xiao</surname> <given-names>W</given-names></name></person-group>. <article-title>DeepHE: Accurately predicting human essential genes based on deep learning</article-title>. <source>PLoS Comput Biol</source>. (<year>2020</year>) <volume>16</volume>:<fpage>e1008229</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1008229</pub-id><pub-id pub-id-type="pmid">32936825</pub-id></citation></ref>
<ref id="B107">
<label>107.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xiao</surname> <given-names>W</given-names></name> <name><surname>Zhang</surname> <given-names>X</given-names></name> <name><surname>Xiao</surname> <given-names>W</given-names></name></person-group>. <article-title>A Deep Learning Framework for Predicting Human Essential Genes by Integrating Sequence and Functional data</article-title>. <source>bioRxiv</source>. (<year>2020</year>). p. <fpage>2020</fpage>&#x02013;<lpage>08</lpage>.</citation>
</ref>
<ref id="B108">
<label>108.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schapke</surname> <given-names>J</given-names></name> <name><surname>Tavares</surname> <given-names>A</given-names></name> <name><surname>Recamonde-Mendoza</surname> <given-names>M</given-names></name></person-group>. <article-title>EPGAT: gene essentiality prediction with graph attention networks</article-title>. <source>IEEE/ACM Trans Comput Biol Bioinform</source>. (<year>2021</year>) <volume>19</volume>:<fpage>1615</fpage>&#x02013;<lpage>26</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2021.3054738</pub-id><pub-id pub-id-type="pmid">33497339</pub-id></citation></ref>
<ref id="B109">
<label>109.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ma</surname> <given-names>J</given-names></name> <name><surname>Song</surname> <given-names>J</given-names></name> <name><surname>Young</surname> <given-names>ND</given-names></name> <name><surname>Chang</surname> <given-names>BC</given-names></name> <name><surname>Korhonen</surname> <given-names>PK</given-names></name> <name><surname>Campos</surname> <given-names>TL</given-names></name> <etal/></person-group>. <article-title>&#x02018;Bingo&#x00027;&#x02013;a large language model-and graph neural network-based workflow for the prediction of essential genes from protein data</article-title>. <source>Brief Bioinform</source>. (<year>2024</year>) <volume>25</volume>:<fpage>bbad472</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbad472</pub-id><pub-id pub-id-type="pmid">38152979</pub-id></citation></ref>
<ref id="B110">
<label>110.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nunes</surname> <given-names>S</given-names></name> <name><surname>Sousa</surname> <given-names>RT</given-names></name> <name><surname>Pesquita</surname> <given-names>C</given-names></name></person-group>. <article-title>Multi-domain knowledge graph embeddings for gene-disease association prediction</article-title>. <source>J Biomed Semantics</source>. (<year>2023</year>) <volume>14</volume>:<fpage>11</fpage>. <pub-id pub-id-type="doi">10.1186/s13326-023-00291-x</pub-id><pub-id pub-id-type="pmid">37580835</pub-id></citation></ref>
<ref id="B111">
<label>111.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fan</surname> <given-names>K</given-names></name> <name><surname>Zhang</surname> <given-names>Y</given-names></name></person-group>. <article-title>Pseudo2GO: a graph-based deep learning method for pseudogene function prediction by borrowing information from coding genes</article-title>. <source>Front Genet</source>. (<year>2020</year>) <volume>11</volume>:<fpage>538028</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2020.00807</pub-id><pub-id pub-id-type="pmid">33014009</pub-id></citation></ref>
<ref id="B112">
<label>112.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>M</given-names></name> <name><surname>Alkhairy</surname> <given-names>S</given-names></name> <name><surname>Lee</surname> <given-names>I</given-names></name> <name><surname>Pillich</surname> <given-names>RT</given-names></name> <name><surname>Fong</surname> <given-names>D</given-names></name> <name><surname>Smith</surname> <given-names>K</given-names></name> <etal/></person-group>. <article-title>Evaluation of large language models for discovery of gene set function</article-title>. <source>Nat Methods</source>. (<year>2023</year>) <volume>22</volume>:<fpage>82</fpage>&#x02013;<lpage>91</lpage>. <pub-id pub-id-type="doi">10.1038/s41592-024-02525-x</pub-id><pub-id pub-id-type="pmid">39609565</pub-id></citation></ref>
<ref id="B113">
<label>113.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Arango-Argoty</surname> <given-names>GA</given-names></name> <name><surname>Heath</surname> <given-names>LS</given-names></name> <name><surname>Pruden</surname> <given-names>A</given-names></name> <name><surname>Vikesland</surname> <given-names>PJ</given-names></name> <name><surname>Zhang</surname> <given-names>L</given-names></name></person-group>. <article-title>MetaMLP: a fast word embedding based classifier to profile target gene databases in metagenomic samples</article-title>. <source>J Comput Biol</source>. (<year>2021</year>) <volume>28</volume>:<fpage>1063</fpage>&#x02013;<lpage>74</lpage>. <pub-id pub-id-type="doi">10.1089/cmb.2021.0273</pub-id><pub-id pub-id-type="pmid">34665648</pub-id></citation></ref>
<ref id="B114">
<label>114.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Toufiq</surname> <given-names>M</given-names></name> <name><surname>Rinchai</surname> <given-names>D</given-names></name> <name><surname>Bettacchioli</surname> <given-names>E</given-names></name> <name><surname>Kabeer</surname> <given-names>BSA</given-names></name> <name><surname>Khan</surname> <given-names>T</given-names></name> <name><surname>Subba</surname> <given-names>B</given-names></name> <etal/></person-group>. <article-title>Harnessing large language models (LLMs) for candidate gene prioritization and selection</article-title>. <source>J Transl Med</source>. (<year>2023</year>) <volume>21</volume>:<fpage>728</fpage>. <pub-id pub-id-type="doi">10.1186/s12967-023-04576-8</pub-id><pub-id pub-id-type="pmid">37845713</pub-id></citation></ref>
<ref id="B115">
<label>115.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ai</surname> <given-names>H</given-names></name></person-group>. <article-title>Gsea-sdbe: a gene selection method for breast cancer classification based on gsea and analyzing differences in performance metrics</article-title>. <source>PLoS ONE</source>. (<year>2022</year>) <volume>17</volume>:<fpage>e0263171</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0263171</pub-id><pub-id pub-id-type="pmid">35472078</pub-id></citation></ref>
<ref id="B116">
<label>116.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hulsen</surname> <given-names>T</given-names></name> <name><surname>Huynen</surname> <given-names>M</given-names></name> <name><surname>Vlieg</surname> <given-names>J</given-names></name> <name><surname>Groenen</surname> <given-names>P</given-names></name></person-group>. <article-title>Untitled</article-title>. <source>Genome Biol</source>. (<year>2006</year>) <volume>7</volume>:<fpage>R31</fpage>. <pub-id pub-id-type="doi">10.1186/gb-2006-7-4-r31</pub-id><pub-id pub-id-type="pmid">16613613</pub-id></citation></ref>
<ref id="B117">
<label>117.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Abbasi</surname> <given-names>AF</given-names></name> <name><surname>Asim</surname> <given-names>MN</given-names></name> <name><surname>Ahmed</surname> <given-names>S</given-names></name> <name><surname>Vollmer</surname> <given-names>S</given-names></name> <name><surname>Dengel</surname> <given-names>A</given-names></name></person-group>. <article-title>Survival prediction landscape: an in-depth systematic literature review on activities, methods, tools, diseases, and databases</article-title>. <source>medRxiv</source>. (<year>2024</year>). p. <fpage>2024</fpage>&#x02013;<lpage>01</lpage>. <pub-id pub-id-type="doi">10.1101/2024.01.05.24300889</pub-id><pub-id pub-id-type="pmid">39021434</pub-id></citation></ref>
<ref id="B118">
<label>118.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Deng</surname> <given-names>F</given-names></name> <name><surname>Yang</surname> <given-names>K</given-names></name> <name><surname>Zheng</surname> <given-names>G</given-names></name></person-group>. <article-title>Period family of clock genes as novel predictors of survival in human cancer: a systematic review and meta-analysis</article-title>. <source>Dis Markers</source>. (<year>2020</year>) <volume>2020</volume>:<fpage>1</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1155/2020/6486238</pub-id><pub-id pub-id-type="pmid">32849922</pub-id></citation></ref>
<ref id="B119">
<label>119.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zeng</surname> <given-names>W</given-names></name> <name><surname>Rao</surname> <given-names>N</given-names></name> <name><surname>Zheng</surname> <given-names>J</given-names></name> <name><surname>Wan</surname> <given-names>Y</given-names></name> <name><surname>Wang</surname> <given-names>G</given-names></name> <name><surname>Li</surname> <given-names>Z</given-names></name> <etal/></person-group>. <article-title>Identification of candidate targeted genes in molecular subtypes of gastric cancer</article-title>. <source>J Biomed Sci Eng</source>. (<year>2017</year>) <volume>10</volume>:<fpage>45</fpage>&#x02013;<lpage>53</lpage>. <pub-id pub-id-type="doi">10.4236/jbise.2017.105B005</pub-id></citation>
</ref>
<ref id="B120">
<label>120.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Da&#x0011F;liyan</surname> <given-names>O</given-names></name> <name><surname>&#x000DC;ney Y&#x000FC;ksektepe</surname> <given-names>F</given-names></name> <name><surname>Kavakl&#x000ED;</surname></name> <name><surname>T&#x000FC;rkay</surname> <given-names>M</given-names></name></person-group>. <article-title>Optimization based tumor classification from microarray gene expression data</article-title>. <source>PLoS ONE</source>. (<year>2011</year>) <volume>6</volume>:<fpage>e14579</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0014579</pub-id><pub-id pub-id-type="pmid">21326602</pub-id></citation></ref>
<ref id="B121">
<label>121.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Verma</surname> <given-names>B</given-names></name> <name><surname>Parkinson</surname> <given-names>J</given-names></name></person-group>. <article-title>HiTaxon: a hierarchical ensemble framework for taxonomic classification of short reads</article-title>. <source>Bioinform Adv</source>. (<year>2024</year>) <volume>4</volume>:<fpage>vbae016</fpage>. <pub-id pub-id-type="doi">10.1093/bioadv/vbae016</pub-id><pub-id pub-id-type="pmid">38371920</pub-id></citation></ref>
<ref id="B122">
<label>122.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shi</surname> <given-names>L</given-names></name> <name><surname>Chen</surname> <given-names>B</given-names></name></person-group>. <article-title>LSHvec: a vector representation of DNA sequences using locality sensitive hashing and FastText word embeddings</article-title>. In: <source>Proceedings of the 12th ACM Conference on Bioinformatics, Computational Biology, and Health Informatics</source>. (<year>2021</year>). p. <fpage>1</fpage>&#x02013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1145/3459930.3469521</pub-id></citation>
</ref>
<ref id="B123">
<label>123.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mock</surname> <given-names>F</given-names></name> <name><surname>Kretschmer</surname> <given-names>F</given-names></name> <name><surname>Kriese</surname> <given-names>A</given-names></name> <name><surname>B&#x000F6;cker</surname> <given-names>S</given-names></name> <name><surname>Marz</surname> <given-names>M</given-names></name></person-group>. <article-title>Taxonomic classification of DNA sequences beyond sequence similarity using deep neural networks</article-title>. <source>Proc Natl Acad Sci</source>. (<year>2022</year>) <volume>119</volume>:<fpage>e2122636119</fpage>. <pub-id pub-id-type="doi">10.1073/pnas.2122636119</pub-id><pub-id pub-id-type="pmid">36018838</pub-id></citation></ref>
<ref id="B124">
<label>124.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Coenye</surname> <given-names>T</given-names></name> <name><surname>Gevers</surname> <given-names>D</given-names></name> <name><surname>Peer</surname> <given-names>Y</given-names></name> <name><surname>Vandamme</surname> <given-names>P</given-names></name> <name><surname>Swings</surname> <given-names>J</given-names></name></person-group>. <article-title>Towards a prokaryotic genomic taxonomy</article-title>. <source>FEMS Microbiol Rev</source>. (<year>2005</year>) <volume>29</volume>:<fpage>147</fpage>&#x02013;<lpage>67</lpage>. <pub-id pub-id-type="doi">10.1016/j.femsre.2004.11.004</pub-id><pub-id pub-id-type="pmid">15808739</pub-id></citation></ref>
<ref id="B125">
<label>125.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Commichaux</surname> <given-names>S</given-names></name> <name><surname>Luan</surname> <given-names>T</given-names></name> <name><surname>Muralidharan</surname> <given-names>HS</given-names></name> <name><surname>Pop</surname> <given-names>M</given-names></name></person-group>. <article-title>Database size positively correlates with the loss of species-level taxonomic resolution for the 16S rRNA and other prokaryotic marker genes</article-title>. <source>bioRxiv</source>. (<year>2023</year>) <volume>14</volume>:<fpage>571439</fpage>. <pub-id pub-id-type="doi">10.1101/2023.12.13.571439</pub-id><pub-id pub-id-type="pmid">38168205</pub-id></citation></ref>
<ref id="B126">
<label>126.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pio</surname> <given-names>G</given-names></name> <name><surname>Mignone</surname> <given-names>P</given-names></name> <name><surname>Magazz&#x000FA;</surname> <given-names>G</given-names></name> <name><surname>Zampieri</surname> <given-names>G</given-names></name> <name><surname>Ceci</surname> <given-names>M</given-names></name> <name><surname>Angione</surname> <given-names>C</given-names></name></person-group>. <article-title>Integrating genome-scale metabolic modelling and transfer learning for human gene regulatory network reconstruction</article-title>. <source>Bioinformatics</source>. (<year>2022</year>) <volume>38</volume>:<fpage>487</fpage>&#x02013;<lpage>93</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btab647</pub-id><pub-id pub-id-type="pmid">34499112</pub-id></citation></ref>
<ref id="B127">
<label>127.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Du</surname> <given-names>Z</given-names></name> <name><surname>Zhong</surname> <given-names>X</given-names></name> <name><surname>Wang</surname> <given-names>F</given-names></name> <name><surname>Uversky</surname> <given-names>VN</given-names></name></person-group>. <article-title>Inference of gene regulatory networks based on the light gradient boosting machine</article-title>. <source>Comput Biol Chem</source>. (<year>2022</year>) <volume>101</volume>:<fpage>107769</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiolchem.2022.107769</pub-id><pub-id pub-id-type="pmid">36182867</pub-id></citation></ref>
<ref id="B128">
<label>128.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pio</surname> <given-names>G</given-names></name> <name><surname>Ceci</surname> <given-names>M</given-names></name> <name><surname>Prisciandaro</surname> <given-names>F</given-names></name> <name><surname>Malerba</surname> <given-names>D</given-names></name></person-group>. <article-title>Exploiting causality in gene network reconstruction based on graph embedding</article-title>. <source>Mach Learn</source>. (<year>2020</year>) <volume>109</volume>:<fpage>1231</fpage>&#x02013;<lpage>79</lpage>. <pub-id pub-id-type="doi">10.1007/s10994-019-05861-8</pub-id></citation>
</ref>
<ref id="B129">
<label>129.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Langfelder</surname> <given-names>P</given-names></name> <name><surname>Horvath</surname> <given-names>S</given-names></name></person-group>. <article-title>WGCNA: an r package for weighted correlation network analysis</article-title>. <source>BMC Bioinform</source>. (<year>2008</year>) <volume>9</volume>:<fpage>559</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-9-559</pub-id><pub-id pub-id-type="pmid">19114008</pub-id></citation></ref>
<ref id="B130">
<label>130.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Stanley</surname> <given-names>D</given-names></name> <name><surname>Watson-Haigh</surname> <given-names>N</given-names></name> <name><surname>Cowled</surname> <given-names>C</given-names></name> <name><surname>Moore</surname> <given-names>R</given-names></name></person-group>. <article-title>Genetic architecture of gene expression in the chicken</article-title>. <source>BMC Gen</source>. (<year>2013</year>) <volume>14</volume>:<fpage>13</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2164-14-13</pub-id><pub-id pub-id-type="pmid">23324119</pub-id></citation></ref>
<ref id="B131">
<label>131.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nabeel Asim</surname> <given-names>M</given-names></name> <name><surname>Ali Ibrahim</surname> <given-names>M</given-names></name> <name><surname>Fazeel</surname> <given-names>A</given-names></name> <name><surname>Dengel</surname> <given-names>A</given-names></name> <name><surname>Ahmed</surname> <given-names>S</given-names></name></person-group>. <article-title>DNA-MP: a generalized DNA modifications predictor for multiple species based on powerful sequence encoding method</article-title>. <source>Brief Bioinform</source>. (<year>2023</year>) <volume>24</volume>:<fpage>bbac546</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbac546</pub-id><pub-id pub-id-type="pmid">36528802</pub-id></citation></ref>
<ref id="B132">
<label>132.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Asim</surname> <given-names>MN</given-names></name> <name><surname>Malik</surname> <given-names>MI</given-names></name> <name><surname>Dengel</surname> <given-names>A</given-names></name> <name><surname>Ahmed</surname> <given-names>S</given-names></name></person-group>. <article-title>K-MER neural embedding performance analysis using amino acid codons</article-title>. In: <source>2020 International Joint Conference on Neural Networks (IJCNN)</source>. <publisher-loc>IEEE</publisher-loc> (<year>2020</year>). p. <fpage>1</fpage>&#x02013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1109/IJCNN48605.2020.9206892</pub-id></citation>
</ref>
<ref id="B133">
<label>133.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Asim</surname> <given-names>MN</given-names></name> <name><surname>Ibrahim</surname> <given-names>MA</given-names></name> <name><surname>Malik</surname> <given-names>MI</given-names></name> <name><surname>Razzak</surname> <given-names>I</given-names></name> <name><surname>Dengel</surname> <given-names>A</given-names></name> <name><surname>Ahmed</surname> <given-names>S</given-names></name></person-group>. <article-title>Histone-net: a multi-paradigm computational framework for histone occupancy and modification prediction</article-title>. <source>Complex Intell Syst</source>. (<year>2023</year>) <volume>9</volume>:<fpage>399</fpage>&#x02013;<lpage>419</lpage>. <pub-id pub-id-type="doi">10.1007/s40747-022-00802-w</pub-id></citation>
</ref>
<ref id="B134">
<label>134.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xie</surname> <given-names>H</given-names></name> <name><surname>Ding</surname> <given-names>Y</given-names></name> <name><surname>Qian</surname> <given-names>Y</given-names></name> <name><surname>Tiwari</surname> <given-names>P</given-names></name> <name><surname>Guo</surname> <given-names>F</given-names></name></person-group>. <article-title>Structured sparse regularization based random vector functional link networks for DNA N4-methylcytosine sites prediction</article-title>. <source>Expert Syst Appl</source>. (<year>2024</year>) <volume>235</volume>:<fpage>121157</fpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2023.121157</pub-id></citation>
</ref>
<ref id="B135">
<label>135.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Saha</surname> <given-names>S</given-names></name> <name><surname>Halder</surname> <given-names>RK</given-names></name> <name><surname>Uddin</surname> <given-names>MN</given-names></name></person-group>. <article-title>Particle swarm optimization-assisted multilayer ensemble model to predict DNA 4mC sites</article-title>. <source>Inform Med Unlocked</source>. (<year>2023</year>) <volume>42</volume>:<fpage>101374</fpage>. <pub-id pub-id-type="doi">10.1016/j.imu.2023.101374</pub-id></citation>
</ref>
<ref id="B136">
<label>136.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zulfiqar</surname> <given-names>H</given-names></name> <name><surname>Sun</surname> <given-names>ZJ</given-names></name> <name><surname>Huang</surname> <given-names>QL</given-names></name> <name><surname>Yuan</surname> <given-names>SS</given-names></name> <name><surname>Lv</surname> <given-names>H</given-names></name> <name><surname>Dao</surname> <given-names>FY</given-names></name> <etal/></person-group>. <article-title>Deep-4mCW2V: A sequence-based predictor to identify N4-methylcytosine sites in <italic>Escherichia coli</italic></article-title>. <source>Methods</source>. (<year>2022</year>) <volume>203</volume>:<fpage>558</fpage>&#x02013;<lpage>63</lpage>. <pub-id pub-id-type="doi">10.1016/j.ymeth.2021.07.011</pub-id><pub-id pub-id-type="pmid">34352373</pub-id></citation></ref>
<ref id="B137">
<label>137.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fang</surname> <given-names>G</given-names></name> <name><surname>Zeng</surname> <given-names>F</given-names></name> <name><surname>Li</surname> <given-names>X</given-names></name> <name><surname>Yao</surname> <given-names>L</given-names></name></person-group>. <article-title>Word2vec based deep learning network for DNA N4-methylcytosine sites identification</article-title>. <source>Procedia Comput Sci</source>. (<year>2021</year>) <volume>187</volume>:<fpage>270</fpage>&#x02013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1016/j.procs.2021.04.062</pub-id></citation>
</ref>
<ref id="B138">
<label>138.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Khanal</surname> <given-names>J</given-names></name> <name><surname>Tayara</surname> <given-names>H</given-names></name> <name><surname>Zou</surname> <given-names>Q</given-names></name> <name><surname>Chong</surname> <given-names>KT</given-names></name></person-group>. <article-title>Identifying DNA N4-methylcytosine sites in the rosaceae genome with a deep learning model relying on distributed feature representation</article-title>. <source>Comput Struct Biotechnol J</source>. (<year>2021</year>) <volume>19</volume>:<fpage>1612</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1016/j.csbj.2021.03.015</pub-id><pub-id pub-id-type="pmid">33868598</pub-id></citation></ref>
<ref id="B139">
<label>139.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nguyen-Vo</surname> <given-names>TH</given-names></name> <name><surname>Trinh</surname> <given-names>QH</given-names></name> <name><surname>Nguyen</surname> <given-names>L</given-names></name> <name><surname>Nguyen-Hoang</surname> <given-names>PU</given-names></name> <name><surname>Rahardja</surname> <given-names>S</given-names></name> <name><surname>Nguyen</surname> <given-names>BP</given-names></name></person-group>. <article-title>i4mC-GRU: Identifying DNA N4-Methylcytosine sites in mouse genomes using bidirectional gated recurrent unit and sequence-embedded features</article-title>. <source>Comput Struct Biotechnol J</source>. (<year>2023</year>) <volume>21</volume>:<fpage>3045</fpage>&#x02013;<lpage>53</lpage>. <pub-id pub-id-type="doi">10.1016/j.csbj.2023.05.014</pub-id><pub-id pub-id-type="pmid">37273848</pub-id></citation></ref>
<ref id="B140">
<label>140.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zeng</surname> <given-names>R</given-names></name> <name><surname>Liao</surname> <given-names>M</given-names></name></person-group>. <article-title>Developing a multi-layer deep learning based predictive model to identify DNA N4-methylcytosine modifications</article-title>. <source>Front Bioeng Biotechnol</source>. (<year>2020</year>) <volume>8</volume>:<fpage>274</fpage>. <pub-id pub-id-type="doi">10.3389/fbioe.2020.00274</pub-id><pub-id pub-id-type="pmid">32373597</pub-id></citation></ref>
<ref id="B141">
<label>141.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>H</given-names></name> <name><surname>Jia</surname> <given-names>P</given-names></name> <name><surname>Zhao</surname> <given-names>Z</given-names></name></person-group>. <article-title>Deep4mC: systematic assessment and computational prediction for DNA N4-methylcytosine sites by deep learning</article-title>. <source>Brief Bioinform</source>. (<year>2021</year>) <volume>22</volume>:<fpage>bbaa099</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbaa099</pub-id><pub-id pub-id-type="pmid">32578842</pub-id></citation></ref>
<ref id="B142">
<label>142.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liang</surname> <given-names>Y</given-names></name> <name><surname>Wu</surname> <given-names>Y</given-names></name> <name><surname>Zhang</surname> <given-names>Z</given-names></name> <name><surname>Liu</surname> <given-names>N</given-names></name> <name><surname>Peng</surname> <given-names>J</given-names></name> <name><surname>Tang</surname> <given-names>J</given-names></name></person-group>. <article-title>Hyb4mC: a hybrid DNA2vec-based model for DNA N4-methylcytosine sites prediction</article-title>. <source>BMC Bioinform</source>. (<year>2022</year>) <volume>23</volume>:<fpage>258</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-022-04789-6</pub-id><pub-id pub-id-type="pmid">35768759</pub-id></citation></ref>
<ref id="B143">
<label>143.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>S</given-names></name> <name><surname>Yang</surname> <given-names>Z</given-names></name> <name><surname>Yang</surname> <given-names>J</given-names></name></person-group>. <article-title>4mCBERT: A computing tool for the identification of DNA N4-methylcytosine sites by sequence-and chemical-derived information based on ensemble learning strategies</article-title>. <source>Int J Biol Macromol</source>. (<year>2023</year>) <volume>231</volume>:<fpage>123180</fpage>. <pub-id pub-id-type="doi">10.1016/j.ijbiomac.2023.123180</pub-id><pub-id pub-id-type="pmid">36646347</pub-id></citation></ref>
<ref id="B144">
<label>144.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tsukiyama</surname> <given-names>S</given-names></name> <name><surname>Hasan</surname> <given-names>MM</given-names></name> <name><surname>Deng</surname> <given-names>HW</given-names></name> <name><surname>Kurata</surname> <given-names>H</given-names></name></person-group>. <article-title>BERT6mA: prediction of DNA N6-methyladenine site using deep learning-based approaches</article-title>. <source>Brief Bioinform</source>. (<year>2022</year>) <volume>23</volume>:<fpage>bbac053</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbac053</pub-id><pub-id pub-id-type="pmid">35225328</pub-id></citation></ref>
<ref id="B145">
<label>145.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shi</surname> <given-names>H</given-names></name> <name><surname>Li</surname> <given-names>S</given-names></name> <name><surname>Su</surname> <given-names>X</given-names></name></person-group>. <article-title>Plant6mA: a predictor for predicting N6-methyladenine sites with lightweight structure in plant genomes</article-title>. <source>Methods</source>. (<year>2022</year>) <volume>204</volume>:<fpage>126</fpage>&#x02013;<lpage>31</lpage>. <pub-id pub-id-type="doi">10.1016/j.ymeth.2022.02.009</pub-id><pub-id pub-id-type="pmid">35231584</pub-id></citation></ref>
<ref id="B146">
<label>146.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Navarez</surname> <given-names>AM</given-names></name> <name><surname>Roxas</surname> <given-names>R</given-names></name></person-group>. (<year>2022</year>). An evaluation of multitask transfer learning methods in identifying 6mA and 5mC methylation sites of rice and maize. <pub-id pub-id-type="doi">10.2139/ssrn.4178244</pub-id></citation>
</ref>
<ref id="B147">
<label>147.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Z</given-names></name> <name><surname>Jiang</surname> <given-names>H</given-names></name> <name><surname>Kong</surname> <given-names>L</given-names></name> <name><surname>Chen</surname> <given-names>Y</given-names></name> <name><surname>Lang</surname> <given-names>K</given-names></name> <name><surname>Fan</surname> <given-names>X</given-names></name> <etal/></person-group>. <article-title>Deep6ma: a deep learning framework for exploring similar patterns in DNA n6-methyladenine sites across different species</article-title>. <source>PLoS Comput Biol</source>. (<year>2021</year>) <volume>17</volume>:<fpage>e1008767</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1008767</pub-id><pub-id pub-id-type="pmid">33600435</pub-id></citation></ref>
<ref id="B148">
<label>148.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>Z</given-names></name> <name><surname>Xiao</surname> <given-names>C</given-names></name> <name><surname>Yin</surname> <given-names>J</given-names></name> <name><surname>She</surname> <given-names>J</given-names></name> <name><surname>Duan</surname> <given-names>H</given-names></name> <name><surname>Liu</surname> <given-names>C</given-names></name> <etal/></person-group>. <article-title>PSAC-6mA: 6mA site identifier using self-attention capsule network based on sequence-positioning</article-title>. <source>Comput Biol Med</source>. (<year>2024</year>) <volume>171</volume>:<fpage>108129</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2024.108129</pub-id><pub-id pub-id-type="pmid">38342046</pub-id></citation></ref>
<ref id="B149">
<label>149.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fan</surname> <given-names>XQ</given-names></name> <name><surname>Lin</surname> <given-names>B</given-names></name> <name><surname>Hu</surname> <given-names>J</given-names></name> <name><surname>Guo</surname> <given-names>ZY</given-names></name></person-group>. <article-title>I-DNAN6mA: accurate identification of DNA N6-methyladenine sites using the base-pairing map and deep learning</article-title>. <source>J Chem Inf Model</source>. (<year>2023</year>) <volume>63</volume>:<fpage>1076</fpage>&#x02013;<lpage>86</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.2c01465</pub-id><pub-id pub-id-type="pmid">36722621</pub-id></citation></ref>
<ref id="B150">
<label>150.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Abbas</surname> <given-names>Z</given-names></name> <name><surname>Rehman</surname> <given-names>MU</given-names></name> <name><surname>Chong</surname> <given-names>KT</given-names></name></person-group>. <article-title>TC-6mA-Pred: prediction of DNA N6-methyladenine sites using CNN with transformer</article-title>. In: <source>2022 IEEE International Conference on Bioinformatics and Biomedicine (BIBM)</source>. <publisher-loc>IEEE</publisher-loc> (<year>2022</year>). p. <fpage>2506</fpage>&#x02013;<lpage>2510</lpage>. <pub-id pub-id-type="doi">10.1109/BIBM55620.2022.9995083</pub-id></citation>
</ref>
<ref id="B151">
<label>151.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>Q</given-names></name> <name><surname>Zhou</surname> <given-names>W</given-names></name> <name><surname>Guo</surname> <given-names>F</given-names></name> <name><surname>Xu</surname> <given-names>L</given-names></name> <name><surname>Zhang</surname> <given-names>L</given-names></name></person-group>. <article-title>6mA-Pred: identifying DNA N6-methyladenine sites based on deep learning</article-title>. <source>PeerJ</source>. (<year>2021</year>) <volume>9</volume>:<fpage>e10813</fpage>. <pub-id pub-id-type="doi">10.7717/peerj.10813</pub-id><pub-id pub-id-type="pmid">33604189</pub-id></citation></ref>
<ref id="B152">
<label>152.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Stanojevi&#x00107;</surname> <given-names>D</given-names></name> <name><surname>Li</surname> <given-names>Z</given-names></name> <name><surname>Foo</surname> <given-names>R</given-names></name> <name><surname>&#x00160;iki&#x00107;</surname> <given-names>M</given-names></name></person-group>. <article-title>Rockfish: a transformer-based model for accurate 5-methylcytosine prediction from nanopore sequencing</article-title>. <source>bioRxiv</source>. (<year>2022</year>). p. <fpage>2022</fpage>&#x02013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1101/2022.11.11.513492</pub-id><pub-id pub-id-type="pmid">38961062</pub-id></citation></ref>
<ref id="B153">
<label>153.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tran</surname> <given-names>TA</given-names></name> <name><surname>Pham</surname> <given-names>DM</given-names></name> <name><surname>Ou</surname> <given-names>YY</given-names></name> <etal/></person-group>. <article-title>An extensive examination of discovering 5-Methylcytosine Sites in Genome-Wide DNA Promoters using machine learning based approaches</article-title>. <source>IEEE/ACM Trans Comput Biol Bioinform</source>. (<year>2021</year>) <volume>19</volume>:<fpage>87</fpage>&#x02013;<lpage>94</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2021.3082184</pub-id><pub-id pub-id-type="pmid">34014828</pub-id></citation></ref>
<ref id="B154">
<label>154.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>Y</given-names></name> <name><surname>He</surname> <given-names>W</given-names></name> <name><surname>Jin</surname> <given-names>J</given-names></name> <name><surname>Xiao</surname> <given-names>G</given-names></name> <name><surname>Cui</surname> <given-names>L</given-names></name> <name><surname>Zeng</surname> <given-names>R</given-names></name> <etal/></person-group>. <article-title>iDNA-ABT: advanced deep learning model for detecting DNA methylation with adaptive features and transductive information maximization</article-title>. <source>Bioinformatics</source>. (<year>2021</year>) <volume>37</volume>:<fpage>4603</fpage>&#x02013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btab677</pub-id><pub-id pub-id-type="pmid">34601568</pub-id></citation></ref>
<ref id="B155">
<label>155.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhuo</surname> <given-names>L</given-names></name> <name><surname>Wang</surname> <given-names>R</given-names></name> <name><surname>Fu</surname> <given-names>X</given-names></name> <name><surname>Yao</surname> <given-names>X</given-names></name></person-group>. <article-title>StableDNAm: towards a stable and efficient model for predicting DNA methylation based on adaptive feature correction learning</article-title>. <source>BMC Gen</source>. (<year>2023</year>) <volume>24</volume>:<fpage>742</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-023-09802-7</pub-id><pub-id pub-id-type="pmid">38053026</pub-id></citation></ref>
<ref id="B156">
<label>156.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zeng</surname> <given-names>W</given-names></name> <name><surname>Gautam</surname> <given-names>A</given-names></name> <name><surname>Huson</surname> <given-names>DH</given-names></name></person-group>. <article-title>MuLan-Methyl-multiple transformer-based language models for accurate DNA methylation prediction</article-title>. <source>GigaScience</source>. (<year>2023</year>) <volume>12</volume>:<fpage>giad054</fpage>. <pub-id pub-id-type="doi">10.1093/gigascience/giad054</pub-id><pub-id pub-id-type="pmid">37489753</pub-id></citation></ref>
<ref id="B157">
<label>157.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jin</surname> <given-names>J</given-names></name> <name><surname>Yu</surname> <given-names>Y</given-names></name> <name><surname>Wang</surname> <given-names>R</given-names></name> <name><surname>Zeng</surname> <given-names>X</given-names></name> <name><surname>Pang</surname> <given-names>C</given-names></name> <name><surname>Jiang</surname> <given-names>Y</given-names></name> <etal/></person-group>. <article-title>iDNA-ABF: multi-scale deep biological language learning model for the interpretable prediction of DNA methylations</article-title>. <source>Genome Biol</source>. (<year>2022</year>) <volume>23</volume>:<fpage>219</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-022-02780-1</pub-id><pub-id pub-id-type="pmid">36253864</pub-id></citation></ref>
<ref id="B158">
<label>158.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Z</given-names></name> <name><surname>Xiang</surname> <given-names>S</given-names></name> <name><surname>Zhou</surname> <given-names>C</given-names></name> <name><surname>Xu</surname> <given-names>Q</given-names></name></person-group>. <article-title>DeepMethylation: a deep learning based framework with GloVe and Transformer encoder for DNA methylation prediction</article-title>. <source>PeerJ</source>. (<year>2023</year>) <volume>11</volume>:<fpage>e16125</fpage>. <pub-id pub-id-type="doi">10.7717/peerj.16125</pub-id><pub-id pub-id-type="pmid">37780374</pub-id></citation></ref>
<ref id="B159">
<label>159.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jeong</surname> <given-names>Y</given-names></name> <name><surname>Gerh&#x000E4;user</surname> <given-names>C</given-names></name> <name><surname>Sauter</surname> <given-names>G</given-names></name> <name><surname>Schlomm</surname> <given-names>T</given-names></name> <name><surname>Rohr</surname> <given-names>K</given-names></name> <name><surname>Lutsik</surname> <given-names>P</given-names></name></person-group>. <article-title>MethylBERT: a Transformer-based model for read-level DNA methylation pattern identification and tumour deconvolution</article-title>. <source>bioRxiv</source>. (<year>2023</year>). p. <fpage>2023</fpage>&#x02013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1101/2023.10.29.564590</pub-id><pub-id pub-id-type="pmid">39824848</pub-id></citation></ref>
<ref id="B160">
<label>160.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Song</surname> <given-names>C</given-names></name> <name><surname>Diao</surname> <given-names>J</given-names></name> <name><surname>Brunger</surname> <given-names>A</given-names></name> <name><surname>Quake</surname> <given-names>S</given-names></name></person-group>. <article-title>Simultaneous single-molecule epigenetic imaging of DNA methylation and hydroxymethylation</article-title>. <source>Proc Nat Acad Sci</source>. (<year>2016</year>) <volume>113</volume>:<fpage>4338</fpage>&#x02013;<lpage>43</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1600223113</pub-id><pub-id pub-id-type="pmid">27035984</pub-id></citation></ref>
<ref id="B161">
<label>161.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ma</surname> <given-names>X</given-names></name></person-group>. <article-title>Deep5hmc: predicting genome-wide 5-hydroxymethylcytosine landscape via a multimodal deep learning model</article-title>. <source>Bioinformatics</source>. (<year>2024</year>) <volume>40</volume>:<fpage>btae528</fpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btae528</pub-id><pub-id pub-id-type="pmid">39196755</pub-id></citation></ref>
<ref id="B162">
<label>162.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kohli</surname> <given-names>R</given-names></name> <name><surname>Zhang</surname> <given-names>Y</given-names></name></person-group>. <article-title>Tet enzymes, TDG and the dynamics of DNA demethylation</article-title>. <source>Nature</source>. (<year>2013</year>) <volume>502</volume>:<fpage>472</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1038/nature12750</pub-id><pub-id pub-id-type="pmid">24153300</pub-id></citation></ref>
<ref id="B163">
<label>163.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gialitsis</surname> <given-names>N</given-names></name> <name><surname>Giannakopoulos</surname> <given-names>G</given-names></name> <name><surname>Athanasouli</surname> <given-names>M</given-names></name></person-group>. <article-title>Evaluation of distributed DNA representations on the classification of conserved non-coding elements</article-title>. In: <source>11th Hellenic Conference on Artificial Intelligence</source>. (<year>2020</year>). p. <fpage>41</fpage>&#x02013;<lpage>47</lpage>. <pub-id pub-id-type="doi">10.1145/3411408.3411463</pub-id></citation>
</ref>
<ref id="B164">
<label>164.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Akal&#x00131;n</surname> <given-names>F</given-names></name> <name><surname>Yumu&#x0015F;ak</surname> <given-names>N</given-names></name></person-group>. <article-title>Classification of exon and intron regions on DNA sequences with hybrid use of sbert and anfis approaches</article-title>. <source>Politeknik Dergisi</source>. (<year>2023</year>) <volume>27</volume>:<fpage>1043</fpage>&#x02013;<lpage>1053</lpage>. <pub-id pub-id-type="doi">10.2339/politeknik.1187808</pub-id></citation>
</ref>
<ref id="B165">
<label>165.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Do</surname> <given-names>DT</given-names></name> <name><surname>Le</surname> <given-names>NQK</given-names></name></person-group>. <article-title>A sequence-based approach for identifying recombination spots in Saccharomyces cerevisiae by using hyper-parameter optimization in FastText and support vector machine</article-title>. <source>Chemometr Intell Labor Syst</source>. (<year>2019</year>) <volume>194</volume>:<fpage>103855</fpage>. <pub-id pub-id-type="doi">10.1016/j.chemolab.2019.103855</pub-id></citation>
</ref>
<ref id="B166">
<label>166.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Soldner</surname> <given-names>F</given-names></name> <name><surname>Jaenisch</surname> <given-names>R</given-names></name></person-group>. <article-title>Dissecting risk haplotypes in sporadic Alzheimer&#x00027;s disease</article-title>. <source>Cell Stem Cell</source>. (<year>2015</year>) <volume>16</volume>:<fpage>341</fpage>&#x02013;<lpage>2</lpage>. <pub-id pub-id-type="doi">10.1016/j.stem.2015.03.010</pub-id><pub-id pub-id-type="pmid">25842969</pub-id></citation></ref>
<ref id="B167">
<label>167.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Paigen</surname> <given-names>K</given-names></name> <name><surname>Petkov</surname> <given-names>P</given-names></name></person-group>. <article-title>Mammalian recombination hot spots: properties, control and evolution</article-title>. <source>Nat Rev Genet</source>. (<year>2010</year>) <volume>11</volume>:<fpage>221</fpage>&#x02013;<lpage>33</lpage>. <pub-id pub-id-type="doi">10.1038/nrg2712</pub-id><pub-id pub-id-type="pmid">20168297</pub-id></citation></ref>
<ref id="B168">
<label>168.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Z</given-names></name> <name><surname>Tian</surname> <given-names>D</given-names></name> <name><surname>Wang</surname> <given-names>B</given-names></name> <name><surname>Wang</surname> <given-names>J</given-names></name> <name><surname>Wang</surname> <given-names>S</given-names></name> <name><surname>Chen</surname> <given-names>H</given-names></name> <etal/></person-group>. <article-title>Microbes drive global soil nitrogen mineralization and availability</article-title>. <source>Glob Chang Biol</source>. (<year>2019</year>) <volume>25</volume>:<fpage>1078</fpage>&#x02013;<lpage>88</lpage>. <pub-id pub-id-type="doi">10.1111/gcb.14557</pub-id><pub-id pub-id-type="pmid">30589163</pub-id></citation></ref>
<ref id="B169">
<label>169.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Knops</surname> <given-names>J</given-names></name> <name><surname>Bradley</surname> <given-names>K</given-names></name> <name><surname>Wedin</surname> <given-names>D</given-names></name></person-group>. <article-title>Mechanisms of plant species impacts on ecosystem nitrogen cycling</article-title>. <source>Ecol Lett</source>. (<year>2002</year>) <volume>5</volume>:<fpage>454</fpage>&#x02013;<lpage>66</lpage>. <pub-id pub-id-type="doi">10.1046/j.1461-0248.2002.00332.x</pub-id></citation>
</ref>
<ref id="B170">
<label>170.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Santos</surname> <given-names>P</given-names></name> <name><surname>Fang</surname> <given-names>Z</given-names></name> <name><surname>Mason</surname> <given-names>S</given-names></name> <name><surname>Set&#x000FA;bal</surname> <given-names>J</given-names></name> <name><surname>Dixon</surname> <given-names>R</given-names></name></person-group>. <article-title>Distribution of nitrogen fixation and nitrogenase-like sequences amongst microbial genomes</article-title>. <source>BMC Gen</source>. (<year>2012</year>) <volume>13</volume>:<fpage>162</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2164-13-162</pub-id><pub-id pub-id-type="pmid">22554235</pub-id></citation></ref>
<ref id="B171">
<label>171.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Narayanan</surname> <given-names>S</given-names></name> <name><surname>Ramachandran</surname> <given-names>A</given-names></name> <name><surname>Aakur</surname> <given-names>SN</given-names></name> <name><surname>Bagavathi</surname> <given-names>A</given-names></name></person-group>. <article-title>Genome sequence classification for animal diagnostics with graph representations and deep neural networks</article-title>. <source>arXiv preprint arXiv:200712791</source>. (<year>2020</year>).</citation>
</ref>
<ref id="B172">
<label>172.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>F</given-names></name> <name><surname>Gan</surname> <given-names>R</given-names></name> <name><surname>Zhang</surname> <given-names>F</given-names></name> <name><surname>Ren</surname> <given-names>C</given-names></name> <name><surname>Yu</surname> <given-names>L</given-names></name> <name><surname>Si</surname> <given-names>Y</given-names></name> <etal/></person-group>. <article-title>PHISDetector: a tool to detect diverse <italic>in silico</italic> phage-host interaction signals for virome studies</article-title>. <source>Gen Proteom Bioinform</source>. (<year>2022</year>) <volume>20</volume>:<fpage>508</fpage>&#x02013;<lpage>23</lpage>. <pub-id pub-id-type="doi">10.1016/j.gpb.2022.02.003</pub-id><pub-id pub-id-type="pmid">35272051</pub-id></citation></ref>
<ref id="B173">
<label>173.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Y&#x00131;lmaz</surname> <given-names>A</given-names></name></person-group>. <article-title>Assessment of mutation susceptibility in DNA sequences with word vectors</article-title>. <source>J Intel Syst</source>. (<year>2020</year>) <volume>3</volume>:<fpage>1</fpage>&#x02013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.38016/jista.674910</pub-id></citation>
</ref>
<ref id="B174">
<label>174.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Z</given-names></name> <name><surname>Li</surname> <given-names>X</given-names></name> <name><surname>Yang</surname> <given-names>M</given-names></name> <name><surname>Zhang</surname> <given-names>H</given-names></name> <name><surname>Xu</surname> <given-names>X</given-names></name></person-group>. <article-title>Optimization of deep learning models for the prediction of gene mutations using unsupervised clustering</article-title>. <source>J Pathol Clin Res</source>. (<year>2022</year>) <volume>9</volume>:<fpage>3</fpage>&#x02013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.1002/cjp2.302</pub-id><pub-id pub-id-type="pmid">36376239</pub-id></citation></ref>
<ref id="B175">
<label>175.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qiu</surname> <given-names>J</given-names></name> <name><surname>Nie</surname> <given-names>W</given-names></name> <name><surname>Ding</surname> <given-names>H</given-names></name> <name><surname>Dai</surname> <given-names>J</given-names></name> <name><surname>Wei</surname> <given-names>Y</given-names></name> <name><surname>Li</surname> <given-names>D</given-names></name> <etal/></person-group>. <article-title>PB-LKS: a python package for predicting phage-bacteria interaction through local K-MER strategy</article-title>. <source>Brief Bioinform</source>. (<year>2024</year>) <volume>25</volume>:<fpage>bbae010</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbae010</pub-id><pub-id pub-id-type="pmid">38344864</pub-id></citation></ref>
<ref id="B176">
<label>176.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Y</given-names></name> <name><surname>Sun</surname> <given-names>H</given-names></name> <name><surname>Wang</surname> <given-names>H</given-names></name> <name><surname>Li</surname> <given-names>D</given-names></name> <name><surname>Zhao</surname> <given-names>W</given-names></name> <name><surname>Jiang</surname> <given-names>X</given-names></name> <etal/></person-group>. <article-title>An effective model for predicting phage-host interactions via graph embedding representation learning with multi-head attention mechanism</article-title>. <source>IEEE J Biomed Health Inf</source>. (<year>2023</year>) <volume>27</volume>:<fpage>3061</fpage>&#x02013;<lpage>3071</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2023.3261319</pub-id><pub-id pub-id-type="pmid">37030796</pub-id></citation></ref>
<ref id="B177">
<label>177.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pan</surname> <given-names>J</given-names></name> <name><surname>You</surname> <given-names>W</given-names></name> <name><surname>Lu</surname> <given-names>X</given-names></name> <name><surname>Wang</surname> <given-names>S</given-names></name> <name><surname>You</surname> <given-names>Z</given-names></name> <name><surname>Sun</surname> <given-names>Y</given-names></name> <etal/></person-group>. <article-title>A novel deep learning model for predicting phage-host interactions via multiple biological information</article-title>. <source>Comput Struct Biotechnol J</source>. (<year>2023</year>) <volume>21</volume>:<fpage>3404</fpage>&#x02013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1016/j.csbj.2023.06.014</pub-id><pub-id pub-id-type="pmid">37397626</pub-id></citation></ref>
<ref id="B178">
<label>178.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Edwards</surname> <given-names>R</given-names></name> <name><surname>McNair</surname> <given-names>K</given-names></name> <name><surname>Faust</surname> <given-names>K</given-names></name> <name><surname>Raes</surname> <given-names>J</given-names></name> <name><surname>Dutilh</surname> <given-names>B</given-names></name></person-group>. <article-title>Computational approaches to predict bacteriophage-host relationships</article-title>. <source>FEMS Microbiol Rev</source>. (<year>2015</year>) <volume>40</volume>:<fpage>258</fpage>&#x02013;<lpage>72</lpage>. <pub-id pub-id-type="doi">10.1093/femsre/fuv048</pub-id><pub-id pub-id-type="pmid">26657537</pub-id></citation></ref>
<ref id="B179">
<label>179.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Harbeck</surname> <given-names>N</given-names></name> <name><surname>Thomssen</surname> <given-names>C</given-names></name></person-group>. <article-title>A new look at node-negative breast cancer</article-title>. <source>Oncologist</source>. (<year>2010</year>) <volume>15</volume>:<fpage>29</fpage>&#x02013;<lpage>38</lpage>. <pub-id pub-id-type="doi">10.1634/theoncologist.2010-S5-29</pub-id><pub-id pub-id-type="pmid">21138953</pub-id></citation></ref>
<ref id="B180">
<label>180.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Osseni</surname> <given-names>MA</given-names></name> <name><surname>Tossou</surname> <given-names>P</given-names></name> <name><surname>Laviolette</surname> <given-names>F</given-names></name> <name><surname>Corbeil</surname> <given-names>J</given-names></name></person-group>. <article-title>MOT: a Multi-Omics Transformer for multiclass classification tumour types predictions</article-title>. <source>BioRxiv</source>. (<year>2022</year>). p. <fpage>2022</fpage>&#x02013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1101/2022.11.14.516459</pub-id></citation>
</ref>
<ref id="B181">
<label>181.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ghosh</surname> <given-names>A</given-names></name> <name><surname>Singh</surname> <given-names>T</given-names></name> <name><surname>Singla</surname> <given-names>V</given-names></name> <name><surname>Bagga</surname> <given-names>R</given-names></name> <name><surname>Srinivasan</surname> <given-names>R</given-names></name> <name><surname>Khandelwal</surname> <given-names>N</given-names></name></person-group>. <article-title>DTI histogram parameters correlate with the extent of myoinvasion and tumor type in endometrial carcinoma: a preliminary analysis</article-title>. <source>Acta Radiol</source>. (<year>2019</year>) <volume>61</volume>:<fpage>675</fpage>&#x02013;<lpage>84</lpage>. <pub-id pub-id-type="doi">10.1177/0284185119875019</pub-id><pub-id pub-id-type="pmid">31533436</pub-id></citation></ref>
<ref id="B182">
<label>182.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kulakovskiy</surname> <given-names>IV</given-names></name> <name><surname>Vorontsov</surname> <given-names>IE</given-names></name> <name><surname>Yevshin</surname> <given-names>IS</given-names></name> <name><surname>Sharipov</surname> <given-names>RN</given-names></name> <name><surname>Fedorova</surname> <given-names>AD</given-names></name> <name><surname>Rumynskiy</surname> <given-names>EI</given-names></name> <etal/></person-group>. <article-title>HOCOMOCO: towards a complete collection of transcription factor binding models for human and mouse via large-scale ChIP-Seq analysis</article-title>. <source>Nucleic Acids Res</source>. (<year>2018</year>) <volume>46</volume>:<fpage>D252</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkx1106</pub-id><pub-id pub-id-type="pmid">29140464</pub-id></citation></ref>
<ref id="B183">
<label>183.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pujar</surname> <given-names>S</given-names></name> <name><surname>O&#x00027;Leary</surname> <given-names>NA</given-names></name> <name><surname>Farrell</surname> <given-names>CM</given-names></name> <name><surname>Loveland</surname> <given-names>JE</given-names></name> <name><surname>Mudge</surname> <given-names>JM</given-names></name> <name><surname>Wallin</surname> <given-names>C</given-names></name> <etal/></person-group>. <article-title>Consensus coding sequence (CCDS) database: a standardized set of human and mouse protein-coding regions supported by expert curation</article-title>. <source>Nucleic Acids Res</source>. (<year>2018</year>) <volume>46</volume>:<fpage>D221</fpage>&#x02013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkx1031</pub-id><pub-id pub-id-type="pmid">29126148</pub-id></citation></ref>
<ref id="B184">
<label>184.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liberzon</surname> <given-names>A</given-names></name> <name><surname>Subramanian</surname> <given-names>A</given-names></name> <name><surname>Pinchback</surname> <given-names>R</given-names></name> <name><surname>Thorvaldsd&#x000F3;ttir</surname> <given-names>H</given-names></name> <name><surname>Tamayo</surname> <given-names>P</given-names></name> <name><surname>Mesirov</surname> <given-names>JP</given-names></name></person-group>. <article-title>Molecular signatures database (MSigDB) 30</article-title>. <source>Bioinformatics</source>. (<year>2011</year>) <volume>27</volume>:<fpage>1739</fpage>&#x02013;<lpage>40</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btr260</pub-id><pub-id pub-id-type="pmid">21546393</pub-id></citation></ref>
<ref id="B185">
<label>185.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shimada</surname> <given-names>K</given-names></name> <name><surname>Muhlich</surname> <given-names>JL</given-names></name> <name><surname>Mitchison</surname> <given-names>TJ</given-names></name></person-group>. <article-title>A tool for browsing the Cancer Dependency Map reveals functional connections between genes and helps predict the efficacy and selectivity of candidate cancer drugs</article-title>. <source>bioRxiv</source>. (<year>2019</year>). p. <fpage>2019</fpage>&#x02013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1101/2019.12.13.874776</pub-id></citation>
</ref>
<ref id="B186">
<label>186.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fornes</surname> <given-names>O</given-names></name> <name><surname>Castro-Mondragon</surname> <given-names>JA</given-names></name> <name><surname>Khan</surname> <given-names>A</given-names></name> <name><surname>Van der Lee</surname> <given-names>R</given-names></name> <name><surname>Zhang</surname> <given-names>X</given-names></name> <name><surname>Richmond</surname> <given-names>PA</given-names></name> <etal/></person-group>. <article-title>JASPAR 2020: update of the open-access database of transcription factor binding profiles</article-title>. <source>Nucleic Acids Res</source>. (<year>2020</year>) <volume>48</volume>:<fpage>D87</fpage>&#x02013;<lpage>92</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkz1001</pub-id><pub-id pub-id-type="pmid">31701148</pub-id></citation></ref>
<ref id="B187">
<label>187.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>R</given-names></name> <name><surname>Ou</surname> <given-names>HY</given-names></name> <name><surname>Zhang</surname> <given-names>CT</given-names></name></person-group>. <article-title>DEG: a database of essential genes</article-title>. <source>Nucl Acids Res</source>. (<year>2004</year>) <volume>32</volume>:<fpage>D271</fpage>&#x02013;<lpage>D272</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkh024</pub-id><pub-id pub-id-type="pmid">14681410</pub-id></citation></ref>
<ref id="B188">
<label>188.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>de</surname> <given-names>Souza N</given-names></name></person-group>. <article-title>The ENCODE project</article-title>. <source>Nat Methods</source>. (<year>2012</year>) <volume>9</volume>:<fpage>1046</fpage>&#x02013;<lpage>1046</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.2238</pub-id><pub-id pub-id-type="pmid">23281567</pub-id></citation></ref>
<ref id="B189">
<label>189.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>El Allali</surname> <given-names>A</given-names></name> <name><surname>Rose</surname> <given-names>JR</given-names></name></person-group>. <article-title>MGC: Gene calling in metagenomic sequences</article-title>. In: <source>8 TH International Symposium On Bioinformatics Research and Applications</source>.</citation>
</ref>
<ref id="B190">
<label>190.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Saxonov</surname> <given-names>S</given-names></name> <name><surname>Daizadeh</surname> <given-names>I</given-names></name> <name><surname>Fedorov</surname> <given-names>A</given-names></name> <name><surname>Gilbert</surname> <given-names>W</given-names></name></person-group>. <article-title>EID the Exon-Intron Database&#x02013;an exhaustive database of protein-coding intron-containing genes</article-title>. <source>Nucleic Acids Res</source>. (<year>2000</year>) <volume>28</volume>:<fpage>185</fpage>&#x02013;<lpage>90</lpage>. <pub-id pub-id-type="doi">10.1093/nar/28.1.185</pub-id><pub-id pub-id-type="pmid">10592221</pub-id></citation></ref>
<ref id="B191">
<label>191.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hubbard</surname> <given-names>T</given-names></name> <name><surname>Barker</surname> <given-names>D</given-names></name> <name><surname>Birney</surname> <given-names>E</given-names></name> <name><surname>Cameron</surname> <given-names>G</given-names></name> <name><surname>Chen</surname> <given-names>Y</given-names></name> <name><surname>Clark</surname> <given-names>L</given-names></name> <etal/></person-group>. <article-title>The Ensembl genome database project</article-title>. <source>Nucleic Acids Res</source>. (<year>2002</year>) <volume>30</volume>:<fpage>38</fpage>&#x02013;<lpage>41</lpage>. <pub-id pub-id-type="doi">10.1093/nar/30.1.38</pub-id><pub-id pub-id-type="pmid">11752248</pub-id></citation></ref>
<ref id="B192">
<label>192.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Salgado</surname> <given-names>H</given-names></name> <name><surname>Santos</surname> <given-names>A</given-names></name> <name><surname>Garza-Ramos</surname> <given-names>U</given-names></name> <name><surname>van Helden</surname> <given-names>J</given-names></name> <name><surname>D&#x000ED;az</surname> <given-names>E</given-names></name> <name><surname>Collado-Vides</surname> <given-names>J</given-names></name></person-group>. <article-title>RegulonDB (version 20): a database on transcriptional regulation in <italic>Escherichia coli</italic></article-title>. <source>Nucl Acids Res</source>. (<year>1999</year>) <volume>27</volume>:<fpage>59</fpage>&#x02013;<lpage>60</lpage>. <pub-id pub-id-type="doi">10.1093/nar/27.1.59</pub-id><pub-id pub-id-type="pmid">9847141</pub-id></citation></ref>
<ref id="B193">
<label>193.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>P&#x000E9;rier</surname> <given-names>RC</given-names></name> <name><surname>Praz</surname> <given-names>V</given-names></name> <name><surname>Junier</surname> <given-names>T</given-names></name> <name><surname>Bonnard</surname> <given-names>C</given-names></name> <name><surname>Bucher</surname> <given-names>P</given-names></name></person-group>. <article-title>The eukaryotic promoter database (EPD)</article-title>. <source>Nucleic Acids Res</source>. (<year>2000</year>) <volume>28</volume>:<fpage>302</fpage>&#x02013;<lpage>303</lpage>. <pub-id pub-id-type="doi">10.1093/nar/28.1.302</pub-id><pub-id pub-id-type="pmid">10592254</pub-id></citation></ref>
<ref id="B194">
<label>194.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Su</surname> <given-names>W</given-names></name> <name><surname>Liu</surname> <given-names>ML</given-names></name> <name><surname>Yang</surname> <given-names>YH</given-names></name> <name><surname>Wang JS Li</surname> <given-names>SH</given-names></name> <name><surname>Lv</surname> <given-names>H</given-names></name> <etal/></person-group>. <article-title>PPD: a manually curated database for experimentally verified prokaryotic promoters</article-title>. <source>J Mol Biol</source>. (<year>2021</year>) <volume>433</volume>:<fpage>166860</fpage>. <pub-id pub-id-type="doi">10.1016/j.jmb.2021.166860</pub-id><pub-id pub-id-type="pmid">33539888</pub-id></citation></ref>
<ref id="B195">
<label>195.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>S</given-names></name> <name><surname>Li</surname> <given-names>L</given-names></name> <name><surname>Meng</surname> <given-names>X</given-names></name> <name><surname>Sun</surname> <given-names>P</given-names></name> <name><surname>Liu</surname> <given-names>Y</given-names></name> <name><surname>Song</surname> <given-names>Y</given-names></name> <etal/></person-group>. <article-title>DREAM: a database of experimentally supported protein-coding RNAs and drug associations in human cancer</article-title>. <source>Mol Cancer</source>. (<year>2021</year>) <volume>20</volume>:<fpage>1</fpage>&#x02013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1186/s12943-021-01436-1</pub-id><pub-id pub-id-type="pmid">34774049</pub-id></citation></ref>
<ref id="B196">
<label>196.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>B</given-names></name> <name><surname>Zheng</surname> <given-names>L</given-names></name> <name><surname>Long</surname> <given-names>C</given-names></name> <name><surname>Song</surname> <given-names>M</given-names></name> <name><surname>Li</surname> <given-names>T</given-names></name> <name><surname>Yang</surname> <given-names>L</given-names></name> <etal/></person-group>. <article-title>EmExplorer: a database for exploring time activation of gene expression in mammalian embryos</article-title>. <source>Open Biol</source>. (<year>2019</year>) <volume>9</volume>:<fpage>190054</fpage>. <pub-id pub-id-type="doi">10.1098/rsob.190054</pub-id><pub-id pub-id-type="pmid">31164042</pub-id></citation></ref>
<ref id="B197">
<label>197.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Karczewski</surname> <given-names>K</given-names></name> <name><surname>Francioli</surname> <given-names>L</given-names></name></person-group>. <source>The genome aggregation database (gnomAD)</source>. MacArthur Lab. (<year>2017</year>). p. <fpage>1</fpage>&#x02013;<lpage>10</lpage>.</citation>
</ref>
<ref id="B198">
<label>198.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>F</given-names></name> <name><surname>Luo</surname> <given-names>H</given-names></name> <name><surname>Zhang</surname> <given-names>CT</given-names></name></person-group>. <article-title>DeOri: a database of eukaryotic DNA replication origins</article-title>. <source>Bioinformatics</source>. (<year>2012</year>) <volume>28</volume>:<fpage>1551</fpage>&#x02013;<lpage>2</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bts151</pub-id><pub-id pub-id-type="pmid">22467915</pub-id></citation></ref>
<ref id="B199">
<label>199.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>J</given-names></name> <name><surname>Roy</surname> <given-names>A</given-names></name> <name><surname>Zhang</surname> <given-names>Y</given-names></name></person-group>. <article-title>BioLiP: a semi-manually curated database for biologically relevant ligand-protein interactions</article-title>. <source>Nucleic Acids Res</source>. (<year>2012</year>) <volume>41</volume>:<fpage>D1096</fpage>&#x02013;<lpage>103</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gks966</pub-id><pub-id pub-id-type="pmid">23087378</pub-id></citation></ref>
<ref id="B200">
<label>200.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Buniello</surname> <given-names>A</given-names></name> <name><surname>MacArthur</surname> <given-names>JAL</given-names></name> <name><surname>Cerezo</surname> <given-names>M</given-names></name> <name><surname>Harris</surname> <given-names>LW</given-names></name> <name><surname>Hayhurst</surname> <given-names>J</given-names></name> <name><surname>Malangone</surname> <given-names>C</given-names></name> <etal/></person-group>. <article-title>The NHGRI-EBI GWAS Catalog of published genome-wide association studies, targeted arrays and summary statistics 2019</article-title>. <source>Nucleic Acids Res</source>. (<year>2019</year>) <volume>47</volume>:<fpage>D1005</fpage>&#x02013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gky1120</pub-id><pub-id pub-id-type="pmid">30445434</pub-id></citation></ref>
<ref id="B201">
<label>201.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Calvin</surname> <given-names>C</given-names></name></person-group>. <article-title>Descartes: A new generation system for neutronic calculations</article-title>. In: <source>International Topical Meeting on Mathematics and Computation, Supercomputing, Reactor Physics and Nuclear and Biological Applications (M and C 2005)</source>. <publisher-loc>Avignon</publisher-loc> (<year>2005</year>).</citation>
</ref>
<ref id="B202">
<label>202.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>T</given-names></name> <name><surname>Qian</surname> <given-names>J</given-names></name></person-group>. <article-title>EnhancerAtlas 2</article-title>.0: an updated resource with enhancer annotation in 586 tissue/cell types across nine species. <source>Nucleic Acids Res</source>. (<year>2020</year>) <volume>48</volume>:<fpage>D58</fpage>&#x02013;<lpage>64</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkz980</pub-id><pub-id pub-id-type="pmid">31740966</pub-id></citation></ref>
<ref id="B203">
<label>203.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tate</surname> <given-names>JG</given-names></name> <name><surname>Bamford</surname> <given-names>S</given-names></name> <name><surname>Jubb</surname> <given-names>HC</given-names></name> <name><surname>Sondka</surname> <given-names>Z</given-names></name> <name><surname>Beare</surname> <given-names>DM</given-names></name> <name><surname>Bindal</surname> <given-names>N</given-names></name> <etal/></person-group>. <article-title>COSMIC: the catalogue of somatic mutations in cancer</article-title>. <source>Nucleic Acids Res</source>. (<year>2019</year>) <volume>47</volume>:<fpage>D941</fpage>&#x02013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gky1015</pub-id><pub-id pub-id-type="pmid">30371878</pub-id></citation></ref>
<ref id="B204">
<label>204.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pi nero</surname> <given-names>J</given-names></name> <name><surname>Queralt-Rosinach</surname> <given-names>N</given-names></name> <name><surname>Bravo</surname> <given-names>A</given-names></name> <name><surname>Deu-Pons</surname> <given-names>J</given-names></name> <name><surname>Bauer-Mehren</surname> <given-names>A</given-names></name> <name><surname>Baron</surname> <given-names>M</given-names></name> <etal/></person-group>. <article-title>DisGeNET: a discovery platform for the dynamical exploration of human diseases and their genes</article-title>. <source>Database</source>. (<year>2015</year>) <volume>2015</volume>:<fpage>bav028</fpage>. <pub-id pub-id-type="doi">10.1093/database/bav028</pub-id><pub-id pub-id-type="pmid">25877637</pub-id></citation></ref>
<ref id="B205">
<label>205.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Landrum</surname> <given-names>MJ</given-names></name> <name><surname>Lee</surname> <given-names>JM</given-names></name> <name><surname>Benson</surname> <given-names>M</given-names></name> <name><surname>Brown</surname> <given-names>G</given-names></name> <name><surname>Chao</surname> <given-names>C</given-names></name> <name><surname>Chitipiralla</surname> <given-names>S</given-names></name> <etal/></person-group>. <article-title>ClinVar: public archive of interpretations of clinically relevant variants</article-title>. <source>Nucleic Acids Res</source>. (<year>2016</year>) <volume>44</volume>:<fpage>D862</fpage>&#x02013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkv1222</pub-id><pub-id pub-id-type="pmid">26582918</pub-id></citation></ref>
<ref id="B206">
<label>206.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ghandi</surname> <given-names>M</given-names></name> <name><surname>Huang</surname> <given-names>FW</given-names></name> <name><surname>Jan&#x000E9;-Valbuena</surname> <given-names>J</given-names></name> <name><surname>Kryukov</surname> <given-names>GV</given-names></name> <name><surname>Lo</surname> <given-names>CC</given-names></name> <name><surname>McDonald III</surname> <given-names>ER</given-names></name> <etal/></person-group>. <article-title>Next-generation characterization of the cancer cell line encyclopedia</article-title>. <source>Nature</source>. (<year>2019</year>) <volume>569</volume>:<fpage>503</fpage>&#x02013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-019-1186-3</pub-id><pub-id pub-id-type="pmid">31068700</pub-id></citation></ref>
<ref id="B207">
<label>207.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Frankish</surname> <given-names>A</given-names></name> <name><surname>Diekhans</surname> <given-names>M</given-names></name> <name><surname>Jungreis</surname> <given-names>I</given-names></name> <name><surname>Lagarde</surname> <given-names>J</given-names></name> <name><surname>Loveland</surname> <given-names>JE</given-names></name> <name><surname>Mudge</surname> <given-names>JM</given-names></name> <etal/></person-group>. <article-title>GENCODE 2021</article-title>. <source>Nucleic Acids Res</source>. (<year>2021</year>) <volume>49</volume>:<fpage>D916</fpage>&#x02013;<lpage>23</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkaa1087</pub-id><pub-id pub-id-type="pmid">33270111</pub-id></citation></ref>
<ref id="B208">
<label>208.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Consortium</surname> <given-names>GO</given-names></name></person-group>. <article-title>The Gene Ontology (GO) database and informatics resource</article-title>. <source>Nucleic Acids Res</source>. (<year>2004</year>) <volume>32</volume>:<fpage>D258</fpage>&#x02013;<lpage>D261</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkh036</pub-id><pub-id pub-id-type="pmid">14681407</pub-id></citation></ref>
<ref id="B209">
<label>209.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Suzuki</surname> <given-names>Y</given-names></name> <name><surname>Yamashita</surname> <given-names>R</given-names></name> <name><surname>Nakai</surname> <given-names>K</given-names></name> <name><surname>Sugano</surname> <given-names>S</given-names></name> <name><surname>DBTSS</surname></name></person-group>. <article-title>DataBase of human transcriptional start sites and full-length cDNAs</article-title>. <source>Nucleic Acids Res</source>. (<year>2002</year>) <volume>30</volume>:<fpage>328</fpage>&#x02013;<lpage>31</lpage>. <pub-id pub-id-type="doi">10.1093/nar/30.1.328</pub-id><pub-id pub-id-type="pmid">11752328</pub-id></citation></ref>
<ref id="B210">
<label>210.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Clough</surname> <given-names>E</given-names></name> <name><surname>Barrett</surname> <given-names>T</given-names></name></person-group>. <article-title>The gene expression omnibus database</article-title>. In:<person-group person-group-type="editor"><name><surname>Math&#x000E9;</surname></name> <name><surname>E.</surname></name> <name><surname>Davis</surname></name> <name><surname>S.</surname></name></person-group>, editor. <source>Statistical Genomics. Methods in Molecular Biology</source>. <publisher-loc>New York, NY</publisher-loc>: <publisher-name>Humana Press</publisher-name>.</citation>
</ref>
<ref id="B211">
<label>211.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kanehisa</surname> <given-names>M</given-names></name></person-group>. <article-title>The KEGG database</article-title>. In: &#x02018;<italic>In silico&#x00027; simulation of biological processes: Novartis Foundation Symposium</italic>. Wiley Online Library (<year>2002</year>). p. <fpage>91</fpage>&#x02013;<lpage>103</lpage>. <pub-id pub-id-type="doi">10.1002/0470857897.ch8</pub-id></citation>
</ref>
<ref id="B212">
<label>212.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Geer</surname> <given-names>LY</given-names></name> <name><surname>Marchler-Bauer</surname> <given-names>A</given-names></name> <name><surname>Geer</surname> <given-names>RC</given-names></name> <name><surname>Han</surname> <given-names>L</given-names></name> <name><surname>He</surname> <given-names>J</given-names></name> <name><surname>He</surname> <given-names>S</given-names></name> <etal/></person-group>. <article-title>The NCBI biosystems database</article-title>. <source>Nucleic Acids Res</source>. (<year>2010</year>) <volume>38</volume>:<fpage>D492</fpage>&#x02013;<lpage>D496</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkp858</pub-id><pub-id pub-id-type="pmid">19854944</pub-id></citation></ref>
<ref id="B213">
<label>213.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Benson</surname> <given-names>DA</given-names></name> <name><surname>Karsch-Mizrachi</surname> <given-names>I</given-names></name> <name><surname>Lipman</surname> <given-names>DJ</given-names></name> <name><surname>Ostell</surname> <given-names>J</given-names></name> <name><surname>Rapp</surname> <given-names>BA</given-names></name> <name><surname>Wheeler</surname> <given-names>DL</given-names></name></person-group>. <article-title>GenBank</article-title>. <source>Nucleic Acids Res</source>. (<year>2000</year>) <volume>28</volume>:<fpage>15</fpage>&#x02013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1093/nar/28.1.15</pub-id><pub-id pub-id-type="pmid">10592170</pub-id></citation></ref>
<ref id="B214">
<label>214.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sherry</surname> <given-names>ST</given-names></name> <name><surname>Ward</surname> <given-names>M</given-names></name> <name><surname>Sirotkin</surname> <given-names>K</given-names></name></person-group>. <article-title>dbSNP&#x02013;database for single nucleotide polymorphisms and other classes of minor genetic variation</article-title>. <source>Genome Res</source>. (<year>1999</year>) <volume>9</volume>:<fpage>677</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1101/gr.9.8.677</pub-id><pub-id pub-id-type="pmid">10447503</pub-id></citation></ref>
<ref id="B215">
<label>215.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sherry</surname> <given-names>ST</given-names></name> <name><surname>Ward</surname> <given-names>MH</given-names></name> <name><surname>Kholodov</surname> <given-names>M</given-names></name> <name><surname>Baker</surname> <given-names>J</given-names></name> <name><surname>Phan</surname> <given-names>L</given-names></name> <name><surname>Smigielski</surname> <given-names>EM</given-names></name> <etal/></person-group>. <article-title>dbSNP: the NCBI database of genetic variation</article-title>. <source>Nucleic Acids Res</source>. (<year>2001</year>) <volume>29</volume>:<fpage>308</fpage>&#x02013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1093/nar/29.1.308</pub-id><pub-id pub-id-type="pmid">11125122</pub-id></citation></ref>
<ref id="B216">
<label>216.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hamosh</surname> <given-names>A</given-names></name> <name><surname>Scott</surname> <given-names>AF</given-names></name> <name><surname>Amberger</surname> <given-names>J</given-names></name> <name><surname>Valle</surname> <given-names>D</given-names></name> <name><surname>McKusick</surname> <given-names>VA</given-names></name></person-group>. <article-title>Online Mendelian inheritance in man (OMIM)</article-title>. <source>Hum Mutat</source>. (<year>2000</year>) <volume>15</volume>:<fpage>57</fpage>&#x02013;<lpage>61</lpage>. <pub-id pub-id-type="doi">10.1002/(SICI)1098-1004(200001)15:1&#x0003C;57::AID-HUMU12&#x0003E;3.0.CO;2-G</pub-id></citation>
</ref>
<ref id="B217">
<label>217.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lilleberg</surname> <given-names>J</given-names></name> <name><surname>Zhu</surname> <given-names>Y</given-names></name> <name><surname>Zhang</surname> <given-names>Y</given-names></name></person-group>. <article-title>Support vector machines and word2vec for text classification with semantic features</article-title>. In: <source>2015 IEEE 14th International Conference on Cognitive Informatics</source> &#x00026; <italic>Cognitive Computing (ICCI&#x0002A; CC)</italic>. IEEE (<year>2015</year>). p. <fpage>136</fpage>&#x02013;<lpage>140</lpage>. <pub-id pub-id-type="doi">10.1109/ICCI-CC.2015.7259377</pub-id></citation>
</ref>
<ref id="B218">
<label>218.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zou</surname> <given-names>WY</given-names></name> <name><surname>Socher</surname> <given-names>R</given-names></name> <name><surname>Cer</surname> <given-names>D</given-names></name> <name><surname>Manning</surname> <given-names>CD</given-names></name></person-group>. <article-title>Bilingual word embeddings for phrase-based machine translation</article-title>. In: <source>Proceedings of the 2013 Conference on Empirical Methods in Natural Language Processing</source>. (<year>2013</year>). p. <fpage>1393</fpage>&#x02013;<lpage>1398</lpage>. <pub-id pub-id-type="doi">10.18653/v1/D13-1141</pub-id><pub-id pub-id-type="pmid">36568019</pub-id></citation></ref>
<ref id="B219">
<label>219.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Al-Amin</surname> <given-names>M</given-names></name> <name><surname>Islam</surname> <given-names>MS</given-names></name> <name><surname>Uzzal</surname> <given-names>SD</given-names></name></person-group>. <article-title>Sentiment analysis of Bengali comments with Word2Vec and sentiment information of words</article-title>. In: <source>2017 International Conference on Electrical, Computer and Communication Engineering (ECCE)</source>. <publisher-loc>IEEE</publisher-loc> (<year>2017</year>). p. <fpage>186</fpage>&#x02013;<lpage>190</lpage>. <pub-id pub-id-type="doi">10.1109/ECACE.2017.7912903</pub-id></citation>
</ref>
<ref id="B220">
<label>220.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Verma</surname> <given-names>PK</given-names></name> <name><surname>Agrawal</surname> <given-names>P</given-names></name> <name><surname>Amorim</surname> <given-names>I</given-names></name> <name><surname>Prodan</surname> <given-names>R</given-names></name></person-group>. <article-title>WELFake: Word embedding over linguistic features for fake news detection</article-title>. <source>IEEE Trans Comput Soc Syst</source>. (<year>2021</year>) <volume>8</volume>:<fpage>881</fpage>&#x02013;<lpage>93</lpage>. <pub-id pub-id-type="doi">10.1109/TCSS.2021.3068519</pub-id></citation>
</ref>
<ref id="B221">
<label>221.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mikolov</surname> <given-names>T</given-names></name> <name><surname>Chen</surname> <given-names>K</given-names></name> <name><surname>Corrado</surname> <given-names>G</given-names></name> <name><surname>Dean</surname> <given-names>J</given-names></name></person-group>. <article-title>Efficient estimation of word representations in vector space</article-title>. <source>arXiv preprint arXiv:13013781</source>. (<year>2013</year>).<pub-id pub-id-type="pmid">31752376</pub-id></citation></ref>
<ref id="B222">
<label>222.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pennington</surname> <given-names>J</given-names></name> <name><surname>Socher</surname> <given-names>R</given-names></name> <name><surname>Manning</surname> <given-names>CD</given-names></name></person-group>. <article-title>Glove: Global vectors for word representation</article-title>. In: <source>Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP)</source>. (<year>2014</year>). p. <fpage>1532</fpage>&#x02013;<lpage>1543</lpage>. <pub-id pub-id-type="doi">10.3115/v1/D14-1162</pub-id></citation>
</ref>
<ref id="B223">
<label>223.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mikolov</surname> <given-names>T</given-names></name> <name><surname>Grave</surname> <given-names>E</given-names></name> <name><surname>Bojanowski</surname> <given-names>P</given-names></name> <name><surname>Puhrsch</surname> <given-names>C</given-names></name> <name><surname>Joulin</surname> <given-names>A</given-names></name></person-group>. <article-title>Advances in pre-training distributed word representations</article-title>. <source>arXiv preprint arXiv:171209405</source>. (<year>2017</year>).<pub-id pub-id-type="pmid">31501885</pub-id></citation></ref>
<ref id="B224">
<label>224.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Perozzi</surname> <given-names>B</given-names></name> <name><surname>Al-Rfou</surname> <given-names>R</given-names></name> <name><surname>Skiena</surname> <given-names>S</given-names></name></person-group>. <article-title>Deepwalk: online learning of social representations</article-title>. In: <source>Proceedings of the 20th ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</source>. (<year>2014</year>). p. <fpage>701</fpage>&#x02013;<lpage>710</lpage>. <pub-id pub-id-type="doi">10.1145/2623330.2623732</pub-id></citation>
</ref>
<ref id="B225">
<label>225.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Grover</surname> <given-names>A</given-names></name> <name><surname>Leskovec</surname> <given-names>J</given-names></name></person-group>. <article-title>node2vec: Scalable feature learning for networks</article-title>. In: <source>Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</source>. (<year>2016</year>). p. <fpage>855</fpage>&#x02013;<lpage>864</lpage>. <pub-id pub-id-type="doi">10.1145/2939672.2939754</pub-id><pub-id pub-id-type="pmid">27853626</pub-id></citation></ref>
<ref id="B226">
<label>226.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Narayanan</surname> <given-names>A</given-names></name> <name><surname>Chandramohan</surname> <given-names>M</given-names></name> <name><surname>Venkatesan</surname> <given-names>R</given-names></name> <name><surname>Chen</surname> <given-names>L</given-names></name> <name><surname>Liu</surname> <given-names>Y</given-names></name> <name><surname>Jaiswal</surname> <given-names>S</given-names></name></person-group>. <article-title>graph2vec: Learning distributed representations of graphs</article-title>. <source>arXiv preprint arXiv:170705005</source>. (<year>2017</year>).</citation>
</ref>
<ref id="B227">
<label>227.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>D</given-names></name> <name><surname>Cui</surname> <given-names>P</given-names></name> <name><surname>Zhu</surname> <given-names>W</given-names></name></person-group>. <article-title>Structural deep network embedding</article-title>. In: <source>Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</source> (<year>2016</year>). p. <fpage>1225</fpage>&#x02013;<lpage>1234</lpage>. <pub-id pub-id-type="doi">10.1145/2939672.2939753</pub-id></citation>
</ref>
<ref id="B228">
<label>228.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tang</surname> <given-names>L</given-names></name> <name><surname>Liu</surname> <given-names>H</given-names></name></person-group>. <article-title>Relational learning via latent social dimensions</article-title>. In: <source>Proceedings of the 15th ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</source>. (<year>2009</year>). p. <fpage>817</fpage>&#x02013;<lpage>826</lpage>. <pub-id pub-id-type="doi">10.1145/1557019.1557109</pub-id></citation>
</ref>
<ref id="B229">
<label>229.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cao</surname> <given-names>S</given-names></name> <name><surname>Lu</surname> <given-names>W</given-names></name> <name><surname>Xu</surname> <given-names>Q</given-names></name></person-group>. <article-title>Grarep: learning graph representations with global structural information</article-title>. In: <source>Proceedings of the 24th ACM International on Conference on Information and Knowledge Management</source>. (<year>2015</year>). p. <fpage>891</fpage>&#x02013;<lpage>900</lpage>. <pub-id pub-id-type="doi">10.1145/2806416.2806512</pub-id></citation>
</ref>
<ref id="B230">
<label>230.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Belkin</surname> <given-names>M</given-names></name> <name><surname>Niyogi</surname> <given-names>P</given-names></name></person-group>. <article-title>Laplacian eigenmaps and spectral techniques for embedding and clustering</article-title>. In: <source>Advances in Neural Information Processing Systems</source>. (<year>2001</year>). p. 14. <pub-id pub-id-type="doi">10.7551/mitpress/1120.003.0080</pub-id><pub-id pub-id-type="pmid">15333211</pub-id></citation></ref>
<ref id="B231">
<label>231.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Roweis</surname> <given-names>ST</given-names></name> <name><surname>Saul</surname> <given-names>LK</given-names></name></person-group>. <article-title>Nonlinear dimensionality reduction by locally linear embedding</article-title>. <source>Science</source>. (<year>2000</year>) <volume>290</volume>:<fpage>2323</fpage>&#x02013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1126/science.290.5500.2323</pub-id><pub-id pub-id-type="pmid">11125150</pub-id></citation></ref>
<ref id="B232">
<label>232.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Smaili</surname> <given-names>FZ</given-names></name> <name><surname>Gao</surname> <given-names>X</given-names></name> <name><surname>Hoehndorf</surname> <given-names>R</given-names></name></person-group>. <article-title>OPA2Vec: combining formal and informal content of biomedical ontologies to improve similarity-based prediction</article-title>. <source>Bioinformatics</source>. (<year>2019</year>) <volume>35</volume>:<fpage>2133</fpage>&#x02013;<lpage>40</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty933</pub-id><pub-id pub-id-type="pmid">30407490</pub-id></citation></ref>
<ref id="B233">
<label>233.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Song</surname> <given-names>T</given-names></name> <name><surname>Song</surname> <given-names>H</given-names></name> <name><surname>Pan</surname> <given-names>Z</given-names></name> <name><surname>Gao</surname> <given-names>Y</given-names></name> <name><surname>Yang</surname> <given-names>Q</given-names></name> <name><surname>Wang</surname> <given-names>X</given-names></name></person-group>. <article-title>DeepDualEPI: predicting promoter-enhancer interactions based on DNA sequence and genomic signals</article-title>. In: <source>2023 IEEE International Conference on Bioinformatics and Biomedicine (BIBM)</source>. <publisher-loc>IEEE</publisher-loc> (<year>2023</year>). p. <fpage>2889</fpage>&#x02013;<lpage>2895</lpage>. <pub-id pub-id-type="doi">10.1109/BIBM58861.2023.10385972</pub-id><pub-id pub-id-type="pmid">36495179</pub-id></citation></ref>
<ref id="B234">
<label>234.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Le</surname> <given-names>NQK</given-names></name></person-group>. <article-title>iN6-methylat (5-step): identifying DNA N 6-methyladenine sites in rice genome using continuous bag of nucleobases via Chou&#x00027;s 5-step rule</article-title>. <source>Molec Gen Genom</source>. (<year>2019</year>) <volume>294</volume>:<fpage>1173</fpage>&#x02013;<lpage>82</lpage>. <pub-id pub-id-type="doi">10.1007/s00438-019-01570-y</pub-id><pub-id pub-id-type="pmid">31055655</pub-id></citation></ref>
<ref id="B235">
<label>235.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Raza</surname> <given-names>A</given-names></name> <name><surname>Tahir</surname> <given-names>M</given-names></name> <name><surname>Alam</surname> <given-names>W</given-names></name></person-group>. <article-title>iPro-TCN: Prediction of DNA promoters recognition and their strength using temporal convolutional network</article-title>. <source>IEEE Access</source>. (<year>2023</year>). <pub-id pub-id-type="doi">10.1109/ACCESS.2023.3285197</pub-id></citation>
</ref>
<ref id="B236">
<label>236.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Y</given-names></name> <name><surname>Tai</surname> <given-names>S</given-names></name> <name><surname>Zhang</surname> <given-names>S</given-names></name> <name><surname>Sheng</surname> <given-names>N</given-names></name> <name><surname>Xie</surname> <given-names>X</given-names></name></person-group>. <article-title>PromGER: promoter prediction based on graph embedding and ensemble learning for eukaryotic sequence</article-title>. <source>Genes</source>. (<year>2023</year>) <volume>14</volume>:<fpage>1441</fpage>. <pub-id pub-id-type="doi">10.3390/genes14071441</pub-id><pub-id pub-id-type="pmid">37510345</pub-id></citation></ref>
<ref id="B237">
<label>237.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Min</surname> <given-names>B</given-names></name> <name><surname>Ross</surname> <given-names>H</given-names></name> <name><surname>Sulem</surname> <given-names>E</given-names></name> <name><surname>Veyseh</surname> <given-names>APB</given-names></name> <name><surname>Nguyen</surname> <given-names>TH</given-names></name> <name><surname>Sainz</surname> <given-names>O</given-names></name> <etal/></person-group>. <article-title>Recent advances in natural language processing via large pre-trained language models: a survey</article-title>. <source>ACM Comput Surv</source>. (<year>2023</year>) <volume>56</volume>:<fpage>1</fpage>&#x02013;<lpage>40</lpage>. <pub-id pub-id-type="doi">10.1145/3605943</pub-id></citation>
</ref>
<ref id="B238">
<label>238.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chang</surname> <given-names>Y</given-names></name> <name><surname>Wang</surname> <given-names>X</given-names></name> <name><surname>Wang</surname> <given-names>J</given-names></name> <name><surname>Wu</surname> <given-names>Y</given-names></name> <name><surname>Yang</surname> <given-names>L</given-names></name> <name><surname>Zhu</surname> <given-names>K</given-names></name> <etal/></person-group>. <article-title>A survey on evaluation of large language models</article-title>. <source>ACM Trans Intell Syst Technol</source>. (<year>2024</year>) <volume>15</volume>:<fpage>1</fpage>&#x02013;<lpage>45</lpage>. <pub-id pub-id-type="doi">10.1145/3641289</pub-id></citation>
</ref>
<ref id="B239">
<label>239.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yue</surname> <given-names>T</given-names></name> <name><surname>Wang</surname> <given-names>Y</given-names></name> <name><surname>Zhang</surname> <given-names>L</given-names></name> <name><surname>Gu</surname> <given-names>C</given-names></name> <name><surname>Xue</surname> <given-names>H</given-names></name> <name><surname>Wang</surname> <given-names>W</given-names></name> <etal/></person-group>. <article-title>Deep learning for genomics: from early neural nets to modern large language models</article-title>. <source>Int J Mol Sci</source>. (<year>2023</year>) <volume>24</volume>:<fpage>15858</fpage>. <pub-id pub-id-type="doi">10.3390/ijms242115858</pub-id><pub-id pub-id-type="pmid">37958843</pub-id></citation></ref>
<ref id="B240">
<label>240.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Y</given-names></name> <name><surname>Zou</surname> <given-names>J</given-names></name></person-group>. <article-title>GenePT: a simple but effective foundation model for genes and cells built from ChatGPT</article-title>. <source>bioRxiv</source>. (<year>2023</year>). <pub-id pub-id-type="doi">10.1101/2023.10.16.562533</pub-id><pub-id pub-id-type="pmid">37905130</pub-id></citation></ref>
<ref id="B241">
<label>241.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vaswani</surname> <given-names>A</given-names></name> <name><surname>Shazeer</surname> <given-names>N</given-names></name> <name><surname>Parmar</surname> <given-names>N</given-names></name> <name><surname>Uszkoreit</surname> <given-names>J</given-names></name> <name><surname>Jones</surname> <given-names>L</given-names></name> <name><surname>Gomez</surname> <given-names>AN</given-names></name> <etal/></person-group>. <article-title>Attention is all you need</article-title>. In: <source>Advances in Neural Information Processing Systems</source>. (<year>2017</year>). p. 30.</citation>
</ref>
<ref id="B242">
<label>242.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>Z</given-names></name> <name><surname>Dai</surname> <given-names>Z</given-names></name> <name><surname>Yang</surname> <given-names>Y</given-names></name> <name><surname>Carbonell</surname> <given-names>J</given-names></name> <name><surname>Salakhutdinov</surname> <given-names>R</given-names></name> <name><surname>Le</surname> <given-names>QV</given-names></name></person-group>. <article-title>XLNet: generalized autoregressive pretraining for language understanding</article-title>. <source>arXiv preprint arXiv:190608237</source>. (<year>2019</year>).</citation>
</ref>
<ref id="B243">
<label>243.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Howard</surname> <given-names>J</given-names></name> <name><surname>Ruder</surname> <given-names>S</given-names></name></person-group>. <article-title>Universal language model fine-tuning for text classification</article-title>. <source>arXiv preprint arXiv:180106146</source>. (<year>2018</year>).</citation>
</ref>
<ref id="B244">
<label>244.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Devlin</surname> <given-names>J</given-names></name> <name><surname>Chang</surname> <given-names>MW</given-names></name> <name><surname>Lee</surname> <given-names>K</given-names></name> <name><surname>Toutanova</surname> <given-names>K</given-names></name></person-group>. <article-title>Bert: Pre-training of deep bidirectional transformers for language understanding</article-title>. <source>arXiv preprint arXiv:181004805</source>. (<year>2018</year>).</citation>
</ref>
<ref id="B245">
<label>245.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lan</surname> <given-names>Z</given-names></name> <name><surname>Chen</surname> <given-names>M</given-names></name> <name><surname>Goodman</surname> <given-names>S</given-names></name> <name><surname>Gimpel</surname> <given-names>K</given-names></name> <name><surname>Sharma</surname> <given-names>P</given-names></name> <name><surname>Soricut</surname> <given-names>R</given-names></name></person-group>. <article-title>Albert: a lite bert for self-supervised learning of language representations</article-title>. <source>arXiv preprint arXiv:190911942</source>. (<year>2019</year>).</citation>
</ref>
<ref id="B246">
<label>246.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Clark</surname> <given-names>K</given-names></name> <name><surname>Luong</surname> <given-names>MT</given-names></name> <name><surname>Le</surname> <given-names>QV</given-names></name> <name><surname>Manning</surname> <given-names>CD</given-names></name></person-group>. <article-title>Electra: Pre-training text encoders as discriminators rather than generators</article-title>. <source>arXiv preprint arXiv:200310555</source>. (<year>2020</year>).<pub-id pub-id-type="pmid">34330259</pub-id></citation></ref>
<ref id="B247">
<label>247.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brown</surname> <given-names>T</given-names></name> <name><surname>Mann</surname> <given-names>B</given-names></name> <name><surname>Ryder</surname> <given-names>N</given-names></name> <name><surname>Subbiah</surname> <given-names>M</given-names></name> <name><surname>Kaplan</surname> <given-names>JD</given-names></name> <name><surname>Dhariwal</surname> <given-names>P</given-names></name> <etal/></person-group>. <article-title>Language models are few-shot learners</article-title>. <source>Adv Neural Inf Process Syst</source>. (<year>2020</year>) <volume>33</volume>:<fpage>1877</fpage>&#x02013;<lpage>901</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.2005.14165</pub-id></citation>
</ref>
<ref id="B248">
<label>248.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huson</surname> <given-names>D</given-names></name> <name><surname>Zeng</surname> <given-names>W</given-names></name></person-group>. <article-title>MR-DNA: flexible 5mC-methylation-site recognition in DNA sequences using token classification</article-title>. <source>bioRxiv</source>. (<year>2023</year>).</citation>
</ref>
<ref id="B249">
<label>249.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>M</given-names></name> <name><surname>Huang</surname> <given-names>L</given-names></name> <name><surname>Huang</surname> <given-names>H</given-names></name> <name><surname>Tang</surname> <given-names>H</given-names></name> <name><surname>Zhang</surname> <given-names>N</given-names></name> <name><surname>Yang</surname> <given-names>H</given-names></name> <etal/></person-group>. <article-title>Integrating convolution and self-attention improves language model of human genome for interpreting non-coding regions at base-resolution</article-title>. <source>Nucleic Acids Res</source>. (<year>2022</year>) <volume>50</volume>:<fpage>e81</fpage>&#x02013;<lpage>e81</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkac326</pub-id><pub-id pub-id-type="pmid">35536244</pub-id></citation></ref>
<ref id="B250">
<label>250.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>J</given-names></name> <name><surname>Chen</surname> <given-names>Q</given-names></name> <name><surname>Braun</surname> <given-names>PR</given-names></name> <name><surname>Perzel Mandell</surname> <given-names>KA</given-names></name> <name><surname>Jaffe</surname> <given-names>AE</given-names></name> <name><surname>Tan</surname> <given-names>HY</given-names></name> <etal/></person-group>. <article-title>Deep learning predicts DNA methylation regulatory variants in the human brain and elucidates the genetics of psychiatric disorders</article-title>. <source>Proc Nat Acad Sci</source>. (<year>2022</year>) <volume>119</volume>:<fpage>e2206069119</fpage>. <pub-id pub-id-type="doi">10.1073/pnas.2206069119</pub-id><pub-id pub-id-type="pmid">35969790</pub-id></citation></ref>
<ref id="B251">
<label>251.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ligeti</surname> <given-names>B</given-names></name> <name><surname>Szepesi-Nagy</surname> <given-names>I</given-names></name> <name><surname>Bodn&#x000E1;r</surname> <given-names>B</given-names></name> <name><surname>Ligeti-Nagy</surname> <given-names>N</given-names></name> <name><surname>Juh&#x000E1;sz</surname> <given-names>J</given-names></name></person-group>. <article-title>ProkBERT family: genomic language models for microbiome applications</article-title>. <source>Front Microbiol</source>. (<year>2024</year>) <volume>14</volume>:<fpage>1331233</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2023.1331233</pub-id><pub-id pub-id-type="pmid">38282738</pub-id></citation></ref>
<ref id="B252">
<label>252.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Carvalho</surname> <given-names>V</given-names></name> <name><surname>Nepomuceno</surname> <given-names>T</given-names></name> <name><surname>Poleto</surname> <given-names>T</given-names></name> <name><surname>Turet</surname> <given-names>J</given-names></name> <name><surname>Costa</surname> <given-names>A</given-names></name></person-group>. <article-title>Mining public opinions on COVID-19 vaccination: a temporal analysis to support combating misinformation</article-title>. <source>Trop Med Infect Dis</source>. (<year>2022</year>) <volume>7</volume>:<fpage>256</fpage>. <pub-id pub-id-type="doi">10.3390/tropicalmed7100256</pub-id><pub-id pub-id-type="pmid">36287997</pub-id></citation></ref>
<ref id="B253">
<label>253.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lestari</surname> <given-names>N</given-names></name></person-group>. <article-title>A comparison of artificial neural network and naive Bayes classification using unbalanced data handling</article-title>. <source>Barekeng Jurnal Ilmu Matematika Dan Terapan</source>. (<year>2023</year>) <volume>17</volume>:<fpage>1585</fpage>&#x02013;<lpage>94</lpage>. <pub-id pub-id-type="doi">10.30598/barekengvol17iss3pp1585-1594</pub-id></citation>
</ref>
<ref id="B254">
<label>254.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ayyadevara</surname> <given-names>V</given-names></name></person-group>. <article-title>Random forest</article-title>. (<year>2018</year>). p. <fpage>105</fpage>&#x02013;<lpage>116</lpage>. <pub-id pub-id-type="doi">10.1007/978-1-4842-3564-5_5</pub-id></citation>
</ref>
<ref id="B255">
<label>255.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Weitschek</surname> <given-names>E</given-names></name> <name><surname>Fiscon</surname> <given-names>G</given-names></name> <name><surname>Felici</surname> <given-names>G</given-names></name></person-group>. <article-title>Supervised DNA barcodes species classification: analysis, comparisons and results</article-title>. <source>Biodata Min</source>. (<year>2014</year>) <volume>7</volume>:<fpage>4</fpage>. <pub-id pub-id-type="doi">10.1186/1756-0381-7-4</pub-id><pub-id pub-id-type="pmid">24721333</pub-id></citation></ref>
<ref id="B256">
<label>256.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gata</surname> <given-names>W</given-names></name> <name><surname>Bayhaqy</surname> <given-names>A</given-names></name></person-group>. <article-title>Analysis sentiment about islamophobia when Christchurch attack on social media</article-title>. <source>Telkomnika</source>. (<year>2020</year>) <volume>18</volume>:<fpage>1819</fpage>. <pub-id pub-id-type="doi">10.12928/telkomnika.v18i4.14179</pub-id></citation>
</ref>
<ref id="B257">
<label>257.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tamrakar</surname> <given-names>S</given-names></name> <name><surname>Bal</surname> <given-names>B</given-names></name> <name><surname>Thapa</surname> <given-names>R</given-names></name></person-group>. <article-title>Aspect based sentiment analysis of Nepali text using support vector machine and naive Bayes</article-title>. <source>Technical Journal</source>. (<year>2020</year>) <volume>2</volume>:<fpage>22</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.3126/tj.v2i1.32824</pub-id></citation>
</ref>
<ref id="B258">
<label>258.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Muludi</surname> <given-names>K</given-names></name> <name><surname>Akbar</surname> <given-names>M</given-names></name> <name><surname>Shofiana</surname> <given-names>D</given-names></name> <name><surname>Syarif</surname> <given-names>A</given-names></name></person-group>. <article-title>Sentiment analysis of energy independence tweets using simple recurrent neural network</article-title>. <source>IJCCS</source>. (<year>2021</year>) <volume>15</volume>:<fpage>339</fpage>. <pub-id pub-id-type="doi">10.22146/ijccs.66016</pub-id></citation>
</ref>
<ref id="B259">
<label>259.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ardianto</surname> <given-names>R</given-names></name> <name><surname>Rivanie</surname> <given-names>T</given-names></name> <name><surname>Alkhalifi</surname> <given-names>Y</given-names></name> <name><surname>Nugraha</surname> <given-names>F</given-names></name> <name><surname>Gata</surname> <given-names>W</given-names></name></person-group>. <article-title>Sentiment analysis on e-sports for education curriculum using naive Bayes and support vector machine</article-title>. <source>Jurnal Ilmu Komputer Dan Informasi</source>. (<year>2020</year>) <volume>13</volume>:<fpage>109</fpage>&#x02013;<lpage>22</lpage>. <pub-id pub-id-type="doi">10.21609/jiki.v13i2.885</pub-id></citation>
</ref>
<ref id="B260">
<label>260.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Arilya</surname> <given-names>R</given-names></name> <name><surname>Azhar</surname> <given-names>Y</given-names></name> <name><surname>Chandranegara</surname> <given-names>D</given-names></name></person-group>. <article-title>Sentiment analysis on work from home policy using na&#x000EF;ve bayes method and particle swarm optimization</article-title>. <source>Jurnal Ilmiah Teknik Elektro Komputer Dan Informatika</source>. (<year>2021</year>) <volume>7</volume>:<fpage>433</fpage>. <pub-id pub-id-type="doi">10.26555/jiteki.v7i3.22080</pub-id></citation>
</ref>
<ref id="B261">
<label>261.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kamath</surname> <given-names>U</given-names></name> <name><surname>Shehu</surname> <given-names>A</given-names></name> <name><surname>Jong</surname> <given-names>K</given-names></name> <name><surname>A</surname></name></person-group>. <article-title>two-stage evolutionary approach for effective classification of hypersensitive DNA sequences</article-title>. <source>J Bioinform Comput Biol</source>. (<year>2011</year>) <volume>09</volume>:<fpage>399</fpage>&#x02013;<lpage>413</lpage>. <pub-id pub-id-type="doi">10.1142/S0219720011005586</pub-id><pub-id pub-id-type="pmid">21714132</pub-id></citation></ref>
<ref id="B262">
<label>262.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jannah</surname> <given-names>N</given-names></name></person-group>. <article-title>Comparison of na&#x000EF;ve Bayes and SVM in sentiment analysis of product reviews on marketplaces</article-title>. <source>Sinkron</source>. (<year>2024</year>) <volume>8</volume>:<fpage>727</fpage>&#x02013;<lpage>33</lpage>. <pub-id pub-id-type="doi">10.33395/sinkron.v8i2.13559</pub-id></citation>
</ref>
<ref id="B263">
<label>263.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gunasekaran</surname> <given-names>H</given-names></name> <name><surname>Ramalakshmi</surname> <given-names>K</given-names></name> <name><surname>Arokiaraj</surname> <given-names>A</given-names></name> <name><surname>Kanmani</surname> <given-names>S</given-names></name> <name><surname>Dhas</surname> <given-names>C</given-names></name></person-group>. <article-title>Analysis of DNA sequence classification using CNN and hybrid models</article-title>. <source>Comput Math Methods Med</source>. (<year>2021</year>) <volume>2021</volume>:<fpage>1</fpage>&#x02013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1155/2021/1835056</pub-id><pub-id pub-id-type="pmid">34306171</pub-id></citation></ref>
<ref id="B264">
<label>264.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Asim</surname> <given-names>MN</given-names></name> <name><surname>Malik</surname> <given-names>MI</given-names></name> <name><surname>Zehe</surname> <given-names>C</given-names></name> <name><surname>Trygg</surname> <given-names>J</given-names></name> <name><surname>Dengel</surname> <given-names>A</given-names></name> <name><surname>Ahmed</surname> <given-names>S</given-names></name> <etal/></person-group>. <article-title>robust and precise convnet for small non-coding RNA classification (RPC-snrc)</article-title>. <source>IEEE Access</source>. (<year>2020</year>) <volume>9</volume>:<fpage>19379</fpage>&#x02013;<lpage>90</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2020.3037642</pub-id></citation>
</ref>
<ref id="B265">
<label>265.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Asim</surname> <given-names>MN</given-names></name> <name><surname>Ibrahim</surname> <given-names>MA</given-names></name> <name><surname>Imran Malik</surname> <given-names>M</given-names></name> <name><surname>Dengel</surname> <given-names>A</given-names></name> <name><surname>Ahmed</surname> <given-names>S</given-names></name></person-group>. <article-title>Circ-LocNet: a computational framework for circular RNA sub-cellular localization prediction</article-title>. <source>Int J Mol Sci</source>. (<year>2022</year>) <volume>23</volume>:<fpage>8221</fpage>. <pub-id pub-id-type="doi">10.3390/ijms23158221</pub-id><pub-id pub-id-type="pmid">35897818</pub-id></citation></ref>
<ref id="B266">
<label>266.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chong</surname> <given-names>S</given-names></name> <name><surname>Peyton</surname> <given-names>PJ</given-names></name></person-group>. <article-title>A meta-analysis of the accuracy and precision of the ultrasonic cardiac output monitor (USCOM)</article-title>. <source>Anaesthesia</source>. (<year>2012</year>) <volume>67</volume>:<fpage>1266</fpage>&#x02013;<lpage>71</lpage>. <pub-id pub-id-type="doi">10.1111/j.1365-2044.2012.07311.x</pub-id><pub-id pub-id-type="pmid">22928650</pub-id></citation></ref>
<ref id="B267">
<label>267.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Asim</surname> <given-names>MN</given-names></name> <name><surname>Ibrahim</surname> <given-names>MA</given-names></name> <name><surname>Malik</surname> <given-names>MI</given-names></name> <name><surname>Zehe</surname> <given-names>C</given-names></name> <name><surname>Cloarec</surname> <given-names>O</given-names></name> <name><surname>Trygg</surname> <given-names>J</given-names></name> <etal/></person-group>. <article-title>EL-RMLocNet: An explainable LSTM network for RNA-associated multi-compartment localization prediction</article-title>. <source>Comput Struct Biotechnol J</source>. (<year>2022</year>) <volume>20</volume>:<fpage>3986</fpage>&#x02013;<lpage>4002</lpage>. <pub-id pub-id-type="doi">10.1016/j.csbj.2022.07.031</pub-id><pub-id pub-id-type="pmid">35983235</pub-id></citation></ref>
<ref id="B268">
<label>268.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Saloom</surname> <given-names>RH</given-names></name> <name><surname>Khafaji</surname> <given-names>HK</given-names></name></person-group>. <article-title>Mutation types and pathogenicity classification using multi-label multi-class deep networks</article-title>. In: <source>AIP Conference Proceedings</source>. AIP Publishing (<year>2024</year>). <pub-id pub-id-type="doi">10.1063/5.0213291</pub-id></citation>
</ref>
<ref id="B269">
<label>269.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Butucea</surname> <given-names>C</given-names></name> <name><surname>Ndaoud</surname> <given-names>M</given-names></name> <name><surname>Stepanova</surname> <given-names>NA</given-names></name> <name><surname>Tsybakov</surname> <given-names>AB</given-names></name></person-group>. <article-title>Variable selection with Hamming loss</article-title>. <source>Ann Statist</source>. (<year>2018</year>) <volume>46</volume>:<fpage>1837</fpage>&#x02013;<lpage>1875</lpage>. <pub-id pub-id-type="doi">10.1214/17-AOS1572</pub-id></citation>
</ref>
<ref id="B270">
<label>270.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Allen</surname> <given-names>DM</given-names></name></person-group>. <article-title>Mean square error of prediction as a criterion for selecting variables</article-title>. <source>Technometrics</source>. (<year>1971</year>) <volume>13</volume>:<fpage>469</fpage>&#x02013;<lpage>75</lpage>. <pub-id pub-id-type="doi">10.1080/00401706.1971.10488811</pub-id></citation>
</ref>
<ref id="B271">
<label>271.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hodson</surname> <given-names>TO</given-names></name></person-group>. <article-title>Root mean square error (RMSE) or mean absolute error (MAE): when to use them or not</article-title>. <source>Geosci Model Dev Disc</source>. (<year>2022</year>) <volume>2022</volume>:<fpage>1</fpage>&#x02013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.5194/gmd-2022-64</pub-id></citation>
</ref>
<ref id="B272">
<label>272.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kato</surname> <given-names>T</given-names></name></person-group>. <article-title>Prediction of photovoltaic power generation output and network operation</article-title>. In: <source>Integration of Distributed Energy Resources in Power Systems</source>. <publisher-loc>Elsevier</publisher-loc> (<year>2016</year>). p. <fpage>77</fpage>&#x02013;<lpage>108</lpage>. <pub-id pub-id-type="doi">10.1016/B978-0-12-803212-1.00004-0</pub-id></citation>
</ref>
<ref id="B273">
<label>273.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname> <given-names>S</given-names></name> <name><surname>Kim</surname> <given-names>H</given-names></name> <name><surname>A</surname></name></person-group>. <article-title>new metric of absolute percentage error for intermittent demand forecasts</article-title>. <source>Int J Forecast</source>. (<year>2016</year>) <volume>32</volume>:<fpage>669</fpage>&#x02013;<lpage>79</lpage>. <pub-id pub-id-type="doi">10.1016/j.ijforecast.2015.12.003</pub-id></citation>
</ref>
<ref id="B274">
<label>274.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cameron</surname> <given-names>AC</given-names></name> <name><surname>Windmeijer</surname> <given-names>FA</given-names></name></person-group>. <article-title>An R-squared measure of goodness of fit for some common nonlinear regression models</article-title>. <source>J Econom</source>. (<year>1997</year>) <volume>77</volume>:<fpage>329</fpage>&#x02013;<lpage>42</lpage>. <pub-id pub-id-type="doi">10.1016/S0304-4076(96)01818-0</pub-id><pub-id pub-id-type="pmid">39526294</pub-id></citation></ref>
<ref id="B275">
<label>275.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Azhar</surname> <given-names>M</given-names></name> <name><surname>Blanc</surname> <given-names>P</given-names></name> <name><surname>Asim</surname> <given-names>M</given-names></name> <name><surname>Imran</surname> <given-names>S</given-names></name> <name><surname>Hayat</surname> <given-names>N</given-names></name> <name><surname>Shahid</surname> <given-names>H</given-names></name> <etal/></person-group>. <article-title>The evaluation of reanalysis and analysis products of solar radiation for Sindh province, Pakistan</article-title>. <source>Renewable Energy</source>. (<year>2020</year>) <volume>145</volume>:<fpage>347</fpage>&#x02013;<lpage>62</lpage>. <pub-id pub-id-type="doi">10.1016/j.renene.2019.04.107</pub-id></citation>
</ref>
<ref id="B276">
<label>276.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dudoit</surname> <given-names>S</given-names></name> <name><surname>Gentleman</surname> <given-names>R</given-names></name></person-group>. <article-title>Cluster analysis in DNA microarray experiments</article-title>. In: <source>Bioconductor Short Course Winter</source>. (<year>2002</year>). p. 506.</citation>
</ref>
<ref id="B277">
<label>277.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rahmanian</surname> <given-names>M</given-names></name> <name><surname>Mansoori</surname> <given-names>EG</given-names></name></person-group>. <article-title>An unsupervised gene selection method based on multivariate normalized mutual information of genes</article-title>. <source>Chemometr Intell Lab Syst</source>. (<year>2022</year>) <volume>222</volume>:<fpage>104512</fpage>. <pub-id pub-id-type="doi">10.1016/j.chemolab.2022.104512</pub-id></citation>
</ref>
<ref id="B278">
<label>278.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lord</surname> <given-names>E</given-names></name> <name><surname>Diallo</surname> <given-names>AB</given-names></name> <name><surname>Makarenkov</surname> <given-names>V</given-names></name></person-group>. <article-title>Classification of bioinformatics workflows using weighted versions of partitioning and hierarchical clustering algorithms</article-title>. <source>BMC Bioinform</source>. (<year>2015</year>) <volume>16</volume>:<fpage>1</fpage>&#x02013;<lpage>19</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-015-0508-1</pub-id><pub-id pub-id-type="pmid">25887434</pub-id></citation></ref>
<ref id="B279">
<label>279.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hossen</surname> <given-names>MB</given-names></name> <name><surname>Auwul</surname> <given-names>MR</given-names></name></person-group>. <article-title>Comparative study of K-means, partitioning around medoids, agglomerative hierarchical, and DIANA clustering algorithms by using cancer datasets</article-title>. <source>Biomed Stat Inform</source>. (<year>2020</year>) <volume>5</volume>:<fpage>20</fpage>. <pub-id pub-id-type="doi">10.11648/j.bsi.20200501.14</pub-id></citation>
</ref>
<ref id="B280">
<label>280.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Singh</surname> <given-names>AK</given-names></name> <name><surname>Mittal</surname> <given-names>S</given-names></name> <name><surname>Malhotra</surname> <given-names>P</given-names></name> <name><surname>Srivastava</surname> <given-names>YV</given-names></name></person-group>. <article-title>Clustering evaluation by davies-bouldin index (DBI) in cereal data using k-means</article-title>. In: <source>2020 Fourth international conference on computing methodologies and communication (ICCMC)</source>. <publisher-loc>IEEE</publisher-loc> (<year>2020</year>). p. <fpage>306</fpage>&#x02013;<lpage>310</lpage>. <pub-id pub-id-type="doi">10.1109/ICCMC48092.2020.ICCMC-00057</pub-id></citation>
</ref>
<ref id="B281">
<label>281.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Le</surname> <given-names>NQK</given-names></name> <name><surname>Ho</surname> <given-names>QT</given-names></name></person-group>. <article-title>Deep transformers and convolutional neural network in identifying DNA N6-methyladenine sites in cross-species genomes</article-title>. <source>Methods</source>. (<year>2022</year>) <volume>204</volume>:<fpage>199</fpage>&#x02013;<lpage>206</lpage>. <pub-id pub-id-type="doi">10.1016/j.ymeth.2021.12.004</pub-id><pub-id pub-id-type="pmid">34915158</pub-id></citation></ref>
<ref id="B282">
<label>282.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>S</given-names></name> <name><surname>Liu</surname> <given-names>Y</given-names></name> <name><surname>Liu</surname> <given-names>Y</given-names></name> <name><surname>Zhang</surname> <given-names>Y</given-names></name> <name><surname>Zhu</surname> <given-names>X</given-names></name></person-group>. <article-title>BERT-5mC: an interpretable model for predicting 5-methylcytosine sites of DNA based on BERT</article-title>. <source>PeerJ</source>. (<year>2023</year>) <volume>11</volume>:<fpage>e16600</fpage>. <pub-id pub-id-type="doi">10.7717/peerj.16600</pub-id><pub-id pub-id-type="pmid">38089911</pub-id></citation></ref>
<ref id="B283">
<label>283.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Martinek</surname> <given-names>V</given-names></name> <name><surname>Cechak</surname> <given-names>D</given-names></name> <name><surname>Gresova</surname> <given-names>K</given-names></name> <name><surname>Alexiou</surname> <given-names>P</given-names></name> <name><surname>Simecek</surname> <given-names>P</given-names></name></person-group>. <article-title>Fine-tuning transformers for genomic tasks</article-title>. <source>bioRxiv</source>. (<year>2022</year>). p. <fpage>2022</fpage>&#x02013;<lpage>02</lpage>. <pub-id pub-id-type="doi">10.1101/2022.02.07.479412</pub-id></citation>
</ref>
<ref id="B284">
<label>284.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>X</given-names></name> <name><surname>Wang</surname> <given-names>GA</given-names></name> <name><surname>Wei</surname> <given-names>Z</given-names></name> <name><surname>Wang</surname> <given-names>H</given-names></name> <name><surname>Zhu</surname> <given-names>X</given-names></name></person-group>. <article-title>Protein-DNA interface hotspots prediction based on fusion features of embeddings of protein language model and handcrafted features</article-title>. <source>Comput Biol Chem</source>. (<year>2023</year>) <volume>107</volume>:<fpage>107970</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiolchem.2023.107970</pub-id><pub-id pub-id-type="pmid">37866116</pub-id></citation></ref>
<ref id="B285">
<label>285.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>W</given-names></name> <name><surname>Liu</surname> <given-names>W</given-names></name> <name><surname>Guo</surname> <given-names>Y</given-names></name> <name><surname>Wang</surname> <given-names>B</given-names></name> <name><surname>Qing</surname> <given-names>H</given-names></name></person-group>. <article-title>Deep contextual representation learning for identifying essential proteins via integrating multisource protein features</article-title>. <source>Chinese J Electr</source>. (<year>2023</year>) <volume>32</volume>:<fpage>868</fpage>&#x02013;<lpage>81</lpage>. <pub-id pub-id-type="doi">10.23919/cje.2022.00.053</pub-id></citation>
</ref>
<ref id="B286">
<label>286.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yan</surname> <given-names>Y</given-names></name> <name><surname>Li</surname> <given-names>W</given-names></name> <name><surname>Wang</surname> <given-names>S</given-names></name> <name><surname>Huang</surname> <given-names>T</given-names></name></person-group>. <article-title>Seq-RBPPred: predicting RNA-binding proteins from sequence</article-title>. <source>ACS Omega</source>. (<year>2024</year>) <volume>9</volume>:<fpage>12734</fpage>&#x02013;<lpage>42</lpage>. <pub-id pub-id-type="doi">10.1021/acsomega.3c08381</pub-id><pub-id pub-id-type="pmid">38524500</pub-id></citation></ref>
<ref id="B287">
<label>287.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>X</given-names></name> <name><surname>Guo</surname> <given-names>H</given-names></name> <name><surname>Zhang</surname> <given-names>F</given-names></name> <name><surname>Wang</surname> <given-names>X</given-names></name> <name><surname>Wu</surname> <given-names>K</given-names></name> <name><surname>Qiu</surname> <given-names>S</given-names></name> <etal/></person-group>. <article-title>HNetGO: protein function prediction via heterogeneous network transformer</article-title>. <source>Brief Bioinform</source>. (<year>2023</year>) <volume>24</volume>:<fpage>bbab556</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbab556</pub-id><pub-id pub-id-type="pmid">37861172</pub-id></citation></ref>
<ref id="B288">
<label>288.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jha</surname> <given-names>K</given-names></name> <name><surname>Saha</surname> <given-names>S</given-names></name> <name><surname>Karmakar</surname> <given-names>S</given-names></name></person-group>. <article-title>Prediction of protein-protein interactions using vision transformer and language model</article-title>. <source>IEEE/ACM Trans Comput Biol Bioinform</source>. (<year>2023</year>) <volume>20</volume>:<fpage>3215</fpage>&#x02013;<lpage>25</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2023.3248797</pub-id><pub-id pub-id-type="pmid">37027644</pub-id></citation></ref>
<ref id="B289">
<label>289.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yu DL Yu</surname> <given-names>ZG</given-names></name> <name><surname>Han GS Li</surname> <given-names>J</given-names></name> <name><surname>Anh</surname> <given-names>V</given-names></name></person-group>. <article-title>Heterogeneous types of miRNA-disease associations stratified by multi-layer network embedding and prediction</article-title>. <source>Biomedicines</source>. (<year>2021</year>) <volume>9</volume>:<fpage>1152</fpage>. <pub-id pub-id-type="doi">10.3390/biomedicines9091152</pub-id><pub-id pub-id-type="pmid">34572337</pub-id></citation></ref>
<ref id="B290">
<label>290.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Cai</surname> <given-names>K</given-names></name> <name><surname>Zhu</surname> <given-names>Y</given-names></name></person-group>. <article-title>A method for identifying essential proteins based on deep convolutional neural network architecture with particle swarm optimization</article-title>. In: <source>2022 Asia Conference on Advanced Robotics, Automation, and Control Engineering (ARACE)</source>. <publisher-loc>IEEE</publisher-loc> (<year>2022</year>). p. <fpage>7</fpage>&#x02013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1109/ARACE56528.2022.00010</pub-id></citation>
</ref>
<ref id="B291">
<label>291.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Duan</surname> <given-names>T</given-names></name> <name><surname>Kuang</surname> <given-names>Z</given-names></name> <name><surname>Wang</surname> <given-names>J</given-names></name> <name><surname>Ma</surname> <given-names>Z</given-names></name></person-group>. <article-title>GBDTLRL2D predicts LncRNA-disease associations using MetaGraph2Vec and K-means based on heterogeneous network</article-title>. <source>Front Cell Dev Biol</source>. (<year>2021</year>) <volume>9</volume>:<fpage>753027</fpage>. <pub-id pub-id-type="doi">10.3389/fcell.2021.753027</pub-id><pub-id pub-id-type="pmid">34977011</pub-id></citation></ref>
<ref id="B292">
<label>292.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>H</given-names></name> <name><surname>Zheng</surname> <given-names>H</given-names></name> <name><surname>Chen</surname> <given-names>DZ</given-names></name> <name><surname>TANGO</surname></name></person-group>. <article-title>A GO-term embedding based method for protein semantic similarity prediction</article-title>. <source>IEEE/ACM Trans Comput Biol Bioinform</source>. (<year>2022</year>) <volume>20</volume>:<fpage>694</fpage>&#x02013;<lpage>706</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2022.3143480</pub-id><pub-id pub-id-type="pmid">35030084</pub-id></citation></ref>
<ref id="B293">
<label>293.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wei MM Yu</surname> <given-names>CQ</given-names></name> <name><surname>Li</surname> <given-names>LP</given-names></name> <name><surname>You</surname> <given-names>ZH</given-names></name> <name><surname>Ren</surname> <given-names>ZH</given-names></name> <name><surname>Guan</surname> <given-names>YJ</given-names></name> <etal/></person-group>. <article-title>LPIH2V: LncRNA-protein interactions prediction using HIN2Vec based on heterogeneous networks model</article-title>. <source>Front Genet</source>. (<year>2023</year>) <volume>14</volume>:<fpage>1122909</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2023.1122909</pub-id><pub-id pub-id-type="pmid">36845392</pub-id></citation></ref>
<ref id="B294">
<label>294.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>JR</given-names></name> <name><surname>You</surname> <given-names>ZH</given-names></name> <name><surname>Cheng</surname> <given-names>L</given-names></name> <name><surname>Ji</surname> <given-names>BY</given-names></name></person-group>. <article-title>Prediction of lncRNA-disease associations via an embedding learning HOPE in heterogeneous information networks</article-title>. <source>Molec Ther-Nucleic Acids</source>. (<year>2021</year>) <volume>23</volume>:<fpage>277</fpage>&#x02013;<lpage>85</lpage>. <pub-id pub-id-type="doi">10.1016/j.omtn.2020.10.040</pub-id><pub-id pub-id-type="pmid">33425486</pub-id></citation></ref>
<ref id="B295">
<label>295.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Z</given-names></name> <name><surname>Gu</surname> <given-names>Y</given-names></name> <name><surname>Zheng</surname> <given-names>S</given-names></name> <name><surname>Yang</surname> <given-names>L</given-names></name> <name><surname>Li</surname> <given-names>J</given-names></name></person-group>. <article-title>MGREL: a multi-graph representation learning-based ensemble learning method for gene-disease association prediction</article-title>. <source>Comput Biol Med</source>. (<year>2023</year>) <volume>155</volume>:<fpage>106642</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2023.106642</pub-id><pub-id pub-id-type="pmid">36805231</pub-id></citation></ref>
<ref id="B296">
<label>296.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>J</given-names></name> <name><surname>Zhang</surname> <given-names>J</given-names></name> <name><surname>Cai</surname> <given-names>Y</given-names></name> <name><surname>Deng</surname> <given-names>L</given-names></name></person-group>. <article-title>Deepmir2go: inferring functions of human micrornas using a deep multi-label classification model</article-title>. <source>Int J Mol Sci</source>. (<year>2019</year>) <volume>20</volume>:<fpage>6046</fpage>. <pub-id pub-id-type="doi">10.3390/ijms20236046</pub-id><pub-id pub-id-type="pmid">31801264</pub-id></citation></ref>
<ref id="B297">
<label>297.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Su</surname> <given-names>XR</given-names></name> <name><surname>Hu</surname> <given-names>L</given-names></name> <name><surname>You</surname> <given-names>ZH</given-names></name> <name><surname>Hu</surname> <given-names>PW</given-names></name> <name><surname>Zhao</surname> <given-names>BW</given-names></name></person-group>. <article-title>Multi-view heterogeneous molecular network representation learning for protein-protein interaction prediction</article-title>. <source>BMC Bioinform</source>. (<year>2022</year>) <volume>23</volume>:<fpage>234</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-022-04766-z</pub-id><pub-id pub-id-type="pmid">35710342</pub-id></citation></ref>
<ref id="B298">
<label>298.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>L</given-names></name> <name><surname>Wu</surname> <given-names>M</given-names></name> <name><surname>Wu</surname> <given-names>Y</given-names></name> <name><surname>Zhang</surname> <given-names>X</given-names></name> <name><surname>Li</surname> <given-names>S</given-names></name> <name><surname>He</surname> <given-names>M</given-names></name> <etal/></person-group>. <article-title>Prediction of the disease causal genes based on heterogeneous network and multi-feature combination method</article-title>. <source>Comput Biol Chem</source>. (<year>2022</year>) <volume>97</volume>:<fpage>107639</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiolchem.2022.107639</pub-id><pub-id pub-id-type="pmid">35217251</pub-id></citation></ref>
<ref id="B299">
<label>299.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>K</given-names></name> <name><surname>Zhou</surname> <given-names>D</given-names></name> <name><surname>Slonim</surname> <given-names>D</given-names></name> <name><surname>Hu</surname> <given-names>X</given-names></name> <name><surname>Cowen</surname> <given-names>L</given-names></name></person-group>. <article-title>MELISSA: semi-supervised embedding for protein function prediction across multiple networks</article-title>. <source>bioRxiv</source>. (<year>2023</year>). p. <fpage>2023</fpage>&#x02013;<lpage>08</lpage>.</citation>
</ref>
<ref id="B300">
<label>300.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wan</surname> <given-names>C</given-names></name> <name><surname>Cozzetto</surname> <given-names>D</given-names></name> <name><surname>Fa</surname> <given-names>R</given-names></name> <name><surname>Jones</surname> <given-names>DT</given-names></name></person-group>. <article-title>Using deep maxout neural networks to improve the accuracy of function prediction from protein interaction networks</article-title>. <source>PLoS ONE</source>. (<year>2019</year>) <volume>14</volume>:<fpage>e0209958</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0209958</pub-id><pub-id pub-id-type="pmid">31335894</pub-id></citation></ref>
<ref id="B301">
<label>301.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Madeddu</surname> <given-names>L</given-names></name> <name><surname>Stilo</surname> <given-names>G</given-names></name> <name><surname>Velardi</surname> <given-names>P</given-names></name></person-group>. <article-title>Network-based methods for disease-gene prediction</article-title>. <source>arXiv preprint arXiv:190210117</source>. (<year>2019</year>). <pub-id pub-id-type="doi">10.1504/IJDMB.2020.109502</pub-id><pub-id pub-id-type="pmid">35009967</pub-id></citation></ref>
<ref id="B302">
<label>302.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Cao</surname> <given-names>W</given-names></name> <name><surname>Chen</surname> <given-names>Y</given-names></name> <name><surname>Yang</surname> <given-names>JY</given-names></name> <name><surname>Xue</surname> <given-names>FY</given-names></name> <name><surname>Yu</surname> <given-names>ZH</given-names></name> <name><surname>Feng</surname> <given-names>J</given-names></name> <etal/></person-group>. <article-title>Metapath-aggregated multilevel graph embedding for miRNA-disease association prediction</article-title>. In: <source>2023 IEEE International Conference on Bioinformatics and Biomedicine (BIBM)</source>. <publisher-loc>IEEE</publisher-loc> (<year>2023</year>). p. <fpage>468</fpage>&#x02013;<lpage>473</lpage>. <pub-id pub-id-type="doi">10.1109/BIBM58861.2023.10385762</pub-id></citation>
</ref>
<ref id="B303">
<label>303.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Prabhakar</surname> <given-names>V</given-names></name> <name><surname>Liu</surname> <given-names>K</given-names></name></person-group>. <article-title>Unsupervised co-optimization of a graph neural network and a knowledge graph embedding model to prioritize causal genes for Alzheimer&#x00027;s Disease</article-title>. <source>medRxiv</source>. (<year>2022</year>). p. <fpage>2022</fpage>&#x02013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1101/2022.10.03.22280657</pub-id></citation>
</ref>
<ref id="B304">
<label>304.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tian</surname> <given-names>Z</given-names></name> <name><surname>Han</surname> <given-names>C</given-names></name> <name><surname>Xu</surname> <given-names>L</given-names></name> <name><surname>Teng</surname> <given-names>Z</given-names></name> <name><surname>Song</surname> <given-names>W</given-names></name></person-group>. <article-title>MGCNSS: miRNA-disease association prediction with multi-layer graph convolution and distance-based negative sample selection strategy</article-title>. <source>Brief Bioinform</source>. (<year>2024</year>) <volume>25</volume>:<fpage>bbae168</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbae168</pub-id><pub-id pub-id-type="pmid">38622356</pub-id></citation></ref>
<ref id="B305">
<label>305.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>XF</given-names></name> <name><surname>Yu</surname> <given-names>CQ</given-names></name> <name><surname>You</surname> <given-names>ZH</given-names></name> <name><surname>Qiao</surname> <given-names>Y</given-names></name> <name><surname>Li</surname> <given-names>ZW</given-names></name> <name><surname>Huang</surname> <given-names>WZ</given-names></name> <etal/></person-group>. <article-title>KS-CMI: A circRNA-miRNA interaction prediction method based on the signed graph neural network and denoising autoencoder</article-title>. <source>Iscience</source>. (<year>2023</year>) <volume>26</volume>:<fpage>107478</fpage>. <pub-id pub-id-type="doi">10.1016/j.isci.2023.107478</pub-id><pub-id pub-id-type="pmid">37583550</pub-id></citation></ref>
<ref id="B306">
<label>306.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chu</surname> <given-names>X</given-names></name> <name><surname>Guan</surname> <given-names>B</given-names></name> <name><surname>Dai</surname> <given-names>L</given-names></name> <name><surname>Liu</surname> <given-names>Jx</given-names></name> <name><surname>Li</surname> <given-names>F</given-names></name> <name><surname>Shang</surname> <given-names>J</given-names></name></person-group>. <article-title>Network embedding framework for driver gene discovery by combining functional and structural information</article-title>. <source>BMC Genom</source>. (<year>2023</year>) <volume>24</volume>:<fpage>426</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-023-09515-x</pub-id><pub-id pub-id-type="pmid">37516822</pub-id></citation></ref>
<ref id="B307">
<label>307.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>L</given-names></name> <name><surname>Peng</surname> <given-names>X</given-names></name> <name><surname>Zeng</surname> <given-names>L</given-names></name> <name><surname>Peng</surname> <given-names>L</given-names></name></person-group>. <article-title>Finding potential lncRNA-disease associations using a boosting-based ensemble learning model</article-title>. <source>Front Genet</source>. (<year>2024</year>) <volume>15</volume>:<fpage>1356205</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2024.1356205</pub-id><pub-id pub-id-type="pmid">38495672</pub-id></citation></ref>
<ref id="B308">
<label>308.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>J</given-names></name> <name><surname>Li</surname> <given-names>J</given-names></name> <name><surname>Kong</surname> <given-names>M</given-names></name> <name><surname>Wang</surname> <given-names>D</given-names></name> <name><surname>Fu</surname> <given-names>K</given-names></name> <name><surname>Shi</surname> <given-names>J</given-names></name> <etal/></person-group>. <article-title>predicting lncRNA-disease associations by Singular Value Decomposition and node2vec</article-title>. <source>BMC Bioinform</source>. (<year>2021</year>) <volume>22</volume>:<fpage>1</fpage>&#x02013;<lpage>18</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-021-04457-1</pub-id><pub-id pub-id-type="pmid">34727886</pub-id></citation></ref>
<ref id="B309">
<label>309.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mallick</surname> <given-names>K</given-names></name> <name><surname>Bandyopadhyay</surname> <given-names>S</given-names></name> <name><surname>Chakraborty</surname> <given-names>S</given-names></name> <name><surname>Choudhuri</surname> <given-names>R</given-names></name> <name><surname>Bose</surname> <given-names>S</given-names></name></person-group>. <article-title>Topo2vec: a novel node embedding generation based on network topology for link prediction</article-title>. <source>IEEE Trans Comput Soc Syst</source>. (<year>2019</year>) <volume>6</volume>:<fpage>1306</fpage>&#x02013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.1109/TCSS.2019.2950589</pub-id></citation>
</ref>
<ref id="B310">
<label>310.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vilela</surname> <given-names>J</given-names></name> <name><surname>Asif</surname> <given-names>M</given-names></name> <name><surname>Marques</surname> <given-names>AR</given-names></name> <name><surname>Santos</surname> <given-names>JX</given-names></name> <name><surname>Rasga</surname> <given-names>C</given-names></name> <name><surname>Vicente</surname> <given-names>A</given-names></name> <etal/></person-group>. <article-title>Biomedical knowledge graph embeddings for personalized medicine: Predicting disease-gene associations</article-title>. <source>Expert Systems</source>. (<year>2023</year>) <volume>40</volume>:<fpage>e13181</fpage>. <pub-id pub-id-type="doi">10.1111/exsy.13181</pub-id></citation>
</ref>
<ref id="B311">
<label>311.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Franke</surname> <given-names>JK</given-names></name> <name><surname>Runge</surname> <given-names>F</given-names></name> <name><surname>Koeksal</surname> <given-names>R</given-names></name> <name><surname>Backofen</surname> <given-names>R</given-names></name> <name><surname>Hutter</surname> <given-names>F</given-names></name></person-group>. <article-title>RNAformer: a simple yet effective deep learning model for rna secondary structure prediction</article-title>. <source>bioRxiv</source>. (<year>2024</year>) p. <fpage>2024</fpage>&#x02013;<lpage>02</lpage>. <pub-id pub-id-type="doi">10.1101/2024.02.12.579881</pub-id></citation>
</ref>
<ref id="B312">
<label>312.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>M</given-names></name> <name><surname>Yuan</surname> <given-names>F</given-names></name> <name><surname>Yang</surname> <given-names>K</given-names></name> <name><surname>Ju</surname> <given-names>F</given-names></name> <name><surname>Su</surname> <given-names>J</given-names></name> <name><surname>Wang</surname> <given-names>H</given-names></name> <etal/></person-group>. <article-title>Exploring evolution-aware &#x00026;-free protein language models as protein function predictors</article-title>. <source>Adv Neural Inf Process Syst</source>. (<year>2022</year>) <volume>35</volume>:<fpage>38873</fpage>&#x02013;<lpage>84</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.2206.06583</pub-id><pub-id pub-id-type="pmid">38965579</pub-id></citation></ref>
<ref id="B313">
<label>313.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shah</surname> <given-names>SMA</given-names></name> <name><surname>Ou</surname> <given-names>YY</given-names></name></person-group>. <article-title>Disto-TRP: an approach for identifying transient receptor potential (TRP) channels using structural information generated by AlphaFold</article-title>. <source>Gene</source>. (<year>2023</year>) <volume>871</volume>:<fpage>147435</fpage>. <pub-id pub-id-type="doi">10.1016/j.gene.2023.147435</pub-id><pub-id pub-id-type="pmid">37075925</pub-id></citation></ref>
<ref id="B314">
<label>314.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>J</given-names></name> <name><surname>Chen</surname> <given-names>S</given-names></name> <name><surname>Yuan</surname> <given-names>Q</given-names></name> <name><surname>Chen</surname> <given-names>J</given-names></name> <name><surname>Li</surname> <given-names>D</given-names></name> <name><surname>Wang</surname> <given-names>L</given-names></name> <etal/></person-group>. <article-title>Predicting the effects of mutations on protein solubility using graph convolution network and protein language model representation</article-title>. <source>J Comput Chem</source>. (<year>2024</year>) <volume>45</volume>:<fpage>436</fpage>&#x02013;<lpage>45</lpage>. <pub-id pub-id-type="doi">10.1002/jcc.27249</pub-id><pub-id pub-id-type="pmid">37933773</pub-id></citation></ref>
<ref id="B315">
<label>315.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hou</surname> <given-names>X</given-names></name> <name><surname>Wang</surname> <given-names>Y</given-names></name> <name><surname>Bu</surname> <given-names>D</given-names></name> <name><surname>Wang</surname> <given-names>Y</given-names></name> <name><surname>Sun</surname> <given-names>S</given-names></name></person-group>. <article-title>EMNGly: predicting N-linked glycosylation sites using the language models for feature extraction</article-title>. <source>Bioinformatics</source>. (<year>2023</year>) <volume>39</volume>:<fpage>btad650</fpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btad650</pub-id><pub-id pub-id-type="pmid">37930896</pub-id></citation></ref>
<ref id="B316">
<label>316.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Roche</surname> <given-names>R</given-names></name> <name><surname>Moussad</surname> <given-names>B</given-names></name> <name><surname>Shuvo</surname> <given-names>MH</given-names></name> <name><surname>Tarafder</surname> <given-names>S</given-names></name> <name><surname>Bhattacharya</surname> <given-names>D</given-names></name></person-group>. <article-title>EquiPNAS: improved protein-nucleic acid binding site prediction using protein-language-model-informed equivariant deep graph neural networks</article-title>. <source>Nucleic Acids Res</source>. (<year>2024</year>) <volume>52</volume>:<fpage>e27</fpage>&#x02013;<lpage>e27</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkae039</pub-id><pub-id pub-id-type="pmid">38281252</pub-id></citation></ref>
<ref id="B317">
<label>317.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ma</surname> <given-names>J</given-names></name> <name><surname>Zhao</surname> <given-names>Z</given-names></name> <name><surname>Li</surname> <given-names>T</given-names></name> <name><surname>Liu</surname> <given-names>Y</given-names></name> <name><surname>Ma</surname> <given-names>J</given-names></name> <name><surname>Zhang</surname> <given-names>R</given-names></name></person-group>. <article-title>GraphsformerCPI: graph transformer for compound-protein interaction prediction</article-title>. <source>Interdiscip Sci</source>. (<year>2024</year>) <volume>16</volume>:<fpage>361</fpage>&#x02013;<lpage>377</lpage>. <pub-id pub-id-type="doi">10.1007/s12539-024-00609-y</pub-id><pub-id pub-id-type="pmid">38457109</pub-id></citation></ref>
<ref id="B318">
<label>318.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dai</surname> <given-names>Z</given-names></name> <name><surname>Deng</surname> <given-names>F</given-names></name></person-group>. <article-title>LncPNdeep: a long non-coding RNA classifier based on Large Language Model with peptide and nucleotide embedding</article-title>. <source>bioRxiv</source>. (<year>2023</year>). p. <fpage>2023</fpage>&#x02013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1101/2023.11.29.569323</pub-id></citation>
</ref>
<ref id="B319">
<label>319.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Meier</surname> <given-names>J</given-names></name> <name><surname>Rao</surname> <given-names>R</given-names></name> <name><surname>Verkuil</surname> <given-names>R</given-names></name> <name><surname>Liu</surname> <given-names>J</given-names></name> <name><surname>Sercu</surname> <given-names>T</given-names></name> <name><surname>Rives</surname> <given-names>A</given-names></name></person-group>. <article-title>Language models enable zero-shot prediction of the effects of mutations on protein function</article-title>. <source>Adv Neural Inf Process Syst</source>. (<year>2021</year>) <volume>34</volume>:<fpage>29287</fpage>&#x02013;<lpage>303</lpage>. <pub-id pub-id-type="doi">10.1101/2021.07.09.450648</pub-id><pub-id pub-id-type="pmid">37695922</pub-id></citation></ref>
<ref id="B320">
<label>320.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>S</given-names></name> <name><surname>Onoda</surname> <given-names>A</given-names></name></person-group>. <article-title>Accurate and fast prediction of intrinsically disordered protein by multiple protein language models and ensemble learning</article-title>. <source>J Chem Inf Model</source>. (<year>2023</year>) <volume>64</volume>:<fpage>2901</fpage>&#x02013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.3c01202</pub-id><pub-id pub-id-type="pmid">37883249</pub-id></citation></ref>
<ref id="B321">
<label>321.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Y</given-names></name> <name><surname>Guo</surname> <given-names>Z</given-names></name> <name><surname>Wang</surname> <given-names>K</given-names></name> <name><surname>Gao</surname> <given-names>X</given-names></name> <name><surname>Wang</surname> <given-names>G</given-names></name></person-group>. <article-title>End-to-end interpretable disease-gene association prediction</article-title>. <source>Brief Bioinform</source>. (<year>2023</year>) <volume>24</volume>:<fpage>bbad118</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbad118</pub-id><pub-id pub-id-type="pmid">36987781</pub-id></citation></ref>
<ref id="B322">
<label>322.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zou</surname> <given-names>H</given-names></name> <name><surname>Ji</surname> <given-names>B</given-names></name> <name><surname>Zhang</surname> <given-names>M</given-names></name> <name><surname>Liu</surname> <given-names>F</given-names></name> <name><surname>Xie</surname> <given-names>X</given-names></name> <name><surname>Peng</surname> <given-names>S</given-names></name></person-group>. <article-title>MHGTMDA: molecular heterogeneous graph transformer based on biological entity graph for miRNA-disease associations prediction</article-title>. <source>Molec Ther-Nucleic Acids</source>. (<year>2024</year>) <volume>35</volume>:<fpage>102139</fpage>. <pub-id pub-id-type="doi">10.1016/j.omtn.2024.102139</pub-id><pub-id pub-id-type="pmid">38384447</pub-id></citation></ref>
<ref id="B323">
<label>323.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Melnyk</surname> <given-names>I</given-names></name> <name><surname>Chenthamarakshan</surname> <given-names>V</given-names></name> <name><surname>Chen</surname> <given-names>PY</given-names></name> <name><surname>Das</surname> <given-names>P</given-names></name> <name><surname>Dhurandhar</surname> <given-names>A</given-names></name> <name><surname>Padhi</surname> <given-names>I</given-names></name> <etal/></person-group>. <article-title>Reprogramming pretrained language models for antibody sequence infilling</article-title>. In: <source>International Conference on Machine Learning</source>. <publisher-loc>PMLR</publisher-loc> (<year>2023</year>). p. <fpage>24398</fpage>&#x02013;<lpage>24419</lpage>.</citation>
</ref>
<ref id="B324">
<label>324.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Saadat</surname> <given-names>M</given-names></name> <name><surname>Behjati</surname> <given-names>A</given-names></name> <name><surname>Zare-Mirakabad</surname> <given-names>F</given-names></name> <name><surname>Gharaghani</surname> <given-names>S</given-names></name></person-group>. <article-title>Drug-target binding affinity prediction using transformers</article-title>. (<year>2021</year>). <pub-id pub-id-type="doi">10.1101/2021.09.30.462610</pub-id></citation>
</ref>
<ref id="B325">
<label>325.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lennox</surname> <given-names>M</given-names></name> <name><surname>Robertson</surname> <given-names>N</given-names></name> <name><surname>Devereux</surname> <given-names>B</given-names></name></person-group>. <article-title>Modelling drug-target binding affinity using a BERT based graph neural network</article-title>. In: <source>2021 43rd Annual International Conference of the IEEE Engineering in Medicine</source> &#x00026; <italic>Biology Society (EMBC)</italic>. IEEE (<year>2021</year>). p. <fpage>4348</fpage>&#x02013;<lpage>4353</lpage>. <pub-id pub-id-type="doi">10.1109/EMBC46164.2021.9629695</pub-id><pub-id pub-id-type="pmid">34892183</pub-id></citation></ref>
<ref id="B326">
<label>326.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Elnaggar</surname> <given-names>A</given-names></name> <name><surname>Essam</surname> <given-names>H</given-names></name> <name><surname>Salah-Eldin</surname> <given-names>W</given-names></name> <name><surname>Moustafa</surname> <given-names>W</given-names></name> <name><surname>Elkerdawy</surname> <given-names>M</given-names></name> <name><surname>Rochereau</surname> <given-names>C</given-names></name> <etal/></person-group>. <article-title>Ankh: optimized protein language model unlocks general-purpose modelling</article-title>. <source>arXiv preprint arXiv:230106568</source>. (<year>2023</year>).</citation>
</ref>
<ref id="B327">
<label>327.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Haselbeck</surname> <given-names>F</given-names></name> <name><surname>John</surname> <given-names>M</given-names></name> <name><surname>Zhang</surname> <given-names>Y</given-names></name> <name><surname>Pirnay</surname> <given-names>J</given-names></name> <name><surname>Fuenzalida-Werner</surname> <given-names>JP</given-names></name> <name><surname>Costa</surname> <given-names>RD</given-names></name> <etal/></person-group>. <article-title>Superior protein thermophilicity prediction with protein language model embeddings</article-title>. <source>NAR Gen Bioinform</source>. (<year>2023</year>) <volume>5</volume>:<fpage>lqad087</fpage>. <pub-id pub-id-type="doi">10.1093/nargab/lqad087</pub-id><pub-id pub-id-type="pmid">37829176</pub-id></citation></ref>
<ref id="B328">
<label>328.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yuan</surname> <given-names>Q</given-names></name> <name><surname>Tian</surname> <given-names>C</given-names></name> <name><surname>Song</surname> <given-names>Y</given-names></name> <name><surname>Ou</surname> <given-names>P</given-names></name> <name><surname>Zhu</surname> <given-names>M</given-names></name> <name><surname>Zhao</surname> <given-names>H</given-names></name> <etal/></person-group>. <article-title>GPSFun: geometry-aware protein sequence function predictions with language models</article-title>. <source>Nucleic Acids Res</source>. (<year>2024</year>) <volume>52</volume>:<fpage>gkae381</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkae381</pub-id><pub-id pub-id-type="pmid">38738636</pub-id></citation></ref>
<ref id="B329">
<label>329.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>F</given-names></name> <name><surname>Yang</surname> <given-names>R</given-names></name> <name><surname>Zhang</surname> <given-names>C</given-names></name> <name><surname>Zhang</surname> <given-names>L</given-names></name></person-group>. <article-title>A deep learning framework combined with word embedding to identify DNA replication origins</article-title>. <source>Sci Rep</source>. (<year>2021</year>) <volume>11</volume>:<fpage>844</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-020-80670-x</pub-id><pub-id pub-id-type="pmid">33436981</pub-id></citation></ref>
<ref id="B330">
<label>330.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>Y</given-names></name> <name><surname>Wu</surname> <given-names>T</given-names></name> <name><surname>Jiang</surname> <given-names>Y</given-names></name> <name><surname>Li</surname> <given-names>Y</given-names></name> <name><surname>Li</surname> <given-names>K</given-names></name> <name><surname>Quan</surname> <given-names>L</given-names></name> <etal/></person-group>. <article-title>DeepNup: prediction of nucleosome positioning from DNA sequences using deep neural network</article-title>. <source>Genes</source>. (<year>2022</year>) <volume>13</volume>:<fpage>1983</fpage>. <pub-id pub-id-type="doi">10.3390/genes13111983</pub-id><pub-id pub-id-type="pmid">36360220</pub-id></citation></ref>
<ref id="B331">
<label>331.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lei</surname> <given-names>R</given-names></name> <name><surname>Jia</surname> <given-names>J</given-names></name> <name><surname>Qin</surname> <given-names>L</given-names></name> <name><surname>Wei</surname> <given-names>X</given-names></name></person-group>. <article-title>iPro2L-DG: Hybrid network based on improved densenet and global attention mechanism for identifying promoter sequences</article-title>. <source>Heliyon</source>. (<year>2024</year>) <volume>10</volume>:<fpage>e27364</fpage>. <pub-id pub-id-type="doi">10.1016/j.heliyon.2024.e27364</pub-id><pub-id pub-id-type="pmid">38510021</pub-id></citation></ref>
<ref id="B332">
<label>332.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zheng</surname> <given-names>L</given-names></name> <name><surname>Liu</surname> <given-names>L</given-names></name> <name><surname>Zhu</surname> <given-names>W</given-names></name> <name><surname>Ding</surname> <given-names>Y</given-names></name> <name><surname>Wu</surname> <given-names>F</given-names></name></person-group>. <article-title>Predicting enhancer-promoter interaction based on epigenomic signals</article-title>. <source>Front Genet</source>. (<year>2023</year>) <volume>14</volume>:<fpage>1133775</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2023.1133775</pub-id><pub-id pub-id-type="pmid">37144127</pub-id></citation></ref>
<ref id="B333">
<label>333.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>W</given-names></name> <name><surname>Li</surname> <given-names>M</given-names></name> <name><surname>Xiao</surname> <given-names>H</given-names></name> <name><surname>Guan</surname> <given-names>L</given-names></name></person-group>. <article-title>Essential genes identification model based on sequence feature map and graph convolutional neural network</article-title>. <source>BMC Genomics</source>. (<year>2024</year>) <volume>25</volume>:<fpage>47</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-024-09958-w</pub-id><pub-id pub-id-type="pmid">38200437</pub-id></citation></ref>
<ref id="B334">
<label>334.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Reddy</surname> <given-names>AJ</given-names></name> <name><surname>Herschl</surname> <given-names>MH</given-names></name> <name><surname>Kolli</surname> <given-names>S</given-names></name> <name><surname>Lu</surname> <given-names>AX</given-names></name> <name><surname>Geng</surname> <given-names>X</given-names></name> <name><surname>Kumar</surname> <given-names>A</given-names></name> <etal/></person-group>. <article-title>Pretraining strategies for effective promoter-driven gene expression prediction</article-title>. <source>bioRxiv</source>. (<year>2023</year>).<pub-id pub-id-type="pmid">36909524</pub-id></citation></ref>
<ref id="B335">
<label>335.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Le</surname> <given-names>NQK</given-names></name> <name><surname>Xu</surname> <given-names>L</given-names></name></person-group>. <article-title>Optimizing hyperparameter tuning in machine learning to improve the predictive performance of cross-species N6-methyladenosine sites</article-title>. <source>ACS Omega</source>. (<year>2023</year>) <volume>8</volume>:<fpage>39420</fpage>&#x02013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1021/acsomega.3c05074</pub-id><pub-id pub-id-type="pmid">37901522</pub-id></citation></ref>
<ref id="B336">
<label>336.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dai</surname> <given-names>Z</given-names></name> <name><surname>Yang</surname> <given-names>Z</given-names></name> <name><surname>Yang</surname> <given-names>Y</given-names></name> <name><surname>Carbonell</surname> <given-names>J</given-names></name> <name><surname>Le</surname> <given-names>QV</given-names></name> <name><surname>Salakhutdinov</surname> <given-names>R</given-names></name></person-group>. <article-title>Transformer-xl: Attentive language models beyond a fixed-length context</article-title>. <source>arXiv preprint arXiv:190102860</source>. (<year>2019</year>).</citation>
</ref>
<ref id="B337">
<label>337.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Radford</surname> <given-names>A</given-names></name> <name><surname>Narasimhan</surname> <given-names>K</given-names></name> <name><surname>Salimans</surname> <given-names>T</given-names></name> <name><surname>Sutskever</surname> <given-names>I</given-names></name></person-group>. <article-title>Improving language understanding by generative pre-training</article-title>. <source>bioRxiv</source>. (<year>2018</year>).</citation>
</ref>
<ref id="B338">
<label>338.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Radford</surname> <given-names>A</given-names></name> <name><surname>Wu</surname> <given-names>J</given-names></name> <name><surname>Child</surname> <given-names>R</given-names></name> <name><surname>Luan</surname> <given-names>D</given-names></name> <name><surname>Amodei</surname> <given-names>D</given-names></name> <name><surname>Sutskever</surname> <given-names>I</given-names></name> <etal/></person-group>. <article-title>Language models are unsupervised multitask learners</article-title>. <source>OpenAI blog</source>. (<year>2019</year>) <volume>1</volume>:<fpage>9</fpage>.<pub-id pub-id-type="pmid">35637722</pub-id></citation></ref>
<ref id="B339">
<label>339.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Achiam</surname> <given-names>J</given-names></name> <name><surname>Adler</surname> <given-names>S</given-names></name> <name><surname>Agarwal</surname> <given-names>S</given-names></name> <name><surname>Ahmad</surname> <given-names>L</given-names></name> <name><surname>Akkaya</surname> <given-names>I</given-names></name> <name><surname>Aleman</surname> <given-names>FL</given-names></name> <etal/></person-group>. <article-title>Gpt-4 technical report</article-title>. <source>arXiv preprint arXiv:230308774</source>. (<year>2023</year>).</citation>
</ref>
<ref id="B340">
<label>340.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ratajczak</surname> <given-names>F</given-names></name> <name><surname>Joblin</surname> <given-names>M</given-names></name> <name><surname>Hildebrandt</surname> <given-names>M</given-names></name> <name><surname>Ringsquandl</surname> <given-names>M</given-names></name> <name><surname>Falter-Braun</surname> <given-names>P</given-names></name> <name><surname>Heinig</surname> <given-names>M</given-names></name></person-group>. <article-title>Speos: an ensemble graph representation learning framework to predict core gene candidates for complex diseases</article-title>. <source>Nat Commun</source>. (<year>2023</year>) <volume>14</volume>:<fpage>7206</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-023-42975-z</pub-id><pub-id pub-id-type="pmid">37938585</pub-id></citation></ref>
<ref id="B341">
<label>341.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>H</given-names></name> <name><surname>Ryu</surname> <given-names>J</given-names></name> <name><surname>Vinyard</surname> <given-names>ME</given-names></name> <name><surname>Lerer</surname> <given-names>A</given-names></name> <name><surname>Pinello</surname> <given-names>L</given-names></name></person-group>. <article-title>SIMBA: single-cell embedding along with features</article-title>. <source>Nat Methods</source>. (<year>2023</year>) <volume>21</volume>:<fpage>1003</fpage>&#x02013;<lpage>1013</lpage>. <pub-id pub-id-type="doi">10.1038/s41592-023-01899-8</pub-id><pub-id pub-id-type="pmid">37248389</pub-id></citation></ref>
<ref id="B342">
<label>342.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Han GS Li</surname> <given-names>Q</given-names></name> <name><surname>Li</surname> <given-names>Y</given-names></name></person-group>. <article-title>Nucleosome positioning based on DNA sequence embedding and deep learning</article-title>. <source>BMC Genom</source>. (<year>2022</year>) <volume>23</volume>:<fpage>301</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-022-08508-6</pub-id><pub-id pub-id-type="pmid">35418074</pub-id></citation></ref>
<ref id="B343">
<label>343.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Y</given-names></name> <name><surname>Zou</surname> <given-names>J</given-names></name></person-group>. <article-title>GenePT: a simple but effective foundation model for genes and cells built from ChatGPT</article-title>. <source>bioRxiv</source>. (<year>2024</year>). <pub-id pub-id-type="doi">10.1101/2023.10.16.562533</pub-id><pub-id pub-id-type="pmid">37905130</pub-id></citation></ref>
<ref id="B344">
<label>344.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>D</given-names></name> <name><surname>Zhang</surname> <given-names>W</given-names></name> <name><surname>He</surname> <given-names>B</given-names></name> <name><surname>Zhang</surname> <given-names>J</given-names></name> <name><surname>Qin</surname> <given-names>C</given-names></name> <name><surname>Yao</surname> <given-names>J</given-names></name></person-group>. <article-title>Dnagpt: A generalized pretrained tool for multiple DNA sequence analysis tasks</article-title>. <source>bioRxiv</source>. (<year>2023</year>). p. <fpage>2023</fpage>&#x02013;<lpage>07</lpage>. <pub-id pub-id-type="doi">10.1101/2023.07.11.548628</pub-id></citation>
</ref>
<ref id="B345">
<label>345.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Elnaggar</surname> <given-names>A</given-names></name> <name><surname>Heinzinger</surname> <given-names>M</given-names></name> <name><surname>Dallago</surname> <given-names>C</given-names></name> <name><surname>Rehawi</surname> <given-names>G</given-names></name> <name><surname>Wang</surname> <given-names>Y</given-names></name> <name><surname>Jones</surname> <given-names>L</given-names></name> <etal/></person-group>. <article-title>Prottrans: Toward understanding the language of life through self-supervised learning</article-title>. <source>IEEE Trans Pattern Anal Mach Intell</source>. (<year>2021</year>) <volume>44</volume>:<fpage>7112</fpage>&#x02013;<lpage>27</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2021.3095381</pub-id></citation>
</ref>
<ref id="B346">
<label>346.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lai</surname> <given-names>PT</given-names></name> <name><surname>Lu</surname> <given-names>Z</given-names></name></person-group>. <article-title>BERT-GT cross-sentence n-ary relation extraction with BERT and graph transformer</article-title>. <source>Bioinformatics</source>. (<year>2020</year>) <volume>36</volume>:<fpage>5678</fpage>&#x02013;<lpage>85</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa1087</pub-id></citation>
</ref>
<ref id="B347">
<label>347.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cui</surname> <given-names>H</given-names></name> <name><surname>Wang</surname> <given-names>C</given-names></name> <name><surname>Maan</surname> <given-names>H</given-names></name> <name><surname>Pang</surname> <given-names>K</given-names></name> <name><surname>Luo</surname> <given-names>F</given-names></name> <name><surname>Duan</surname> <given-names>N</given-names></name> <etal/></person-group>. <article-title>scGPT: toward building a foundation model for single-cell multi-omics using generative AI</article-title>. <source>Nat Methods</source>. (<year>2024</year>) <volume>21</volume>:<fpage>1470</fpage>&#x02013;<lpage>1480</lpage>. <pub-id pub-id-type="doi">10.1038/s41592-024-02201-0</pub-id></citation>
</ref>
<ref id="B348">
<label>348.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Z</given-names></name> <name><surname>Jin</surname> <given-names>J</given-names></name> <name><surname>Long</surname> <given-names>W</given-names></name> <name><surname>Wei</surname> <given-names>L</given-names></name></person-group>. <article-title>PLPMpro: enhancing promoter sequence prediction with prompt-learning based pre-trained language model</article-title>. <source>Comput Biol Med</source>. (<year>2023</year>) <volume>164</volume>:<fpage>107260</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2023.107260</pub-id></citation>
</ref>
<ref id="B349">
<label>349.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Le</surname> <given-names>NQK</given-names></name> <name><surname>Ho</surname> <given-names>QT</given-names></name> <name><surname>Nguyen</surname> <given-names>VN</given-names></name> <name><surname>Chang</surname> <given-names>JS</given-names></name></person-group>. <article-title>BERT-Promoter: An improved sequence-based predictor of DNA promoter using BERT pre-trained model and SHAP feature selection</article-title>. <source>Comput Biol Chem</source>. (<year>2022</year>) <volume>99</volume>:<fpage>107732</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiolchem.2022.107732</pub-id></citation>
</ref>
<ref id="B350">
<label>350.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tenekeci</surname> <given-names>S</given-names></name> <name><surname>Tekir</surname> <given-names>S</given-names></name></person-group>. <article-title>Identifying promoter and enhancer sequences by graph convolutional networks</article-title>. <source>Comput Biol Chem</source>. (<year>2024</year>) <volume>110</volume>:<fpage>108040</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiolchem.2024.108040</pub-id></citation>
</ref>
<ref id="B351">
<label>351.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ju</surname> <given-names>H</given-names></name> <name><surname>Bai</surname> <given-names>J</given-names></name> <name><surname>Jiang</surname> <given-names>J</given-names></name> <name><surname>Che</surname> <given-names>Y</given-names></name> <name><surname>Chen</surname> <given-names>X</given-names></name></person-group>. <article-title>Comparative evaluation and analysis of DNA N4-methylcytosine methylation sites using deep learning</article-title>. <source>Front Genet</source>. (<year>2023</year>) <volume>14</volume>:<fpage>1254827</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2023.1254827</pub-id></citation>
</ref>
<ref id="B352">
<label>352.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hearst</surname> <given-names>MA</given-names></name> <name><surname>Dumais</surname> <given-names>ST</given-names></name> <name><surname>Osuna</surname> <given-names>E</given-names></name> <name><surname>Platt</surname> <given-names>J</given-names></name> <name><surname>Scholkopf</surname> <given-names>B</given-names></name></person-group>. (<year>1998</year>). <article-title>Support vector machines</article-title>. <source>IEEE Intell. Syst.</source> <volume>13</volume>, <fpage>18</fpage>&#x02013;<lpage>28</lpage>.</citation>
</ref>
</ref-list>
</back>
</article>