<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="review-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Syst. Biol.</journal-id>
<journal-title>Frontiers in Systems Biology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Syst. Biol.</abbrev-journal-title>
<issn pub-type="epub">2674-0702</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1402664</article-id>
<article-id pub-id-type="doi">10.3389/fsysb.2024.1402664</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Systems Biology</subject>
<subj-group>
<subject>Mini Review</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>De novo prediction of functional effects of genetic variants from DNA sequences based on context-specific molecular information</article-title>
<alt-title alt-title-type="left-running-head">Yang et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fsysb.2024.1402664">10.3389/fsysb.2024.1402664</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Yang</surname>
<given-names>Jiaxin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2720608/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Das Adhikari</surname>
<given-names>Sikta</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2690583/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Hao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1973278/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Huang</surname>
<given-names>Binbin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Qi</surname>
<given-names>Wenjie</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Cui</surname>
<given-names>Yuehua</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/33255/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Wang</surname>
<given-names>Jianrong</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1392951/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Computational Mathematics</institution>, <institution>Science and Engineering</institution>, <institution>Michigan State University</institution>, <addr-line>East Lansing</addr-line>, <addr-line>MI</addr-line>, <country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Statistics and Probability</institution>, <institution>Michigan State University</institution>, <addr-line>East Lansing</addr-line>, <addr-line>MI</addr-line>, <country>United States</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Biomedical Engineering</institution>, <institution>Michigan State University</institution>, <addr-line>East Lansing</addr-line>, <addr-line>MI</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/29737/overview">Rongling Wu</ext-link>, The Pennsylvania State University (PSU), United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/143825/overview">Shaoyu Li</ext-link>, University of North Carolina at Charlotte, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Jianrong Wang, <email>wangj164@msu.edu</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>03</day>
<month>06</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>4</volume>
<elocation-id>1402664</elocation-id>
<history>
<date date-type="received">
<day>18</day>
<month>03</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>15</day>
<month>05</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Yang, Das Adhikari, Wang, Huang, Qi, Cui and Wang.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Yang, Das Adhikari, Wang, Huang, Qi, Cui and Wang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Deciphering the functional effects of noncoding genetic variants stands as a fundamental challenge in human genetics. Traditional approaches, such as Genome-Wide Association Studies (GWAS), Transcriptome-Wide Association Studies (TWAS), and Quantitative Trait Loci (QTL) studies, are constrained by obscured the underlying molecular-level mechanisms, making it challenging to unravel the genetic basis of complex traits. The advent of Next-Generation Sequencing (NGS) technologies has enabled context-specific genome-wide measurements, encompassing gene expression, chromatin accessibility, epigenetic marks, and transcription factor binding sites, to be obtained across diverse cell types and tissues, paving the way for decoding genetic variation effects directly from DNA sequences only. The <italic>de novo</italic> predictions of functional effects are pivotal for enhancing our comprehension of transcriptional regulation and its disruptions caused by the plethora of noncoding genetic variants linked to human diseases and traits. This review provides a systematic overview of the state-of-the-art models and algorithms for genetic variant effect predictions, including traditional sequence-based models, Deep Learning models, and the cutting-edge Foundation Models. It delves into the ongoing challenges and prospective directions, presenting an in-depth perspective on contemporary developments in this domain.</p>
</abstract>
<kwd-group>
<kwd>genetic variants</kwd>
<kwd>deep learning</kwd>
<kwd>DNA sequence</kwd>
<kwd>disease genetics</kwd>
<kwd>systems genetics</kwd>
<kwd>cellular context specificity</kwd>
<kwd>foundation models</kwd>
</kwd-group>
<contract-num rid="cn001">R01GM131398</contract-num>
<contract-sponsor id="cn001">National Institutes of Health<named-content content-type="fundref-id">10.13039/100000002</named-content>
</contract-sponsor>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Integrative Genetics and Genomics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>Genetic variants have emerged as pivotal factors in the etiology of severe human diseases (<xref ref-type="bibr" rid="B17">Klein et al., 2005</xref>). Therefore, quantitative and systems-level understandings of the relationship between human diseases and genetic variants are critical in precision medicine and clinical care. Over the past decades, the Genome-wide Association Study (GWAS) (<xref ref-type="bibr" rid="B12">Hirschhorn and Daly, 2005</xref>; <xref ref-type="bibr" rid="B31">Visscher et al., 2012</xref>) has revolutionized the field of complex disease genetics, in which millions of single-nucleotide polymorphisms (SNPs) of individuals are tested to identify significant genotype-phenotype associations. However, GWAS grapples with two pronounced limitations that have spurred the quest for advanced methodologies (<xref ref-type="bibr" rid="B29">Tam et al., 2019</xref>). Firstly, it often limited by low statistical power, mainly stemming from the constraints imposed by limited sample sizes and the arduous multi-testing demands. Secondly, the causal relationships between specific genetic variants and diseases remain obscured, partly owing to the ambiguity induced by Linkage Disequilibrium (LD) (<xref ref-type="bibr" rid="B6">Bulik-Sullivan et al., 2015</xref>) and the paucity of insights into the underlying molecular mechanisms. Traditionally, human disease genetics research has centered around SNPs located in protein coding regions, a mere 1.2% of the human genome (<xref ref-type="bibr" rid="B31">Visscher et al., 2012</xref>). Next-generation Sequencing (NGS) (<xref ref-type="bibr" rid="B5">Buermans and den Dunnen, 2014</xref>) technologies like RNA-seq, DNase-seq, and ChIP-seq (<xref ref-type="bibr" rid="B22">Luo et al., 2020</xref>) have empowered researchers to measure gene expression, chromatin accessibility, and transcription factor (TF) binding genome-wide. This advance fuels an exploration of the vast non-coding genome and gives the potential to analyze the effect of genetic variants on nearby local regions.</p>
<p>Given the DNA sequence&#x2019;s fundamental role as the instruction manual for all aspects of life, understanding the function of regulatory genomic elements that control gene expression is paramount. Moving beyond population-based statistical analyses like GWAS and Transcriptome-Wide Association Studies (TWAS) (<xref ref-type="bibr" rid="B32">Wainberg et al., 2019</xref>), direct predictions of genetic variant effects from DNA sequences are pivotal for elucidating the underlying biological mechanisms. This review will explore the evolution of computational models for predicting genetic variant effects genome-wide. We first review the traditional annotation-based models that rely on simple sequence motifs to estimate variant impacts, then dive into the advancements achieved through <italic>de novo</italic> prediction models that leverage deep learning techniques (<xref ref-type="fig" rid="F1">Figure 1</xref>). We conclude by discussing the current challenges in the field of systems genetics and proposing future research directions that hold promise for further breakthroughs.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The development of models for genetic variants&#x2019; effect predictions based on DNA sequences. <bold>(A)</bold> Traditional models leverage multi-omics data resources to annotate and prioritize genetic variants and use static motif PWMs to analyze the gain- and loss-function of TF bindings. <bold>(B)</bold> Deep Learning models, employing CNN, RNN, and Transformer architectures, are designed to predict functional genomics profiles across various cell types. They determine the effects of genetic variants by comparing the predicted genomic profiles for the reference <italic>versus</italic> alternative alleles. <bold>(C)</bold> Foundation Models utilize a self-supervised pre-training strategy based on DNA sequences only, enabling them to be efficiently fine-tuned for a range of downstream tasks, including the prediction of genetic variant effects across different cellular contexts.</p>
</caption>
<graphic xlink:href="fsysb-04-1402664-g001.tif"/>
</fig>
<sec id="s1-1">
<title>Functional variant annotation and prioritization</title>
<p>The ENCODE (<xref ref-type="bibr" rid="B22">Luo et al., 2020</xref>) and the Roadmap Epigenomics Consortium (<xref ref-type="bibr" rid="B3">Bernstein et al., 2010</xref>) have significantly advanced our understanding of the human genome by profiling a wide array of functional noncoding elements through diverse assays. This wealth of data has enabled the functional annotation of genetic variants across the human genome (<xref ref-type="fig" rid="F1">Figure 1A</xref>). GWAVA (<xref ref-type="bibr" rid="B26">Ritchie et al., 2014</xref>), by leveraging a comprehensive suite of genomic and epigenomic annotations, predicts the functional impact of noncoding variants. Its features encompass open chromatin regions, TF binding sites, histone modifications, RNA polymerase interactions, CpG islands, genomic segmentation, evolutionary conservation, genic context, and sequence context. These annotations are synthesized to mitigate the challenges posed by context dependency and the variability of evolutionary conservation signals within regulatory elements. Furthermore, pattern recognition algorithms help to identify DNA sequence motifs overrepresented in regulatory regions of co-expressed genes, enhancing our understanding of gene regulation (<xref ref-type="bibr" rid="B28">Stormo and Fields, 1998</xref>). The Position Weight Matrix (PWM) (<xref ref-type="bibr" rid="B28">Stormo and Fields, 1998</xref>) represents DNA binding sites of different TFs by scoring each potential base at a given genomic position, thereby quantifying the specificity of protein-DNA interactions and facilitating the prediction of new binding sites. An annotation-based approach, Funseq2 (<xref ref-type="bibr" rid="B9">Fu et al., 2014</xref>), integrates these methodologies to analyze loss-of-function and gain-of-function events in TF binding. It calculates motif-breaking scores for variants within TF binding motifs identified by ChIP-seq peaks, and motif-gaining scores for variants in promoters or regulatory elements significantly associated with genes, based on PWM <italic>p</italic>-values for the mutated allele. Funseq2 also incorporates annotation-based features such as conservation, enhancer-gene links, network centrality, and recurrence across samples. However, reliance solely on regulatory annotations and static PWMs has its drawbacks: many variants in non-coding regions do not overlap with regulatory annotations, and novel motifs cannot be discovered through static PWMs (<xref ref-type="bibr" rid="B35">Zhou and Troyanskaya, 2015</xref>; <xref ref-type="bibr" rid="B16">Kelley et al., 2016</xref>).</p>
<p>Addressing these limitations, kmer-SVM (<xref ref-type="bibr" rid="B20">Lee et al., 2011</xref>) emerged as a pioneering model for predicting regulatory elements directly from DNA sequences, bypassing the need for existing annotated motifs. It counts the frequencies of various k-mers within a piece of DNA sequence, employing a support vector machine (SVM) trained on these k-mer features to assess the likelihood of a sequence being a functional genomic regulatory element or a tissue-specific enhancer. Gapped k-mers, utilized as features in the gkm-SVM (<xref ref-type="bibr" rid="B11">Ghandi et al., 2014</xref>), have further enhanced model accuracy in enhancer identification and TF binding site prediction. Moreover, Delta-SVM (<xref ref-type="bibr" rid="B19">Lee et al., 2015</xref>) incorporates the gkm-SVM predictions to assess the disruptive impacts of genetic variants. Despite these advances, the complexity and non-linearity of the underlying regulatory grammar in DNA sequences require further improvements in model performance (<xref ref-type="bibr" rid="B35">Zhou and Troyanskaya, 2015</xref>; <xref ref-type="bibr" rid="B16">Kelley et al., 2016</xref>).</p>
</sec>
<sec id="s1-2">
<title>De novo prediction of genetic variants&#x2019; effects based on deep learning</title>
<p>Deep learning excels in two key capabilities: 1) extracting and representing features, with enhanced flexibility and power, from semi-structured and unstructured data formats, such as texts and images, and 2) approximating various functions effectively through deep layering, with neural networks comprising stacks of linear transformations interspersed with non-linear activations. For the purpose of predicting the effects of genetic variants (<xref ref-type="fig" rid="F1">Figure 1B</xref>), deep learning models typically represent reference DNA sequences using the one-hot encoding (where A &#x3d; [1,0,0,0], C &#x3d; [0,1,0,0], G &#x3d; [0,0,1,0], T &#x3d; [0,0,0,1], and N &#x3d; [0,0,0,0]). The input DNA fragments are represented accordingly, <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mrow>
<mml:mn>4</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the DNA sequence length. Feature extraction from these one-hot encoded sequences to produce sequence embeddings typically employs two foundational architectures: the 1D Convolutional Neural Network (CNN) (<xref ref-type="bibr" rid="B24">O&#x27;Shea and Nash, 2015</xref>) and the Recurrent Neural Network (RNN) (<xref ref-type="bibr" rid="B27">Sherstinsky, 2018</xref>), such as Long Short-Term Memory Network (LSTM) (<xref ref-type="bibr" rid="B27">Sherstinsky, 2018</xref>).</p>
<p>The CNN architecture focuses on local sequence information, with the initial layer acting as a position-weight matrix, so that the convolution operations are analogous to computing PWM scores across the DNA sequence within each sliding window. Subsequent deep CNN layers capture the non-linear and complex sequence signatures, by utilizing the pooling layers to reduce dimensions after each CNN layer. On the other hand, the LSTM architectures capture sequential dependencies in the genome, by incorporating an internal state that reflects the long-term sequential information. Following these feature representation layers, several fully connected layers are then utilized to generate the final predictions. CNNs, in particular, are adept at learning hierarchical layers of complex, nonlinear patterns without requiring strong prior biological assumptions, thus enabling the discovery of novel sequence motifs and their organizational sequence contexts (<xref ref-type="bibr" rid="B35">Zhou and Troyanskaya, 2015</xref>; <xref ref-type="bibr" rid="B16">Kelley et al., 2016</xref>; <xref ref-type="bibr" rid="B25">Quang and Xie, 2016</xref>).</p>
<p>Pioneering applications of deep neural networks in this field, such as DeepSEA (<xref ref-type="bibr" rid="B35">Zhou and Troyanskaya, 2015</xref>) and Basset (<xref ref-type="bibr" rid="B16">Kelley et al., 2016</xref>), have demonstrated the significant potential of CNNs for predicting genetic variants&#x2019; effects based solely on DNA sequences. DeepSEA leverages a multi-task CNN model to predict TF ChIP-seq, DNase-seq, and histone mark ChIP-seq peaks across a variety of cell types, based on the data from the ENCODE and Roadmap Epigenomics projects. Basset focuses on chromatin accessibility, while DanQ (<xref ref-type="bibr" rid="B25">Quang and Xie, 2016</xref>) combines CNN and LSTM to enhance peak profile prediction performance. Trained on the large-scale multi-omics datasets across different cell types from the reference genome, these deep learning models are thus capable of predicting the peak profiles of distinct regulatory factors in a cell-type specific way. For a specific alternative allele of interest, the model&#x2019;s predictions based on the altered genome sequence are compared to those based on the reference genome. The differences in predictions are then used as indicators of the alternative allele&#x2019;s functional disruptions under specific cellular contexts, leading to mechanistic hypotheses of its downstream effects in complex human diseases.</p>
<p>Further advancements have seen models like Basenji (<xref ref-type="bibr" rid="B15">Kelley et al., 2018</xref>), which employs CNN architectures to predict a wider range of genomic signals, including DNase-seq, histone mark ChIP-seq, and CAGE signals across cell types. By using dilated convolution layers, Basenji is able to capture more contextual information around 32&#xa0;kb DNA sequence windows, thereby identifying relevant regulatory sequences over a broader scope. Additionally, efforts to understand genetic variant effects have expanded from modeling the genomic and epigenomic levels to predicting target genes&#x2019; expressions. For instance, ExPecto (<xref ref-type="bibr" rid="B34">Zhou et al., 2018</xref>) predicts the effects on nearby gene expression in a two-stage strategy. First, ExPecto forecasts histone marks, TF, and DNase profiles from DNA sequences, and second, it aggregates the forecasted signals to make predictions of tissue-specific expression. This approach allows for the interpretation of genetic variants&#x2019; effects in the dysregulation of nearby genes. Moreover, BPNet (<xref ref-type="bibr" rid="B2">Avsec et al., 2021a</xref>) has pushed the boundaries further by predicting base-resolution genomic profiles, utilizing a CNN architecture without pooling layers to achieve the single-base pair resolution predictions.</p>
</sec>
<sec id="s1-3">
<title>Cross-species regulatory information and long-range variant effects</title>
<p>Expanding the training dataset is a well-regarded strategy to enhance the accuracy of deep learning models. While new genome-wide functional genomics profiles grow fast, these new datasets primarily provide information that has already been captured by the model from existing datasets in the human genome. The additional benefits of gathering more functional genomics datasets from additional human genomes may decrease, since the genotypes of different individuals are largely similar. In this context, the quest for significantly different training sequences becomes paramount, with a greater potential to develop and refine more sophisticated and precise models.</p>
<p>An intriguing solution lies in the exploration of non-human species as a reservoir of novel training data. The regulatory DNA sequences of species that are genetically related to humans possess sufficient similarities, enabling the application of machine learning models trained across these diverse genomes. Such cross-species training has the potential to enhance the models&#x2019; understanding of regulatory sequence activities. An example of this approach is the expansion of the Basenji model to simultaneously process functional genomic signal tracks from both the mouse and human genomes (<xref ref-type="bibr" rid="B14">Kelley, 2020</xref>). This cross-species training strategy has been shown to yield more accurate predictions on the test set of sequences which has not been seen by the model previously, compared to those trained exclusively on data from a single species. This innovative approach underscores the utility of integrating diverse genomic data sources to significantly advance the precision of predictive models in functional genomics.</p>
<p>However, CNNs, the key architecture in previous models, often struggle with the problem of capturing semantic dependencies over long genomic distances due to their focus on localized feature extraction, which is limited by the filter size. Besides, RNNs can learn long-term dependencies but are hampered by issues like vanishing gradients and inefficiency in dealing with long genomic sequences. This limitation is particularly challenging in modeling complex cell-type specific gene regulation, where distal enhancers can influence gene expression over large distances (<xref ref-type="bibr" rid="B21">Lieberman-Aiden et al., 2009</xref>; <xref ref-type="bibr" rid="B33">Wang et al., 2021</xref>), underscoring the importance in predicting long-range effects of genetic variants. The Transformer model (<xref ref-type="bibr" rid="B30">Vaswani et al., 2017</xref>) has demonstrated remarkable success beyond its initial applications in natural language processing and computer vision, increasingly supplanting traditional CNN and RNN-based models across various domains. Its exceptional capability to capture long-range dependencies without relying on recurrent units renders it more scalable and adaptable for handling large datasets. At the heart of the Transformer architecture is the multi-head self-attention mechanism, which efficiently models dependencies between genomic locations, regardless of their distance (<xref ref-type="bibr" rid="B30">Vaswani et al., 2017</xref>). This ability allows deeper layers of the model to discern increasingly complex relationships, facilitating the prediction of distal genetic variant effects by capturing interactions between genomic locations separated by considerable distances.</p>
<p>Enformer (<xref ref-type="bibr" rid="B1">Avsec et al., 2021b</xref>), a state-of-the-art model leveraging both CNNs and the Transformer architecture, excels in predicting histone marks, TF binding sites, chromatin accessibility, and gene expression across diverse cell types, including those from the genomes of human and mouse. Its design significantly extends the model&#x2019;s receptive field, enabling the identification of distal regulatory elements up to 100&#xa0;kb away. This expansive reach allows Enformer to integrate information from all pertinent regions, such as enhancers, thereby enhancing gene expression prediction. Moreover, the model&#x2019;s attention weights offer greater interpretability, shedding light on the underlying mechanisms of chromatin and gene regulation. With its superior performance of predictions across &#x3e;5,000 functional genome profiles, including gene expressions, Enformer showcases an unparalleled capacity to forecast both local and distal genetic variant effects. This demonstrates the potential of Transformer-based models in advancing our understanding and prediction of genetic regulations underlying complex traits.</p>
</sec>
</sec>
<sec id="s2">
<title>General sequence grammar of variants learned by foundation models</title>
<p>Traditional deep learning models have achieved impressive results in interpreting functional genomic profiles from DNA sequences through supervised learning, where the models are trained to accurately predict experimental genomic tracks based on the sequence representations. However, this approach necessitates a vast amount of labeled data, constraining the models&#x2019; performance and utility in situations where labeled data is scarce. Obtaining high-quality, labeled datasets is often expensive and time-consuming. Moreover, the available data tends to be biased towards certain well-studied cell types with many tracks, neglecting a broad spectrum of cell types yet to be explored. This imbalance results in overrepresented genomic tracks overshadowing the DNA sequence representation, diminishing the efficacy of genomic variant effect prediction in less studied, underrepresented cell types.</p>
<p>In contrast, the development of Foundation Models originally in the fields such as text and image generation illustrates the potential benefits of leveraging context information through a self-supervised pre-training strategy (<xref ref-type="bibr" rid="B8">Devlin et al., 2018</xref>; <xref ref-type="bibr" rid="B4">Brown et al., 2020</xref>). These models, trained on enormous datasets, have demonstrated capabilities surpassing human performance in certain tasks. The pre-training and fine-tuning framework of Foundation Models involves initial training on vast unlabeled datasets, followed by fine-tuning for specific downstream tasks (<xref ref-type="bibr" rid="B8">Devlin et al., 2018</xref>; <xref ref-type="bibr" rid="B4">Brown et al., 2020</xref>). Applied to disease genetics studies, this approach entails pre-training models on unlabeled genomic sequences, which are subsequently fine-tuned for specific genomic interpretation tasks (<xref ref-type="fig" rid="F1">Figure 1C</xref>). This methodology not only mitigates the challenges associated with data scarcity and bias but also enhances the model&#x2019;s ability to understand and predict across a diverse range of cell types and genomic contexts (<xref ref-type="bibr" rid="B13">Ji et al., 2021</xref>).</p>
<p>DNABERT (<xref ref-type="bibr" rid="B13">Ji et al., 2021</xref>) is a pioneer encoder-based Foundation Model in genetics. It processes DNA sequences by breaking them down into k-mers. For input sequences with lengths up to 512&#xa0;bp, 15% of k-mers are randomly replaced by a [MASK] token. The Transformer encoder then leverages context information to reconstruct these masked k-mers without additional information. By accurately reconstructing the masked k-mers, DNABERT captures the fundamental grammatical structures of DNA sequences, enabling it to generate meaningful representations for any given sequence. This model has demonstrated remarkable efficacy across numerous downstream applications (<xref ref-type="bibr" rid="B13">Ji et al., 2021</xref>), such as promoter identification, TF binding site prediction, and the detection of functional genetic variants. Building on DNABERT&#x2019;s foundation, subsequent iterations like DNABERT2 (<xref ref-type="bibr" rid="B36">Zhou et al., 2023</xref>) and DNABERTS (<xref ref-type="bibr" rid="B37">Zhou et al., 2024</xref>) have broadened the scope of Foundation Models to encompass a wider range of species beyond just humans.</p>
<p>The Nucleotide Transformer (<xref ref-type="bibr" rid="B7">Dalla-Torre et al., 2023</xref>), an advanced and larger encoder-based Foundation Model, is pre-trained on DNA sequences with over 2.5 billion parameters and can handle sequences up to 6&#xa0;kb in length. This model has shown remarkable success in a variety of downstream tasks (<xref ref-type="bibr" rid="B7">Dalla-Torre et al., 2023</xref>) after fine-tuning, demonstrating the beneficial impacts of both increased model size and the ability to process longer sequences. Beyond the Transformer architecture, HyenaDNA (<xref ref-type="bibr" rid="B23">Nguyen et al., 2023</xref>) innovatively extends the contextual reach to up to 1 million tokens at the single nucleotide level through the use of global convolutional filters. This significant enhancement enables the model to effectively leverage long-range chromatin regulation at single base pair resolution. Additionally, HyenaDNA introduces novel downstream adaptation methods, such as a unique soft prompt technique. This approach allows for exceptional downstream results without the necessity of updates to the pre-trained model, thus facilitating the seamless application of the Foundation Model to various tasks, including the prediction of genetic variant effects. This revolution in model design and functionality marks a pivotal advancement in our capacity to understand and interpret complex genetic information.</p>
</sec>
<sec sec-type="discussion" id="s3">
<title>Discussions</title>
<p>This review has explored the evolution of models dedicated to predicting the effects of genetic variants using only DNA sequences (<xref ref-type="table" rid="T1">Table 1</xref>). Enabled by the widespread availability of multi-omics datasets and enhanced computational resources, researchers have transitioned from basic feature annotation and motif recognition to the development of sophisticated deep learning models. These models, trained through both supervised and self-supervised approaches, have progressively achieved more accurate predictions of the genetic variant effects across a variety of cell types.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Summary of computational models.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Tool</th>
<th align="center">Model architecture</th>
<th align="center">Required data</th>
<th align="center">Link</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">GWAVA</td>
<td align="center">Annotation-based</td>
<td align="center">Experimental annotation</td>
<td align="center">
<ext-link ext-link-type="uri" xlink:href="https://www.sanger.ac.uk/tool/gwava/">https://www.sanger.ac.uk/tool/gwava/</ext-link>
</td>
</tr>
<tr>
<td align="center">Funseq2</td>
<td align="center">Annotation &#x2b; PWM</td>
<td align="center">Experimental annotation &#x2b; DNA sequence</td>
<td align="center">
<ext-link ext-link-type="uri" xlink:href="http://funseq2.gersteinlab.org/">http://funseq2.gersteinlab.org/</ext-link>
</td>
</tr>
<tr>
<td align="center">Delta-SVM</td>
<td align="center">SVM</td>
<td align="center">DNA sequence</td>
<td align="center">
<ext-link ext-link-type="uri" xlink:href="https://www.beerlab.org/deltasvm/">https://www.beerlab.org/deltasvm/</ext-link>
</td>
</tr>
<tr>
<td align="center">DeepSEA</td>
<td align="center">CNN</td>
<td align="center">DNA sequence &#x2b; experiment peaks</td>
<td align="center">
<ext-link ext-link-type="uri" xlink:href="https://hb.flatironinstitute.org/deepsea/">https://hb.flatironinstitute.org/deepsea/</ext-link>
</td>
</tr>
<tr>
<td align="center">Basset</td>
<td align="center">CNN</td>
<td align="center">DNA sequence &#x2b; experiment peaks</td>
<td align="center">
<ext-link ext-link-type="uri" xlink:href="https://github.com/davek44/Basset">https://github.com/davek44/Basset</ext-link>
</td>
</tr>
<tr>
<td align="center">DanQ</td>
<td align="center">CNN &#x2b; LSTM</td>
<td align="center">DNA sequence &#x2b; experiment peaks</td>
<td align="center">
<ext-link ext-link-type="uri" xlink:href="https://github.com/uci-cbcl/DanQ">https://github.com/uci-cbcl/DanQ</ext-link>
</td>
</tr>
<tr>
<td align="center">Basenji</td>
<td align="center">CNN</td>
<td align="center">DNA sequence &#x2b; experiment signals</td>
<td align="center">
<ext-link ext-link-type="uri" xlink:href="https://github.com/calico/basenji">https://github.com/calico/basenji</ext-link>
</td>
</tr>
<tr>
<td align="center">ExPecto</td>
<td align="center">CNN &#x2b; regression</td>
<td align="center">DNA sequence &#x2b; experiment signals</td>
<td align="center">
<ext-link ext-link-type="uri" xlink:href="https://github.com/FunctionLab/ExPecto">https://github.com/FunctionLab/ExPecto</ext-link>
</td>
</tr>
<tr>
<td align="center">BPNet</td>
<td align="center">CNN</td>
<td align="center">DNA sequence &#x2b; experiment signals</td>
<td align="center">
<ext-link ext-link-type="uri" xlink:href="https://github.com/kundajelab/bpnet/">https://github.com/kundajelab/bpnet/</ext-link>
</td>
</tr>
<tr>
<td align="center">Basenji2</td>
<td align="center">CNN</td>
<td align="center">DNA sequence &#x2b; experiment signals</td>
<td align="center">
<ext-link ext-link-type="uri" xlink:href="https://github.com/calico/basenji">https://github.com/calico/basenji</ext-link>
</td>
</tr>
<tr>
<td align="center">Enformer</td>
<td align="center">CNN &#x2b; Transformer</td>
<td align="center">DNA sequence &#x2b; experiment signals across species</td>
<td align="center">
<ext-link ext-link-type="uri" xlink:href="https://github.com/google-deepmind/deepmind-research/tree/master/enformer">https://github.com/google-deepmind/deepmind-research/tree/master/enformer</ext-link>
</td>
</tr>
<tr>
<td align="center">DNABERT</td>
<td align="center">Transformer</td>
<td align="center">DNA sequence</td>
<td align="center">
<ext-link ext-link-type="uri" xlink:href="https://github.com/jerryji1993/DNABERT">https://github.com/jerryji1993/DNABERT</ext-link>
</td>
</tr>
<tr>
<td align="center">DNABERT2</td>
<td align="center">Transformer</td>
<td align="center">DNA sequence</td>
<td align="center">
<ext-link ext-link-type="uri" xlink:href="https://github.com/MAGICS-LAB/DNABERT_2">https://github.com/MAGICS-LAB/DNABERT_2</ext-link>
</td>
</tr>
<tr>
<td align="center">DNABERTS</td>
<td align="center">Transformer</td>
<td align="center">DNA sequence</td>
<td align="center">
<ext-link ext-link-type="uri" xlink:href="https://github.com/MAGICS-LAB/DNABERT_S">https://github.com/MAGICS-LAB/DNABERT_S</ext-link>
</td>
</tr>
<tr>
<td align="center">The Nucleotide Transformer</td>
<td align="center">Transformer</td>
<td align="center">DNA sequence</td>
<td align="center">
<ext-link ext-link-type="uri" xlink:href="https://github.com/instadeepai/nucleotide-transformer">https://github.com/instadeepai/nucleotide-transformer</ext-link>
</td>
</tr>
<tr>
<td align="center">HyenaDNA</td>
<td align="center">Hyena</td>
<td align="center">DNA sequence</td>
<td align="center">
<ext-link ext-link-type="uri" xlink:href="https://github.com/HazyResearch/hyena-dna">https://github.com/HazyResearch/hyena-dna</ext-link>
</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Despite their advancements, deep learning models for predicting genetic variant effects face two significant challenges: Firstly, model training predominantly relies on labeled data at the cell type level, which limits their capability to discern the functional effects at the single-cell level. With the advent of single-cell sequencing technologies, such as scRNA-seq, scATAC-seq, and scHi-C, there is an influx of data providing detailed insights into gene expression, chromatin accessibility, and regulation at the single-cell level. This type of data, however, tends to be sparse and noisy. Foundation models, pre-trained on the fundamental sequence grammar, exhibit a strong potential for enhancing their performance through fine-tuning with minimal data, addressing the challenge of integrating single-cell level data. Secondly, the training of current models is anchored to the reference genome, neglecting the diversity and frequency of genetic variations across different genotypes. While these models may excel in predicting genetic profiles based on the reference genome, they primarily capture consensus information, which may not accurately represent the actual effects of genetic variants. The discrepancies between the reference and alternative alleles do not fully encapsulate the impact of genetic variants. CRISPR (<xref ref-type="bibr" rid="B18">Korkmaz et al., 2016</xref>; <xref ref-type="bibr" rid="B10">Fulco et al., 2019</xref>) technology, which elucidates the casual and real effects of genetic variants, offers valuable insights beyond the reference genomic context. The CRISPR-derived data is expected to help to fill the gap between model predictions and biological reality.</p>
</sec>
</body>
<back>
<sec id="s4">
<title>Author contributions</title>
<p>JY: Conceptualization, Writing&#x2013;review and editing, Writing&#x2013;original draft, Data curation, Formal Analysis, Visualization. SD: Conceptualization, Writing&#x2013;review and editing. HW: Conceptualization, Writing&#x2013;review and editing. BH: Conceptualization, Writing&#x2013;review and editing. WQ: Conceptualization, Writing&#x2013;review and editing. YC: Conceptualization, Writing&#x2013;review and editing. JW: Conceptualization, Writing&#x2013;review and editing, Funding acquisition, Project administration, Supervision, Writing&#x2013;original draft.</p>
</sec>
<sec sec-type="funding-information" id="s5">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This work is funded, in part, by the awards R01GM131398 from the National Institutes of Health and NSF1942143 from the National Science Foundation</p>
</sec>
<ack>
<p>The authors thank Pronoy Kanti Mondal and Tairan Song for helpful inputs and discussions.</p>
</ack>
<sec sec-type="COI-statement" id="s6">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
<p>The author(s) declared that they were an editorial board member of Frontiers, at the time of submission. This had no impact on the peer review process and the final decision.</p>
</sec>
<sec sec-type="disclaimer" id="s7">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Avsec</surname>
<given-names>&#x17d;.</given-names>
</name>
<name>
<surname>Agarwal</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Visentin</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ledsam</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>Grabska-Barwinska</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Taylor</surname>
<given-names>K. R.</given-names>
</name>
<etal/>
</person-group> (<year>2021b</year>). <article-title>Effective gene expression prediction from sequence by integrating long-range interactions</article-title>. <source>Nat. Methods</source> <volume>18</volume>, <fpage>1196</fpage>&#x2013;<lpage>1203</lpage>. <pub-id pub-id-type="doi">10.1038/s41592-021-01252-x</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Avsec</surname>
<given-names>&#x17d;.</given-names>
</name>
<name>
<surname>Weilert</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Shrikumar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Krueger</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Alexandari</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dalal</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2021a</year>). <article-title>Base-resolution models of transcription-factor binding reveal soft motif syntax</article-title>. <source>Nat. Genet.</source> <volume>53</volume>, <fpage>354</fpage>&#x2013;<lpage>366</lpage>. <pub-id pub-id-type="doi">10.1038/s41588-021-00782-6</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bernstein</surname>
<given-names>B. E.</given-names>
</name>
<name>
<surname>Stamatoyannopoulos</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Costello</surname>
<given-names>J. F.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Milosavljevic</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Meissner</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>The NIH Roadmap epigenomics mapping Consortium</article-title>. <source>Nat. Biotechnol.</source> <volume>28</volume>, <fpage>1045</fpage>&#x2013;<lpage>1048</lpage>. <pub-id pub-id-type="doi">10.1038/nbt1010-1045</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Brown</surname>
<given-names>T. B.</given-names>
</name>
<name>
<surname>Mann</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Ryder</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Subbiah</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Jared</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Dario</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <source>Language models are few-shot learners</source>. <comment>arXiv:2005.14165</comment>. <pub-id pub-id-type="doi">10.48550/arXiv.2005.14165</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Buermans</surname>
<given-names>H. P. J.</given-names>
</name>
<name>
<surname>den Dunnen</surname>
<given-names>J. T.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Next generation sequencing technology: advances and applications</article-title>. <source>Biochimica Biophysica Acta (BBA) - Mol. Basis Dis.</source> <volume>1842</volume>, <fpage>1932</fpage>&#x2013;<lpage>1941</lpage>. <pub-id pub-id-type="doi">10.1016/j.bbadis.2014.06.015</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bulik-Sullivan</surname>
<given-names>B. K.</given-names>
</name>
<name>
<surname>Loh</surname>
<given-names>P. R.</given-names>
</name>
<name>
<surname>Finucane</surname>
<given-names>H. K.</given-names>
</name>
<name>
<surname>Ripke</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>LD Score regression distinguishes confounding from polygenicity in genome-wide association studies</article-title>. <source>Nat. Genet.</source> <volume>47</volume>, <fpage>291</fpage>&#x2013;<lpage>295</lpage>. <pub-id pub-id-type="doi">10.1038/ng.3211</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dalla-Torre</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Gonzalez</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Mendoza-Revilla</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Carranza</surname>
<given-names>N. L.</given-names>
</name>
<name>
<surname>Grzywaczewski</surname>
<given-names>A. H.</given-names>
</name>
<name>
<surname>Oteri</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>The nucleotide transformer: building and evaluating robust foundation models for human genomics</article-title>. <source>bioRxiv</source> <volume>2023</volume>, <fpage>523679</fpage>. <pub-id pub-id-type="doi">10.1101/2023.01.11.523679</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Devlin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>M.-W.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Toutanova</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2018</year>). <source>BERT: pre-training of deep bidirectional transformers for language understanding</source>. <comment>arXiv:1810.04805</comment>. <pub-id pub-id-type="doi">10.48550/arXiv.1810.04805</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Lou</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bedford</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Mu</surname>
<given-names>X. J.</given-names>
</name>
<name>
<surname>Yip</surname>
<given-names>K. Y.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>FunSeq2: a framework for prioritizing noncoding regulatory variants in cancer</article-title>. <source>Genome Biol.</source> <volume>15</volume>, <fpage>480</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-014-0480-5</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fulco</surname>
<given-names>C. P.</given-names>
</name>
<name>
<surname>Nasser</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>T. R.</given-names>
</name>
<name>
<surname>Munson</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Bergman</surname>
<given-names>D. T.</given-names>
</name>
<name>
<surname>Subramanian</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Activity-by-contact model of enhancer&#x2013;promoter regulation from thousands of CRISPR perturbations</article-title>. <source>Nat. Genet.</source> <volume>51</volume>, <fpage>1664</fpage>&#x2013;<lpage>1669</lpage>. <pub-id pub-id-type="doi">10.1038/s41588-019-0538-0</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ghandi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Mohammad-Noori</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Beer</surname>
<given-names>M. A.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Enhanced regulatory sequence prediction using gapped k-mer features</article-title>. <source>PLoS Comput. Biol.</source> <volume>10</volume>, <fpage>e1003711</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1003711</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hirschhorn</surname>
<given-names>J. N.</given-names>
</name>
<name>
<surname>Daly</surname>
<given-names>M. J.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Genome-wide association studies for common diseases and complex traits</article-title>. <source>Nat. Rev. Genet.</source> <volume>6</volume>, <fpage>95</fpage>&#x2013;<lpage>108</lpage>. <pub-id pub-id-type="doi">10.1038/nrg1521</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ji</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Davuluri</surname>
<given-names>R. V.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>DNABERT: pre-trained bidirectional encoder representations from transformers model for DNA-language in genome</article-title>. <source>Bioinformatics</source> <volume>37</volume>, <fpage>2112</fpage>&#x2013;<lpage>2120</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btab083</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kelley</surname>
<given-names>D. R.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Cross-species regulatory sequence activity prediction</article-title>. <source>PLoS Comput. Biol.</source> <volume>16</volume>, <fpage>e1008050</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1008050</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kelley</surname>
<given-names>D. R.</given-names>
</name>
<name>
<surname>Reshef</surname>
<given-names>Y. A.</given-names>
</name>
<name>
<surname>Bileschi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Belanger</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>McLean</surname>
<given-names>C. Y.</given-names>
</name>
<name>
<surname>Snoek</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Sequential regulatory activity prediction across chromosomes with convolutional neural networks</article-title>. <source>Genome Res.</source> <volume>28</volume>, <fpage>739</fpage>&#x2013;<lpage>750</lpage>. <pub-id pub-id-type="doi">10.1101/gr.227819.117</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kelley</surname>
<given-names>D. R.</given-names>
</name>
<name>
<surname>Snoek</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Rinn</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Basset: learning the regulatory code of the accessible genome with deep convolutional neural networks</article-title>. <source>Genome Res.</source> <volume>26</volume>, <fpage>990</fpage>&#x2013;<lpage>999</lpage>. <pub-id pub-id-type="doi">10.1101/gr.200535.115</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Klein</surname>
<given-names>R. J.</given-names>
</name>
<name>
<surname>Zeiss</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Chew</surname>
<given-names>E. Y.</given-names>
</name>
<name>
<surname>Tsai</surname>
<given-names>J. Y.</given-names>
</name>
<name>
<surname>Sackler</surname>
<given-names>R. S.</given-names>
</name>
<name>
<surname>Haynes</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2005</year>). <article-title>Complement factor H polymorphism in age-related macular degeneration</article-title>. <source>Science</source> <volume>308</volume>, <fpage>385</fpage>&#x2013;<lpage>389</lpage>. <pub-id pub-id-type="doi">10.1126/science.1109557</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Korkmaz</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Lopes</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Ugalde</surname>
<given-names>A. P.</given-names>
</name>
<name>
<surname>Nevedomskaya</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Myacheva</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Functional genetic screens for enhancer elements in the human genome using CRISPR-Cas9</article-title>. <source>Nat. Biotechnol.</source> <volume>34</volume>, <fpage>192</fpage>&#x2013;<lpage>198</lpage>. <pub-id pub-id-type="doi">10.1038/nbt.3450</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Gorkin</surname>
<given-names>D. U.</given-names>
</name>
<name>
<surname>Baker</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Strober</surname>
<given-names>B. J.</given-names>
</name>
<name>
<surname>Asoni</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>McCallion</surname>
<given-names>A. S.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>A method to predict the impact of regulatory variants from DNA sequence</article-title>. <source>Nat. Genet.</source> <volume>47</volume>, <fpage>955</fpage>&#x2013;<lpage>961</lpage>. <pub-id pub-id-type="doi">10.1038/ng.3331</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Karchin</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Beer</surname>
<given-names>M. A.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Discriminative prediction of mammalian enhancers from DNA sequence</article-title>. <source>Genome Res.</source> <volume>21</volume>, <fpage>2167</fpage>&#x2013;<lpage>2180</lpage>. <pub-id pub-id-type="doi">10.1101/gr.121905.111</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lieberman-Aiden</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>van Berkum</surname>
<given-names>N. L.</given-names>
</name>
<name>
<surname>Williams</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Imakaev</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ragoczy</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Telling</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>Comprehensive mapping of long-range interactions reveals folding principles of the human genome</article-title>. <source>Science</source> <volume>326</volume>, <fpage>289</fpage>&#x2013;<lpage>293</lpage>. <pub-id pub-id-type="doi">10.1126/science.1181369</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Luo</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hitz</surname>
<given-names>B. C.</given-names>
</name>
<name>
<surname>Gabdank</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Hilton</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Kagda</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Lam</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>New developments on the Encyclopedia of DNA Elements (ENCODE) data portal</article-title>. <source>Nucleic Acids Res.</source> <volume>48</volume>, <fpage>D882</fpage>&#x2013;<lpage>D889</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkz1062</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Nguyen</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Poli</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Faizi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Thomas</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Aman</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Re</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>) <source>HyenaDNA: long-range genomic sequence modeling at single nucleotide resolution</source>. <comment>arXiv:2306.15794</comment>. <pub-id pub-id-type="doi">10.48550/arXiv.2306.15794</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>O&#x27;Shea</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Nash</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2015</year>). <source>An introduction to convolutional neural networks</source>.</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Quang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>DanQ: a hybrid convolutional and recurrent deep neural network for quantifying the function of DNA sequences</article-title>. <source>Nucleic Acids Res.</source> <volume>44</volume>, <fpage>e107</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkw226</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ritchie</surname>
<given-names>G. R. S.</given-names>
</name>
<name>
<surname>Dunham</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Zeggini</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Flicek</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Functional annotation of noncoding sequence variants</article-title>. <source>Nat. Methods</source> <volume>11</volume>, <fpage>294</fpage>&#x2013;<lpage>296</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.2832</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sherstinsky</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Fundamentals of recurrent neural network (RNN) and long short-term memory (LSTM) network</article-title>. <source>Phys. D. Nonlinear Phenom.</source> <volume>404</volume>, <fpage>132306</fpage>. <pub-id pub-id-type="doi">10.1016/j.physd.2019.132306</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stormo</surname>
<given-names>G. D.</given-names>
</name>
<name>
<surname>Fields</surname>
<given-names>D. S.</given-names>
</name>
</person-group> (<year>1998</year>). <article-title>Specificity, free energy and information content in protein&#x2013;DNA interactions</article-title>. <source>Trends Biochem. Sci.</source> <volume>23</volume>, <fpage>109</fpage>&#x2013;<lpage>113</lpage>. <pub-id pub-id-type="doi">10.1016/s0968-0004(98)01187-6</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tam</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Patel</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Turcotte</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Boss&#xe9;</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Par&#xe9;</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Meyre</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Benefits and limitations of genome-wide association studies</article-title>. <source>Nat. Rev. Genet.</source> <volume>20</volume>, <fpage>467</fpage>&#x2013;<lpage>484</lpage>. <pub-id pub-id-type="doi">10.1038/s41576-019-0127-1</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Vaswani</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). &#x201c;<article-title>Attention is all you need</article-title>,&#x201d; in <source>Advances in neural information processing systems</source>. Editor <person-group person-group-type="editor">
<name>
<surname>Guyon</surname>
<given-names>I.</given-names>
</name>
</person-group> (<publisher-name>Curran Associates, Inc.</publisher-name>), <volume>30</volume>.</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Visscher</surname>
<given-names>P. M.</given-names>
</name>
<name>
<surname>Brown</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>McCarthy</surname>
<given-names>M. I.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Five years of GWAS discovery</article-title>. <source>Am. J. Hum. Genet.</source> <volume>90</volume>, <fpage>7</fpage>&#x2013;<lpage>24</lpage>. <pub-id pub-id-type="doi">10.1016/j.ajhg.2011.11.029</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wainberg</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sinnott-Armstrong</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Mancuso</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Barbeira</surname>
<given-names>A. N.</given-names>
</name>
<name>
<surname>Knowles</surname>
<given-names>D. A.</given-names>
</name>
<name>
<surname>Golan</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Opportunities and challenges for transcriptome-wide association studies</article-title>. <source>Nat. Genet.</source> <volume>51</volume>, <fpage>592</fpage>&#x2013;<lpage>599</lpage>. <pub-id pub-id-type="doi">10.1038/s41588-019-0385-z</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Discover novel disease-associated genes based on regulatory networks of long-range chromatin interactions</article-title>. <source>Methods</source> <volume>189</volume>, <fpage>22</fpage>&#x2013;<lpage>33</lpage>. <pub-id pub-id-type="doi">10.1016/j.ymeth.2020.10.010</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Theesfeld</surname>
<given-names>C. L.</given-names>
</name>
<name>
<surname>Yao</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>K. M.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>A. K.</given-names>
</name>
<name>
<surname>Troyanskaya</surname>
<given-names>O. G.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Deep learning sequence-based <italic>ab initio</italic> prediction of variant effects on expression and disease risk</article-title>. <source>Nat. Genet.</source> <volume>50</volume>, <fpage>1171</fpage>&#x2013;<lpage>1179</lpage>. <pub-id pub-id-type="doi">10.1038/s41588-018-0160-6</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Troyanskaya</surname>
<given-names>O. G.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Predicting effects of noncoding variants with deep learning&#x2013;based sequence model</article-title>. <source>Nat. Methods</source> <volume>12</volume>, <fpage>931</fpage>&#x2013;<lpage>934</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.3547</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ji</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Dutta</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Ramana</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>) <source>DNABERT-2: efficient foundation model and benchmark for multi-species genome</source>. <comment>arXiv:2306.15006</comment>. <pub-id pub-id-type="doi">10.48550/arXiv.2306.1500</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Ho</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <source>DNABERT-S: learning species-aware DNA embedding with genome foundation models</source>. <comment>arXiv:2402.08777</comment>. <pub-id pub-id-type="doi">10.48550/arXiv.2402.0877</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>