<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="brief-report" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1254827</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2023.1254827</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Brief Research Report</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Comparative evaluation and analysis of DNA N4-methylcytosine methylation sites using deep learning</article-title>
<alt-title alt-title-type="left-running-head">Ju et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2023.1254827">10.3389/fgene.2023.1254827</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Ju</surname>
<given-names>Hong</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/684405/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Bai</surname>
<given-names>Jie</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Jiang</surname>
<given-names>Jing</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Che</surname>
<given-names>Yusheng</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Chen</surname>
<given-names>Xin</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2370700/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Heilongjiang Agricultural Engineering Vocational College</institution>, <addr-line>Harbin</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Engineering Research Center of Integration and Application of Digital Learning Technology, Ministry of Education</institution>, <addr-line>Hangzhou</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Beidahuang Industry Group General Hospital</institution>, <addr-line>Harbin</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Department of Neurosurgical Laboratory</institution>, <institution>The First Affiliated Hospital of Harbin Medical University</institution>, <addr-line>Harbin</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/827295/overview">Lei Chen</ext-link>, Shanghai Maritime University, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1220765/overview">Lesong Wei</ext-link>, King Abdullah University of Science and Technology, Saudi Arabia</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/182351/overview">Hao Lin</ext-link>, University of Electronic Science and Technology of China, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Xin Chen, <email>chenxinghmu@163.com</email>; Jie Bai, <email>baij@zucc.edu.cn</email>
</corresp>
<fn fn-type="equal" id="fn1">
<label>
<sup>&#x2020;</sup>
</label>
<p>These authors have contributed equally to this work and share first authorship</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>21</day>
<month>08</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>14</volume>
<elocation-id>1254827</elocation-id>
<history>
<date date-type="received">
<day>07</day>
<month>07</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>31</day>
<month>07</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Ju, Bai, Jiang, Che and Chen.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Ju, Bai, Jiang, Che and Chen</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>DNA N4-methylcytosine (4mC) is significantly involved in biological processes, such as DNA expression, repair, and replication. Therefore, accurate prediction methods are urgently needed. Deep learning methods have transformed applications that previously require sequencing expertise into engineering challenges that do not require expertise to solve. Here, we compare a variety of state-of-the-art deep learning models on six benchmark datasets to evaluate their performance in 4mC methylation site detection. We visualize the statistical analysis of the datasets and the performance of different deep-learning models. We conclude that deep learning can greatly expand the potential of methylation site prediction.</p>
</abstract>
<kwd-group>
<kwd>4mC DNA methylation</kwd>
<kwd>deep learning</kwd>
<kwd>classification</kwd>
<kwd>feature</kwd>
<kwd>visualization</kwd>
<kwd>interpretable ability</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Genomics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>The rapid progress in genome sequencing technologies has facilitated the investigation of the functional effects of DNA chemical modifications with unprecedented precision (<xref ref-type="bibr" rid="B28">Larranaga et al., 2006</xref>; <xref ref-type="bibr" rid="B24">Jiao and Du, 2016</xref>; <xref ref-type="bibr" rid="B15">Hamdy et al., 2022</xref>). DNA methylation, as a vital epigenetic modification, plays a crucial role in normal organism development and essential biological processes (<xref ref-type="bibr" rid="B32">Lv et al., 2021</xref>). In the genomes of both prokaryotic and eukaryotic organisms, the most prevalent kinds of DNA methylation include N6-methyladenine (6mA) (<xref ref-type="bibr" rid="B22">Huang et al., 2020</xref>; <xref ref-type="bibr" rid="B29">Li et al., 2021</xref>; <xref ref-type="bibr" rid="B6">Chen et al., 2022</xref>), C5-methylcytosine (5mC) (<xref ref-type="bibr" rid="B5">Cao et al., 2022</xref>), and N4-methylcytosine (4mC) (<xref ref-type="bibr" rid="B34">Moore et al., 2013</xref>; <xref ref-type="bibr" rid="B37">Plongthongkum et al., 2014</xref>; <xref ref-type="bibr" rid="B1">Ao et al., 2022a</xref>; <xref ref-type="bibr" rid="B55">Zulfiqar et al., 2022a</xref>; <xref ref-type="bibr" rid="B54">Zulfiqar et al., 2022b</xref>). The distribution of 4mC sites in the genome is highly significant as they play a crucial role in regulating gene expression and maintaining genome stability. Accurate identification and analysis of 4mC sites allow for a deeper understanding of the role of DNA methylation in gene regulation and disease mechanisms. This has important implications for the study of epigenetics, cancer etiology, biological evolution, and potential therapeutic strategies. Therefore, the development of efficient and accurate methods for detecting and identifying 4mC sites is of great importance for understanding biological processes and disease research (<xref ref-type="bibr" rid="B38">Razin and Cedar, 1991</xref>; <xref ref-type="bibr" rid="B26">Kulis and Esteller, 2010</xref>).</p>
<p>Several experimental techniques have been utilized to identify epigenetic 4mC sites. These methodologies include methylation-specific PCR, mass spectrometry, 4mC-Tet-assisted bisulfite-sequencing (4mCTABseq), whole-genome bisulfite sequencing, nanopore sequencing, and single-molecule real-time (SMRT) sequencing (<xref ref-type="bibr" rid="B3">Buryanov and Shevchuk, 2005</xref>; <xref ref-type="bibr" rid="B27">Laird, 2010</xref>; <xref ref-type="bibr" rid="B7">Chen et al., 2016</xref>; <xref ref-type="bibr" rid="B8">Chen et al., 2017</xref>; <xref ref-type="bibr" rid="B35">Ni et al., 2019</xref>). These experiment-based methods suffer from limitations such as low throughput, high cost, and restricted detection sensitivity. Nowadays, machine learning has been widely utilized and are successful technology in bioinformatics for extracting knowledge from huge data (<xref ref-type="bibr" rid="B28">Larranaga et al., 2006</xref>; <xref ref-type="bibr" rid="B12">Dwyer et al., 2018</xref>; <xref ref-type="bibr" rid="B19">Hu et al., 2020</xref>; <xref ref-type="bibr" rid="B18">Hu et al., 2021</xref>; <xref ref-type="bibr" rid="B20">Hu et al., 2022a</xref>; <xref ref-type="bibr" rid="B49">Zeng et al., 2022a</xref>; <xref ref-type="bibr" rid="B50">Zeng et al., 2022b</xref>; <xref ref-type="bibr" rid="B30">Li et al., 2023</xref>; <xref ref-type="bibr" rid="B47">Xu et al., 2023</xref>) and numerous computer techniques have been created to anticipate DNA 4mC sites. Both standard machine learning techniques and more current deep learning algorithms have been used to provide a strong result. In the field of 4mC site prediction, researchers have made significant strides by leveraging machine learning algorithms. These approaches utilize computational models to identify and classify 4mC sites within DNA sequences. Various machine learning techniques have been explored, including support vector machine (SVM) (<xref ref-type="bibr" rid="B8">Chen et al., 2017</xref>), random forest (RF), Markov model (MM), and ensemble methods. Additionally, advanced techniques such as extreme gradient boosting (XGBoost) and Laplacian Regularized Sparse Representation have also been employed in this context (<xref ref-type="bibr" rid="B8">Chen et al., 2017</xref>; <xref ref-type="bibr" rid="B33">Manavalan et al., 2019</xref>; <xref ref-type="bibr" rid="B17">He et al., 2019</xref>; <xref ref-type="bibr" rid="B16">Hasan et al., 2020</xref>; <xref ref-type="bibr" rid="B53">Zhao et al., 2020</xref>; <xref ref-type="bibr" rid="B2">Ao et al., 2022b</xref>; <xref ref-type="bibr" rid="B45">Xiao et al., 2022</xref>). However, traditional machine learning algorithms rely significantly on data representations known as features for appropriate performance, and it&#x27;s tough to figure out which features are best for a certain task. Deep learning overcomes the limitations of traditional methods by offering adaptivity, fault tolerance, nonlinearity, and improved input-to-output mapping. Deep learning methods, such as convolutional neural networks (CNNs) and recurrent neural networks (RNNs), have been developed for the detection of 4mC sites, leveraging their ability to capture sequence patterns and dependencies, thereby contributing to accurate identification of these sites and enhancing our understanding of DNA methylation in gene regulation and epigenetics (<xref ref-type="bibr" rid="B46">Xu et al., 2021</xref>; <xref ref-type="bibr" rid="B31">Liu et al., 2022</xref>). Yet there are still many deep learning methods that have not been applied, which have achieved great success in various application scenarios, including computer vision, speech recognition, biomarker identification (<xref ref-type="bibr" rid="B51">Zeng et al., 2020</xref>; <xref ref-type="bibr" rid="B4">Cai et al., 2021</xref>) and drug discovery (<xref ref-type="bibr" rid="B10">Chen et al., 2021</xref>; <xref ref-type="bibr" rid="B52">Zhang et al., 2021</xref>; <xref ref-type="bibr" rid="B21">Hu et al., 2022b</xref>; <xref ref-type="bibr" rid="B11">Dong et al., 2022</xref>; <xref ref-type="bibr" rid="B36">Pan et al., 2022</xref>; <xref ref-type="bibr" rid="B42">Song et al., 2022</xref>).</p>
<p>Choosing an appropriate deep learning model for bioinformatics problems poses a significant challenge for biologists. Understanding and comparing the performance of different models on specific datasets is of paramount importance for guiding practical applications. Therefore, our research focuses on evaluating the performance of multiple deep learning models on the 4mC datasets, aiming to assist biologists in making informed decisions when selecting suitable models.</p>
<p>We selected several common deep learning models, including RNN (Recurrent Nerual Network) (<xref ref-type="bibr" rid="B40">Rumelhart et al., 1986</xref>), long short-term memory (LSTM) (<xref ref-type="bibr" rid="B13">Graves, 2012</xref>), bi-directional long short-term memory (Bi-LSTM) (<xref ref-type="bibr" rid="B14">Graves and Schmidhuber, 2005</xref>; <xref ref-type="bibr" rid="B41">Sharma and Srivastava, 2021</xref>), text convolutional neural network (Text-CNN) (<xref ref-type="bibr" rid="B9">Kim, 2014</xref>), and bidirectional encoder representations from transformers (BERT) (<xref ref-type="bibr" rid="B23">Ji et al., 2021</xref>; <xref ref-type="bibr" rid="B43">Tran and Nguyen, 2022</xref>), and compared their performances on the 4mC datasets through optimization of model hyperparameters. Our research findings provide strong evidence-based support for biologists, aiding them in making informed choices when addressing bioinformatics problems on the 4mC datasets. By comparing the performance of multiple models, we can offer recommendations tailored to different problems and datasets, enabling biologists to better understand and leverage the advantages of deep learning models.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>Materials and methods</title>
<p>The implementation of our experiments relies on the DeepBIO (<xref ref-type="bibr" rid="B44">Wang et al., 2022</xref>) platform, which provides a wide selection of deep learning models and a visual comparison of multiple models. <xref ref-type="fig" rid="F1">Figure 1</xref> illustrates the overall framework of our works. We selected four deep learning models (RNN, LSTM, Bi-LSTM, Text-CNN) and pre-trained BERT models from the DeepBIO platform, and BERT is used as our main method to compare with other methods.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The workflow of the main modules. The benchmark datasets are initially divided into training and test sets. Subsequently, the divided dataset is fed into various deep learning models for prediction. The results of the predictions from different models are then evaluated. Finally, the data generated throughout these steps are visualized and analyzed for further insights.</p>
</caption>
<graphic xlink:href="fgene-14-1254827-g001.tif"/>
</fig>
</sec>
<sec id="s3">
<title>Datasets</title>
<p>The first step in creating a strong and trustworthy classification model is creating high-quality benchmark datasets. In this study, six benchmark datasets were utilized (<xref ref-type="bibr" rid="B48">Yu et al., 2021</xref>). <xref ref-type="table" rid="T1">Table 1</xref> provides a statistical summary of the datasets. The positive samples consisted of sequences that were 41 base pairs (bp) in length and contained a 4mC (4-methylcytosine) site located in the middle. These datasets have undergone rigorous preprocessing and quality control measures to ensure data accuracy and consistency (<xref ref-type="bibr" rid="B25">Jin et al., 2022</xref>). By training and evaluating the model on data from multiple species, including humans, animals, and plants, we ensure its broad applicability and provide valuable insights for biologists in selecting deep learning models.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Statistical summary of benchmark datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Species</th>
<th align="left">Positive sample</th>
<th align="left">Negative sample</th>
<th align="left">Total</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<italic>C. elegans</italic>
</td>
<td align="left">1,554</td>
<td align="left">1,554</td>
<td align="left">3,108</td>
</tr>
<tr>
<td align="left">
<italic>D. melanogaster</italic>
</td>
<td align="left">1,769</td>
<td align="left">1,769</td>
<td align="left">3,538</td>
</tr>
<tr>
<td align="left">
<italic>A. thaliana</italic>
</td>
<td align="left">1,978</td>
<td align="left">1,978</td>
<td align="left">3,956</td>
</tr>
<tr>
<td align="left">
<italic>E. coli</italic>
</td>
<td align="left">388</td>
<td align="left">388</td>
<td align="left">776</td>
</tr>
<tr>
<td align="left">
<italic>G. subterraneus</italic>
</td>
<td align="left">906</td>
<td align="left">906</td>
<td align="left">1,812</td>
</tr>
<tr>
<td align="left">
<italic>G. pickeringii</italic>
</td>
<td align="left">569</td>
<td align="left">569</td>
<td align="left">1,138</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4">
<title>Input feature matrix</title>
<p>Deep learning algorithms possess the capability to autonomously extract valuable features from data, distinguishing them from conventional machine learning methods that necessitate manual feature engineering. Nonetheless, when dealing with a string of nucleotide letters (A, C, G, and T), a conversion into a matrix format is required prior to feeding it into a neural network layer. Unlike prior methods that used several feature encodings schemes to represent the sequence as the input to train the model, this method uses a single feature encoding scheme. We took the dictionary encoding approaches for representing DNA sequences. To represent DNA sequences, we utilized a dictionary encoding method where each nucleotide (A, C, G, and T) is assigned a numeric value. Specifically, A is represented by 1, C by 2, G by 3, and T by 4. This encoding scheme allows us to convert the sequence into an N-dimensional vector, facilitating its input into the neural network for further analysis.</p>
</sec>
<sec id="s5">
<title>Model construction and parameters</title>
<p>We have selected deep-learning models that have received a lot of attention in recent years as follows: RNN, LSTM, Bi-LSTM, Text-CNN, and BERT. The first four deep learning models we used are the models provided by the DeepBIO platform with parameters already set and the BERT model we used is pre-trained DNABERT (<xref ref-type="bibr" rid="B23">Ji et al., 2021</xref>; <xref ref-type="bibr" rid="B39">Ren et al., 2022</xref>), which achieves the best performance on several DNA sequence classification tasks.</p>
<p>RNN is a type of neural network where the output of the previous neuron is fed back as input to the current neuron, creating temporal memory and enabling the processing of dynamic input sequences. RNNs find wide applications in various domains, including voice recognition, time series analysis, DNA sequences, and sequential data processing. One notable variant of RNNs that addresses the issue of capturing long-term dependencies is Long Short-Term Memory (LSTM). LSTM introduces a cell state that serves as a memory component, allowing the network to retain relevant information over extended periods. The forget gate in LSTM controls which information should be discarded and retained by using a sigmoid activation function. Additionally, Bidirectional LSTM (BiLSTM) processes input data in both forward and backward directions, effectively incorporating information from both past and future states. This bidirectional approach enables BiLSTM to capture intricate sequential relationships between words and sentences, making it particularly advantageous for Natural Language Processing (NLP) tasks that require contextual information from both preceding and succeeding elements in the input sequence. The RNN, LSTM, and Bi-LSTM architectures consist of stacked RNN cells, LSTM cells, and bidirectional LSTM cells, respectively. All these architectures share a similar structure, featuring 128 hidden neurons and a single layer for optimal performance. To prevent overfitting and promote generalization, a dropout rate of 0.2 was applied, and the output layer utilized sigmoid activation with a single neuron.</p>
<p>Text-CNN, a powerful deep learning approach for language classification tasks, such as sentiment analysis and question categorization, is a convolutional neural network tailored for text processing. The core structure comprises four layers: an embedding layer, a convolution layer, a pooling layer, and a fully connected layer. In our implementation, we set four convolutional kernel sizes (1, 2, 4, 8), and the number of convolutional kernels is uniformly set to 128. The embeddings undergo convolutional operations with a sliding kernel, producing convolutions that are subsequently downsampled through a Max Pooling layer to manage complexity and computational requirements. The scalar pooling outputs are then concatenated to form a vector representation of the input sequence. To mitigate overfitting, regularization methods, including a dropout layer with a rate of 0.2 and ReLU activation, are employed in the penultimate layer, preventing overfitting of the hidden layer.</p>
<p>BERT, an abbreviation for Bidirectional Encoder Representations from Transformers, originates from the Transformer architecture. In the Transformer model, every output element is intricately connected to every input element, with dynamically calculated weightings based on their connections. BERT is a pre-trained model that benefits from its ability to learn rich contextualized representations by considering the entire input sequence during training. Our study employs the pre-trained DNABert model, which has demonstrated superior performance in several DNA sequence classification tasks. We specifically fine-tune the 6mer-BERT variant on the 4mC methylation site benchmark dataset. Fine-tuning a pre-trained model on a task-specific dataset allows us to transfer the knowledge acquired during pre-training, enabling the model to achieve state-of-the-art performance in predicting DNA 4mC methylation sites. The incorporation of BERT&#x2019;s pre-trained knowledge provides significant advantages, as the model has already learned from vast amounts of data and captures intricate sequence patterns and dependencies. By leveraging pre-trained models like BERT, we achieve robust and accurate predictions, even in scenarios with limited training data.</p>
</sec>
<sec id="s6">
<title>Evaluation metrics</title>
<p>In order to compare with previous related work, we selected the commonly used evaluation indicators comprised of accuracy (ACC), sensitivity (SN), specificity (SP), Matthews&#x2019; coefficient correlation (MCC), and area under the receiver operating characteristic curve (AUC). These indicators are calculated by the following formula:<disp-formula id="equ1">
<mml:math id="m1">
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="{" close="" separators="|">
<mml:mrow>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mfrac>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>where <italic>TP</italic> represents true positives, which is the number of correctly predicted positive samples; <italic>TN</italic> represents true negatives, which is the number of correctly predicted negative samples; <italic>FP</italic> represents false positives, which is the number of negative samples wrongly predicted as positive; and <italic>FN</italic> represents false negatives, which is the number of positive samples wrongly predicted as negative.</p>
</sec>
<sec id="s7">
<title>Experimental setup</title>
<p>In our experimental design, we adopted the default settings of DeepBIO for other hyperparameters. For instance, when performing data set deduplication, we limited the duplication rate to 0.8 using the CDHIT algorithm integrated in the DeepBIO platform. Furthermore, we conducted a grid search on hyperparameters such as learning rate and batch size for each model. Grid search is a method for hyperparameter tuning, where different combinations of hyperparameters are tried to determine the optimal model configuration. Such experimental settings ensure that the models achieve their maximum potential performance while maintaining the reliability, fairness, and accuracy of the experiments.</p>
</sec>
<sec sec-type="results" id="s8">
<title>Result</title>
<p>In this section, we evaluate the performance of the different models and analyze the features extracted by the different models. In addition, we also compare the features learnt from deep-learning models with the traditional manual feature extraction methods applied in other studies to further demonstrate the superiority of deep learning in solving the 4mC methylation site detection problem. To ensure a balanced representation, the samples were randomly divided into training and test datasets for each species. The division was done in a ratio of 9:1, with 90% of the samples allocated to the training dataset and the remaining 10% assigned to the test dataset.</p>
<sec id="s8-1">
<title>Performance evaluation of multiple models</title>
<p>We conducted a comprehensive performance evaluation of four different models on six datasets to assess their performance in various data environments. The evaluation process involved the use of common binary classification metrics, such as accuracy (ACC), sensitivity, specificity, area under the curve (AUC), and Matthews correlation coefficient (MCC), to provide a comprehensive understanding of the models&#x2019; classification capabilities and highlight their performance differences. In addition to these metrics, we also employed receiver operating characteristic (ROC) curves and precision-recall curves (PRC) to further analyze the models&#x2019; performance.</p>
<p>Throughout our evaluation, we observed variations in performance across different datasets. While certain models demonstrated superior predictive performance on most datasets, their performance might vary on specific datasets. As shown in <xref ref-type="fig" rid="F2">Figures 2A, B</xref>, the RNN and TextCNN models exhibited promising performance on the <italic>G. pickeringii</italic> dataset, while DNABERT outperformed others on the <italic>G. subterraneus</italic> dataset. Overall, DNABERT consistently showcased superior performance across the evaluated datasets.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Performance evaluation of multiple models. <bold>(A)</bold> The basic statistics of ACC, Sensitivity, Specificity, AUC, and MCC in different models. <bold>(B)</bold> Performance comparison between DNABERT and other state-of-the-art methods on the benchmark datasets. <bold>(C)</bold> Density distribution of the prediction confidence by different deep learning models. <bold>(D)</bold> VENN and Upset plots show the overlap of different models&#x2019; predictions.</p>
</caption>
<graphic xlink:href="fgene-14-1254827-g002.tif"/>
</fig>
<p>Furthermore, let&#x2019;s consider the results obtained on the <italic>E. coli</italic> dataset. The density distribution of prediction confidence by different deep learning models (<xref ref-type="fig" rid="F2">Figure 2C</xref>) provides insights into the prediction preferences of each model. In the case of LSTM and Text-CNN, their density distribution shows a preference towards the center part of the X-axis, around 0.5. This indicates their poor binary classification ability and confusion in distinguishing between positive and negative instances. On the other hand, the density distribution for DNABERT is skewed towards the right side of the X-axis, indicating a better classification performance. This suggests that DNABERT exhibits a stronger ability to differentiate between positive and negative instances. And this is consistent with the conclusions drawn from the performance comparison in <xref ref-type="fig" rid="F2">Figure 2A</xref>.</p>
<p>We also performed statistics on the overlap of predictions between different models for the same dataset. Take the results obtained on the <italic>G. subterraneus</italic> dataset as an example, the distribution of sets classified as negative classes by different models in the test set is shown in <xref ref-type="fig" rid="F2">Figure 2D</xref>. In the VN diagram on the left, 41.4% of the test set is judged as negative by all models (negative classes account for 50% of the test set in total). The difference in quantity is shown more clearly in the right figure, and we can find that DNABERT may be one of the less effective models for classification under this dataset, as it predicts more negative cases individually. However, given that most of the model predictions converge on the same, we can conclude that most of the models are consistent in their classification results.</p>
</sec>
<sec id="s8-2">
<title>Deep learning model feature analysis</title>
<p>We conduct a comparative study on the features learned by deep learning from biological information. This includes comparisons between different deep learning models as well as comparisons between deep learning features and manually designed features. By conducting feature comparisons, we aim to further validate the superiority of deep learning methods and enhance the interpretability of deep learning models. We select ANF, binary, CKSNAP, and DNC approaches to extracting features and using SVM for unsupervised classification to compare with our deep learning models. <xref ref-type="fig" rid="F3">Figure 3A</xref> presents the ROC and PR curves for all models on the <italic>G. pickeringii</italic> dataset. We only display the two best-performing traditional manual feature methods for comparison. It is evident that most of the deep learning methods outperform the traditional approaches in terms of classification performance.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Deep learning model feature analysis. <bold>(A)</bold> Feature performance comparison between hand-crafted features and the features learned by deep learning models. <bold>(B)</bold> UMAP feature visualization and SHAP feature importance visualization.</p>
</caption>
<graphic xlink:href="fgene-14-1254827-g003.tif"/>
</fig>
<p>To visualize the results of deep learning features, we utilized UMAP (Uniform Manifold Approximation and Projection) and SHAP (Shapley Additive Explanations) plots for display (<xref ref-type="fig" rid="F3">Figure 3B</xref>). The UMAP plot reduces the dimensionality of the features while preserving the underlying data structure. It enables data clustering and categorization by mapping high-dimensional features into a lower-dimensional space, allowing for an analysis of feature similarity between positive and negative instances. The SHAP plot facilitates the understanding of feature importance and contribution to model predictions, providing interpretability to the model and enabling comparison of feature impacts. It helps to comprehend the significance of features in model predictions, enhancing interpretability and facilitating comparison among different features. In the feature visualization figure, each row corresponds to a specific feature, and the x-axis represents the snap value, providing a clearer understanding of the feature. The color gradient indicates the feature value, with higher values represented by redder colors and lower values represented by bluer colors. Each line represents a feature, and the horizontal position represents the SHAP value assigned to that feature in a particular sample. Each point represents a sample. The intensity of the color reflects the impact of the feature, with redder colors indicating a larger impact and bluer colors indicating a smaller impact. The scattered distribution of points indicates a greater influence of the feature.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s9">
<title>Conclusion</title>
<p>In this study, we use several currently popular deep learning models on the problem of 4mC methylation detection of DNA. We first present the current status of DNA 4mC methylation site detection, followed by the design of deep learning model workflows on six benchmark datasets, and finally, we evaluate the output of all models and conclude that deep learning has great potential for methylation detection, leading the way to future sequencing technologies along with newer bio-experimental methods. In fact, deep learning methods consistently outperformed traditional machine learning methods on all datasets. Furthermore, it was observed that pre-trained deep learning models with a higher number of parameters exhibited even better performance. We believe this may be because deep learning models with more parameters capture more features and analyze the features acquired by each model. By attempting to explain the model&#x2019;s internal workings and shed light on its internal representations, we aim to define its &#x201c;black box&#x201d; behavior.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s10">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/Supplementary Material, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s11">
<title>Author contributions</title>
<p>HJ: Data curation, Validation, Writing&#x2013;original draft, Writing&#x2013;review and editing. JB: Writing&#x2013;review and editing. JJ: Data curation, Writing&#x2013;review and editing. YC: Data curation, Writing&#x2013;review and editing. XC: Writing&#x2013;review and editing.</p>
</sec>
<sec id="s12">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This research work was supported by the Innovation Fund of the Ministry of Education&#x2019;s Engineering Research Center for the Integration and Application of Digital Learning Technologies, under project grant number 1221001.</p>
</sec>
<sec sec-type="COI-statement" id="s13">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s14">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ao</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Jiao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2022a</year>). <article-title>Biological sequence classification: A review on data and general methods</article-title>. <source>Research</source> <volume>2022</volume>, <fpage>0011</fpage>. <pub-id pub-id-type="doi">10.34133/research.0011</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ao</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2022b</year>). <article-title>NmRF: Identification of multispecies RNA 2&#x2019;-O-methylation modification sites from RNA sequences</article-title>. <source>Briefings Bioinforma.</source> <volume>23</volume>, <fpage>bbab480</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbab480</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Buryanov</surname>
<given-names>Y. I.</given-names>
</name>
<name>
<surname>Shevchuk</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>DNA methyltransferases and structural-functional specificity of eukaryotic DNA modification</article-title>. <source>Biochem. Mosc.</source> <volume>70</volume>, <fpage>730</fpage>&#x2013;<lpage>742</lpage>. <pub-id pub-id-type="doi">10.1007/s10541-005-0178-0</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cai</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Active semisupervised model for improving the identification of anticancer peptides</article-title>. <source>ACS Omega</source> <volume>6</volume>, <fpage>23998</fpage>&#x2013;<lpage>24008</lpage>. <pub-id pub-id-type="doi">10.1021/acsomega.1c03132</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cao</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kwok</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>webTWAS: a resource for disease candidate susceptibility genes identified by transcriptome-wide association study</article-title>. <source>Nucleic Acids Res.</source> <volume>50</volume>, <fpage>D1123</fpage>&#x2013;<lpage>D1130</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkab957</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>DeepM6ASeq-EL: Prediction of human N6-methyladenosine (m6A) sites with LSTM and ensemble learning</article-title>. <source>Front. Comput. Sci.</source> <volume>16</volume>, <fpage>162302</fpage>. <pub-id pub-id-type="doi">10.1007/s11704-020-0180-0</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>B. S.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Nucleic acid modifications in regulation of gene expression</article-title>. <source>Cell Chem. Biol.</source> <volume>23</volume>, <fpage>74</fpage>&#x2013;<lpage>85</lpage>. <pub-id pub-id-type="doi">10.1016/j.chembiol.2015.11.007</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>iDNA4mC: identifying DNA N4-methylcytosine sites based on nucleotide chemical properties</article-title>. <source>Bioinformatics</source> <volume>33</volume>, <fpage>3518</fpage>&#x2013;<lpage>3523</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btx479</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Muffin: Multi-scale feature fusion for drug&#x2013;drug interaction prediction</article-title>. <source>Bioinformatics</source> <volume>37</volume>, <fpage>2651</fpage>&#x2013;<lpage>2658</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btab169</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dong</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Deep learning in retrosynthesis planning: Datasets, models and tools</article-title>. <source>Briefings Bioinforma.</source> <volume>23</volume>, <fpage>bbab391</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbab391</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dwyer</surname>
<given-names>D. B.</given-names>
</name>
<name>
<surname>Falkai</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Koutsouleris</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Machine learning approaches for clinical psychology and psychiatry</article-title>. <source>Annu. Rev. Clin. Psychol.</source> <volume>14</volume>, <fpage>91</fpage>&#x2013;<lpage>118</lpage>. <pub-id pub-id-type="doi">10.1146/annurev-clinpsy-032816-045037</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Graves</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2012</year>). &#x201c;<article-title>Long Short-Term Memory</article-title>,&#x201d; in <source>Supervised Sequence Labelling with Recurrent Neural Networks. Studies in Computational Intelligence</source> (<publisher-loc>Berlin, Heidelberg</publisher-loc>: <publisher-name>Springer</publisher-name>) <volume>385</volume>, <fpage>37&#x2013;45</fpage>.</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Graves</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Schmidhuber</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Framewise phoneme classification with bidirectional LSTM and other neural network architectures</article-title>. <source>Neural Netw.</source> <volume>18</volume>, <fpage>602</fpage>&#x2013;<lpage>610</lpage>. <pub-id pub-id-type="doi">10.1016/j.neunet.2005.06.042</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hamdy</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Maghraby</surname>
<given-names>F. A.</given-names>
</name>
<name>
<surname>Omar</surname>
<given-names>Y. M. K.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>ConvChrome: Predicting gene expression based on histone modifications using deep learning techniques</article-title>. <source>Curr. Bioinforma.</source> <volume>17</volume>, <fpage>273</fpage>&#x2013;<lpage>283</lpage>. <pub-id pub-id-type="doi">10.2174/1574893616666211214110625</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hasan</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Manavalan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Shoombuatong</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Khatun</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Kurata</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>i4mC-Mouse: Improved identification of DNA N4-methylcytosine sites in the mouse genome using multiple encoding schemes</article-title>. <source>Comput. Struct. Biotechnol. J.</source> <volume>18</volume>, <fpage>906</fpage>&#x2013;<lpage>912</lpage>. <pub-id pub-id-type="doi">10.1016/j.csbj.2020.04.001</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>He</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>4mCPred: Machine learning methods for DNA N4-methylcytosine sites prediction</article-title>. <source>Bioinformatics</source> <volume>35</volume>, <fpage>593</fpage>&#x2013;<lpage>601</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty668</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>J. Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>rs1990622 variant associates with Alzheimer&#x27;s disease and regulates TMEM106B expression in human brain tissues</article-title>. <source>BMC Med.</source> <volume>19</volume>, <fpage>11</fpage>. <pub-id pub-id-type="doi">10.1186/s12916-020-01883-5</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>rs34331204 regulates TSPAN13 expression and contributes to Alzheimer&#x27;s disease with sex differences</article-title>. <source>Brain</source> <volume>143</volume>, <fpage>e95</fpage>. <pub-id pub-id-type="doi">10.1093/brain/awaa302</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2022a</year>). <article-title>Cognitive performance protects against Alzheimer&#x27;s disease independently of educational attainment and intelligence</article-title>. <source>Mol. Psychiatry</source> <volume>27</volume>, <fpage>4297</fpage>&#x2013;<lpage>4306</lpage>. <pub-id pub-id-type="doi">10.1038/s41380-022-01695-4</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2022b</year>). <article-title>Mendelian randomization highlights causal association between genetically increased C-reactive protein levels and reduced Alzheimer&#x27;s disease risk</article-title>. <source>Alzheimers Dement.</source> <volume>18</volume>, <fpage>2003</fpage>&#x2013;<lpage>2006</lpage>. <pub-id pub-id-type="doi">10.1002/alz.12687</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>Q. F.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>L. Y.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>6mA-RicePred: A method for identifying DNA N (6)-methyladenine sites in the rice genome based on feature fusion</article-title>. <source>Front. Plant Sci.</source> <volume>11</volume>, <fpage>4</fpage>. <pub-id pub-id-type="doi">10.3389/fpls.2020.00004</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ji</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Davuluri</surname>
<given-names>R. V.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Dnabert: Pre-trained bidirectional encoder representations from transformers model for DNA-language in genome</article-title>. <source>Bioinformatics</source> <volume>37</volume>, <fpage>2112</fpage>&#x2013;<lpage>2120</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btab083</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Performance measures in evaluating machine learning based bioinformatics predictors for classifications</article-title>. <source>Quant. Biol.</source> <volume>4</volume>, <fpage>320</fpage>&#x2013;<lpage>330</lpage>. <pub-id pub-id-type="doi">10.1007/s40484-016-0081-2</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Pang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>iDNA-ABF: multi-scale deep biological language learning model for the interpretable prediction of DNA methylations</article-title>. <source>Genome Biol.</source> <volume>23</volume>, <fpage>219</fpage>&#x2013;<lpage>223</lpage>. <pub-id pub-id-type="doi">10.1186/s13059-022-02780-1</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2014</year>). <source>Convolutional neural network for sentence classification[J]</source>. <publisher-loc>Waterloo, ON</publisher-loc>: <publisher-name>University of Waterloo</publisher-name>. <comment>arXiv preprint arXiv:1408.5882</comment>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kulis</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Esteller</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>DNA methylation and cancer</article-title>. <source>Adv. Genet.</source> <volume>70</volume>, <fpage>27</fpage>&#x2013;<lpage>56</lpage>. <pub-id pub-id-type="doi">10.1016/B978-0-12-380866-0.60002-2</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Laird</surname>
<given-names>P. W.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Principles and challenges of genome-wide DNA methylation analysis</article-title>. <source>Nat. Rev. Genet.</source> <volume>11</volume>, <fpage>191</fpage>&#x2013;<lpage>203</lpage>. <pub-id pub-id-type="doi">10.1038/nrg2732</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Larranaga</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Calvo</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Santana</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Bielza</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Galdiano</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Inza</surname>
<given-names>I.</given-names>
</name>
<etal/>
</person-group> (<year>2006</year>). <article-title>Machine learning in bioinformatics</article-title>. <source>Briefings Bioinforma.</source> <volume>7</volume>, <fpage>86</fpage>&#x2013;<lpage>112</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbk007</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>S. D.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>HSM6AP: A high-precision predictor for the Homo <italic>sapiens</italic> N6-methyladenosine (m&#x5e;6 A) based on multiple weights and feature stitching</article-title>. <source>Rna Biol.</source> <volume>18</volume>, <fpage>1882</fpage>&#x2013;<lpage>1892</lpage>. <pub-id pub-id-type="doi">10.1080/15476286.2021.1875180</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shao</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>T. Y.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>DSN-DDI: An accurate and generalized framework for drug&#x2013;drug interaction prediction by dual-view representation learning</article-title>. <source>Briefings Bioinforma.</source> <volume>24</volume>, <fpage>bbac597</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbac597</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ogata</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Akutsu</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>MSNet-4mC: Learning effective multi-scale representations for identifying DNA N4-methylcytosine sites</article-title>. <source>Bioinformatics</source> <volume>38</volume>, <fpage>5160</fpage>&#x2013;<lpage>5167</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btac671</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lv</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Dao</surname>
<given-names>F. Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Advances in mapping the epigenetic modifications of 5&#x2010;methylcytosine (5mC), N6&#x2010;methyladenine (6mA), and N4&#x2010;methylcytosine (4mC)</article-title>. <source>Biotechnol. Bioeng.</source> <volume>118</volume>, <fpage>4204</fpage>&#x2013;<lpage>4216</lpage>. <pub-id pub-id-type="doi">10.1002/bit.27911</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Manavalan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Basith</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shin</surname>
<given-names>T. H.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>D. Y.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>4mCpred-EL: An ensemble learning framework for identification of DNA N4-methylcytosine sites in the mouse genome</article-title>. <source>Cells</source> <volume>8</volume>, <fpage>1332</fpage>. <pub-id pub-id-type="doi">10.3390/cells8111332</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moore</surname>
<given-names>L. D.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>DNA methylation and its basic function</article-title>. <source>Neuropsychopharmacology</source> <volume>38</volume>, <fpage>23</fpage>&#x2013;<lpage>38</lpage>. <pub-id pub-id-type="doi">10.1038/npp.2012.112</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ni</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>D. P.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Miao</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>DeepSignal: Detecting DNA methylation state from nanopore sequencing reads using deep-learning</article-title>. <source>Bioinformatics</source> <volume>35</volume>, <fpage>4586</fpage>&#x2013;<lpage>4595</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz276</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>P. S.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Deep learning for drug repurposing: Methods, databases, and applications</article-title>. <source>Wiley Interdiscip. Rev. Comput. Mol. Sci.</source> <volume>12</volume>, <fpage>e1597</fpage>. <pub-id pub-id-type="doi">10.1002/wcms.1597</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Plongthongkum</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Diep</surname>
<given-names>D. H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Advances in the profiling of DNA modifications: Cytosine methylation and beyond</article-title>. <source>Nat. Rev. Genet.</source> <volume>15</volume>, <fpage>647</fpage>&#x2013;<lpage>661</lpage>. <pub-id pub-id-type="doi">10.1038/nrg3772</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Razin</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cedar</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>1991</year>). <article-title>DNA methylation and gene expression</article-title>. <source>Microbiol. Rev.</source> <volume>55</volume>, <fpage>451</fpage>&#x2013;<lpage>458</lpage>. <pub-id pub-id-type="doi">10.1128/mr.55.3.451-458.1991</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ren</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Multidrug representation learning based on pretraining model and molecular graph for drug interaction and combination prediction</article-title>. <source>Bioinformatics</source> <volume>38</volume>, <fpage>4387</fpage>&#x2013;<lpage>4394</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btac538</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rumelhart</surname>
<given-names>D. E.</given-names>
</name>
<name>
<surname>Hinton</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Williams</surname>
<given-names>R. J.</given-names>
</name>
</person-group> (<year>1986</year>). <article-title>Learning representations by back-propagating errors</article-title>. <source>nature</source> <volume>323</volume>, <fpage>533</fpage>&#x2013;<lpage>536</lpage>. <pub-id pub-id-type="doi">10.1038/323533a0</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sharma</surname>
<given-names>A. K.</given-names>
</name>
<name>
<surname>Srivastava</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Protein secondary structure prediction using character bi-gram embedding and Bi-LSTM</article-title>. <source>Curr. Bioinforma.</source> <volume>16</volume>, <fpage>333</fpage>&#x2013;<lpage>338</lpage>. <pub-id pub-id-type="doi">10.2174/1574893615999200601122840</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Niu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Learning spatial structures of proteins improves protein&#x2013;protein interaction prediction</article-title>. <source>Briefings Bioinforma.</source> <volume>23</volume>, <fpage>bbab558</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbab558</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tran</surname>
<given-names>H. V.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>Q. H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>iAnt: Combination of convolutional neural network and random forest models using PSSM and BERT features to identify antioxidant proteins</article-title>. <source>Curr. Bioinforma.</source> <volume>17</volume>, <fpage>184</fpage>&#x2013;<lpage>195</lpage>. <pub-id pub-id-type="doi">10.2174/1574893616666210820095144</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yin</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>DeepBIO is an automated and interpretable deep-learning platform for biological sequence prediction, functional annotation, and visualization analysis</article-title>, <comment>2022.2009.2029.509859. bioRxiv</comment>. <pub-id pub-id-type="doi">10.1101/2022.09.29.509859</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xiao</surname>
<given-names>Z. C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L. Z.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>Y. J.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>L. A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>iEnhancer-MRBF: Identifying enhancers and their strength with a multiple Laplacian-regularized radial basis function network</article-title>. <source>Methods</source> <volume>208</volume>, <fpage>1</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1016/j.ymeth.2022.10.001</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Deep4mC: Systematic assessment and computational prediction for DNA N4-methylcytosine sites by deep learning</article-title>. <source>Briefings Bioinforma.</source> <volume>22</volume>, <fpage>bbaa099</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbaa099</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Nussinov</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Graph embedding and Gaussian mixture variational autoencoder network for end-to-end analysis of single-cell RNA sequencing data</article-title>. <source>Cell Rep. Methods</source> <volume>3</volume>, <fpage>100382</fpage>. <pub-id pub-id-type="doi">10.1016/j.crmeth.2022.100382</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>iDNA-ABT: advanced deep learning model for detecting DNA methylation with adaptive features and transductive information maximization</article-title>. <source>Bioinformatics</source> <volume>37</volume>, <fpage>4603</fpage>&#x2013;<lpage>4610</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btab677</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zeng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>S. G.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lightstone</surname>
<given-names>F. C.</given-names>
</name>
<etal/>
</person-group> (<year>2022a</year>). <article-title>Deep generative molecular design reshapes drug discovery</article-title>. <source>Cell Rep. Med.</source> <volume>4</volume>, <fpage>100794</fpage>. <pub-id pub-id-type="doi">10.1016/j.xcrm.2022.100794</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zeng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xiang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Nussinov</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2022b</year>). <article-title>Accurate prediction of molecular properties and drug targets using a self-supervised image representation learning framework</article-title>. <source>Nat. Mach. Intell.</source> <volume>4</volume>, <fpage>1004</fpage>&#x2013;<lpage>1016</lpage>. <pub-id pub-id-type="doi">10.1038/s42256-022-00557-6</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zeng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Target identification among known drugs by deep learning from heterogeneous networks</article-title>. <source>Chem. Sci.</source> <volume>11</volume>, <fpage>1775</fpage>&#x2013;<lpage>1797</lpage>. <pub-id pub-id-type="doi">10.1039/c9sc04336e</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A novel antibacterial peptide recognition algorithm based on BERT</article-title>. <source>Briefings Bioinforma.</source> <volume>22</volume>, <fpage>bbab200</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbab200</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Accurate prediction of DNA N4-methylcytosine sites via boost-learning various types of sequence features</article-title>. <source>BMC genomics</source> <volume>21</volume>, <fpage>627</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-020-07033-8</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zulfiqar</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Q. L.</given-names>
</name>
<name>
<surname>Lv</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Z. J.</given-names>
</name>
<name>
<surname>Dao</surname>
<given-names>F. Y.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2022b</year>). <article-title>Deep-4mCGP: A deep learning approach to predict 4mC sites in geobacter pickeringii by using correlation-based feature selection technique</article-title>. <source>Int. J. Mol. Sci.</source> <volume>23</volume>, <fpage>1251</fpage>. <pub-id pub-id-type="doi">10.3390/ijms23031251</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zulfiqar</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Z. J.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Q. L.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Lv</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Dao</surname>
<given-names>F. Y.</given-names>
</name>
<etal/>
</person-group> (<year>2022a</year>). <article-title>Deep-4mCW2V: A sequence-based predictor to identify N4-methylcytosine sites in <italic>Escherichia coli</italic>
</article-title>. <source>Methods</source> <volume>203</volume>, <fpage>558</fpage>&#x2013;<lpage>563</lpage>. <pub-id pub-id-type="doi">10.1016/j.ymeth.2021.07.011</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>