<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="methods-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2022.845835</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>i6mA-Vote: Cross-Species Identification of DNA N6-Methyladenine Sites in Plant Genomes Based on Ensemble Learning With Voting</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Teng</surname> <given-names>Zhixia</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1010496/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhao</surname> <given-names>Zhengnan</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1571175/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Li</surname> <given-names>Yanjuan</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1465511/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Tian</surname> <given-names>Zhen</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/654726/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Guo</surname> <given-names>Maozu</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1656076/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Lu</surname> <given-names>Qianzi</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1642112/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Wang</surname> <given-names>Guohua</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1549213/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>College of Information and Computer Engineering, Northeast Forestry University</institution>, <addr-line>Harbin</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>College of Electrical and Information Engineering, Quzhou University</institution>, <addr-line>Quzhou</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>College of Information Engineering, Zhengzhou University</institution>, <addr-line>Zhengzhou</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>College of Electrical and Information Engineering, Beijing University of Civil Engineering and Architecture</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff5"><sup>5</sup><institution>College of Bioinformatics Science and Technology, Harbin Medical University</institution>, <addr-line>Harbin</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Weihua Pan, Agricultural Genomics Institute at Shenzhen, Chinese Academy of Agricultural Sciences (CAAS), China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Guohua Huang, Shaoyang University, China; Yungang Xu, Xi&#x2019;an Jiaotong University, China; Hui Ding, University of Electronic Science and Technology of China, China</p></fn>
<corresp id="c001">&#x002A;Correspondence: Qianzi Lu, <email>245360359@qq.com</email></corresp>
<corresp id="c002">Guohua Wang, <email>ghwang@nefu.edu.cn</email></corresp>
<fn fn-type="other" id="fn004"><p>This article was submitted to Plant Bioinformatics, a section of the journal Frontiers in Plant Science</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>14</day>
<month>02</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>845835</elocation-id>
<history>
<date date-type="received">
<day>30</day>
<month>12</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>24</day>
<month>01</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2022 Teng, Zhao, Li, Tian, Guo, Lu and Wang.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Teng, Zhao, Li, Tian, Guo, Lu and Wang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>DNA N6-Methyladenine (6mA) is a common epigenetic modification, which plays some significant roles in the growth and development of plants. It is crucial to identify 6mA sites for elucidating the functions of 6mA. In this article, a novel model named i6mA-vote is developed to predict 6mA sites of plants. Firstly, DNA sequences were coded into six feature vectors with diverse strategies based on density, physicochemical properties, and position of nucleotides, respectively. To find the best coding strategy, the feature vectors were compared on several machine learning classifiers. The results suggested that the position of nucleotides has a significant positive effect on 6mA sites identification. Thus, the dinucleotide one-hot strategy which can describe position characteristics of nucleotides well was employed to extract DNA features in our method. Secondly, DNA sequences of Rosaceae were divided into a training dataset and a test dataset randomly. Finally, i6mA-vote was constructed by combining five different base-classifiers under a majority voting strategy and trained on the Rosaceae training dataset. The i6mA-vote was evaluated on the task of predicting 6mA sites from the genome of the Rosaceae, Rice, and Arabidopsis separately. In Rosaceae, the performances of i6mA-vote were 0.955 on accuracy (ACC), 0.909 on Matthew correlation coefficients (MCC), 0.955 on sensitivity (SN), and 0.954 on specificity (SP). Those indicators, in the order of ACC, MCC, SN, SP, were 0.882, 0.774, 0.961, and 0.803 on Rice while they were 0.798, 0.617, 0.666, and 0.929 on Arabidopsis. According to the indicators, our method was effectiveness and better than other concerned methods. The results also illustrated that i6mA-vote does not only well in 6mA sites prediction of intraspecies but also interspecies plants. Moreover, it can be seen that the specificity is distinctly lower than the sensitivity in Rice while it is just the opposite in Arabidopsis. It may be resulted from sequence similarity among Rosaceae, Rice and Arabidopsis.</p>
</abstract>
<kwd-group>
<kwd>N6-methyladenine</kwd>
<kwd>plant genomes</kwd>
<kwd>cross-species</kwd>
<kwd>feature encoding</kwd>
<kwd>ensemble learning</kwd>
</kwd-group>
<counts>
<fig-count count="5"/>
<table-count count="3"/>
<equation-count count="6"/>
<ref-count count="45"/>
<page-count count="11"/>
<word-count count="7290"/>
</counts>
</article-meta>
</front>
<body>
<sec id="S1" sec-type="intro">
<title>Introduction</title>
<p>DNA N6-methyladenine (6mA) is a methyl modification at the sixth position of the adenine ring, which was discovered by <xref ref-type="bibr" rid="B36">Vanyushin et al. (1968)</xref>. 6mA is widely found in prokaryotes and eukaryotes (<xref ref-type="bibr" rid="B10">Fu et al., 2015</xref>; <xref ref-type="bibr" rid="B11">Greer et al., 2015</xref>; <xref ref-type="bibr" rid="B45">Zhang et al., 2015</xref>). It is reported that 6mA plays vital roles in DNA replication, repairing nucleotide dislocations, and preventing the invasion of foreign DNA (<xref ref-type="bibr" rid="B40">Wion and Casades&#x00FA;s, 2006</xref>). Although 6mA in animal genomes studies have been well studied, those of plants genomes have still known a little, which hampered to explore their functions. To better understand the molecular mechanism of 6mA in plants, it is the first step to determine the 6mA sites accurately.</p>
<p>To detect 6mA sites, several biochemical methods were developed, such as single-molecule real-time sequencing technology (SMRT-seq) (<xref ref-type="bibr" rid="B7">Davis et al., 2013</xref>) and restriction endonuclease-based 6mA sequencing (6MA-RE-seq) (<xref ref-type="bibr" rid="B10">Fu et al., 2015</xref>). In SMRT-seq, single-nucleotide molecules labeled by different fluorophores were paired with bases of a DNA sequence, and the fluorescence signals were recorded during the process of pairing. The fragment of DNA sequence may be methylated if it showed the continuous same signal during the process of pairing. 6mA-RE-seq explored restriction enzymes to fragment genomic DNA at &#x201C;CATG&#x201D; and &#x201C;GATC&#x201D; motifs that did not contain 6mA and then retained these motifs containing 6mA. In this way, after end-repair and other operations, the methylated motifs would be enriched in the internal positions of DNA fragments. However, these methods are hard to detect 6mA sites from high-throughput sequences because they are time-consuming and expensive.</p>
<p>Therefore, some machine learning models have been developed to identify 6mA sites in recent years because they are efficient and cheap. At first, iDNA6mA-PseKNC (<xref ref-type="bibr" rid="B9">Feng et al., 2019</xref>) was proposed to detect 6mA sites in the mouse genome. In this model, DNA sequences were represented by pseudo-k-tuple nucleotide composition incorporating the physicochemical properties of nucleotides, and then the sequences were classified by a support vector machine (SVM). Subsequently, i6mA-Pred (<xref ref-type="bibr" rid="B5">Chen et al., 2019</xref>) trained a novel SVM model to identify 6mA sites in the rice genome based on the chemical properties of nucleotide such as the loop structure, the hydrogen bond, and the amino groups, and the nucleotide frequency of DNA sequences. To avoid overfitting, i6mA-Pred used the maximum correlation maximum distance approach to select the most representative features. Afterward, iN6-methylate (<xref ref-type="bibr" rid="B20">Le, 2019</xref>), another novel SVM model, used FastText to generate feature vectors for DNA sequences based on the assumption that a DNA sequence is a sentence and a nucleotide is a word. Unlike previous models, MM-6mAPred (<xref ref-type="bibr" rid="B28">Pian et al., 2019</xref>) constructed Markov chains based on DNA sequences with 6mA sites (positive samples) and DNA sequences without 6mA sites (negative samples) in the training dataset. Based on the Markov chains, the positive and negative probabilities of a DNA sequence were calculated separately. It is considered that a sequence contained 6mA site if the ratio of positive probability against negative probability is greater than 1.</p>
<p>To improve the performance of above methods, ensemble learning has been increasingly applied to 6mA sites prediction. In the beginning, iDNA6mA-Rice (<xref ref-type="bibr" rid="B24">Lv et al., 2019</xref>), a rice 6mA site classification model based on random forest, encoded DNA sequences via three feature descriptors, namely the k-nucleotide frequency, the mono-nucleotide binary coding, and the natural vector containing the frequency, average position, and second-order central moment of mono-nucleotides. Soon afterward, on the basis of bagging with CART, i6mA-DNCP (<xref ref-type="bibr" rid="B19">Kong and Zhang, 2019</xref>) represented rice DNA sequences by two novel feature descriptors: dinucleotide frequency and dinucleotide physicochemical properties. In addition, i6mA-DNCP employed heuristic ideas to select the most representative features. Several months later, i6mA-Fuse (<xref ref-type="bibr" rid="B13">Hasan et al., 2020</xref>) was proposed to classify Rosaceae DNA sequences with random forest and linear regression. Subsequently, a random forest-based multi-species 6mA site prediction model 6mA-Finder (<xref ref-type="bibr" rid="B41">Xu et al., 2020</xref>) was developed, which contained three modules for mouse, rice, and a general species admixed by mouse and rice DNA sequences, respectively. i6mA-stack (<xref ref-type="bibr" rid="B17">Khanal et al., 2021</xref>) developed a two-level stacked ensemble classifier based on linear regression, random forest, support vector machine, and gaussian naive bayes to recognize Rosaceae 6mA sites.</p>
<p>With the development of deep learning, some neural network models were also developed for identifying 6mA sites. For example, iDNA6mA (<xref ref-type="bibr" rid="B33">Tahir et al., 2019</xref>) is composed of four layers: two convolution layers which extract features of DNA sequences, a dropout layer which is used to avoid overfitting, and a full-connection layer which performs classification tasks. Subsequently, SNNRice6mA (<xref ref-type="bibr" rid="B43">Yu and Dai, 2019</xref>) was improved iDNA6mA by adding a normalization layer and a pooling layer between the convolution layer and the dropout layer, which aimed to reduce redundant features of DNA sequences according to the correlation of the features. i6mA-DNC (<xref ref-type="bibr" rid="B26">Park et al., 2020</xref>) is similar with the above two models except it extracted features from nucleotide pairs of DNA sequences rather than from single nucleotides. It is worth noting that the three neural network models mentioned above were all developed for predicting 6mA sites in the rice genome.</p>
<p>Because the previously mentioned models are species-specific, Meta-i6mA (<xref ref-type="bibr" rid="B12">Hasan et al., 2021</xref>) was proposed for 6mA site prediction from multiple plants. Although Meta-i6mA has achieved encouraging results in intraspecies, it still has room for improvement in interspecific. To solve this problem, a novel classification model i6mA-vote was developed based on an ensemble learning strategy. In this model, DNA sequences were encoded by nucleotide position-based feature descriptors, and then these sequences were classified by an ensemble classifier integrating random forest, linear discriminant analysis, multi-layer perceptron, stochastic gradient descent, and extreme gradient boosting. The details of i6mA-vote will be introduced in the following sections.</p>
</sec>
<sec id="S2" sec-type="materials|methods">
<title>Materials and Methods</title>
<sec id="S2.SS1">
<title>Framework of i6mA-Vote</title>
<p>In our study, as shown in <xref ref-type="fig" rid="F1">Figure 1</xref>, i6mA-vote was constructed by four steps. Firstly, positive samples of Rosaceae, Rice, and Arabidopsis were derived from MDR (<xref ref-type="bibr" rid="B23">Liu et al., 2019</xref>), GEO (<xref ref-type="bibr" rid="B8">Edgar et al., 2002</xref>), and MethSMRT (<xref ref-type="bibr" rid="B42">Ye et al., 2017</xref>) databases, and negative samples of these plants were downloaded from NCBI. For each plant, the positive and negative samples were filtered by CD-HIT (<xref ref-type="bibr" rid="B21">Li and Godzik, 2006</xref>) to reduce high similar samples. Then all samples were divided into three datasets according to organisms for the subsequent experiments. The Rosaceae dataset was split into a training dataset and a test dataset, and datasets for the remaining two species were used as cross-species evaluation datasets. Secondly, to transform DNA sequences into feature vectors, one-hot encoding method was performed on dinucleotides (e.g., AA, AG, &#x2026;) of DNA sequences. Because the known nucleotides can be represented by four symbols (A, G, C, T) and other unknown nucleotides can be denoted by symbol N, in this way, there were twenty-five dinucleotide combinations. Thirdly, an ensemble learning model, named i6mA-vote, was built by integrating random forest (RF), multi-layer perceptron (MLP), stochastic gradient descent (SGD), linear discriminant analysis (LDA), extreme gradient boosting (XGB), based on majority voting strategy. Then all samples were represented by feature vectors and the ensemble learning model was trained on the samples. Finally, to evaluate the performance of the model, i6mA-vote was used to perform simulation a task on test datasets, and its superiority was demonstrated by accuracy, Matthew correlation coefficient, sensitivity, and specificity. In the following sections, the detail process of constructing the i6mA-vote model will be illustrated step by step.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption><p>Frame diagram of i6mA-vote. (DS1, DS2, DS3, and DS4 refer to Rosaceae training dataset, Rosaceae test dataset, Rice test dataset, and Arabidopsis test dataset; In the DNA sequences, the letter &#x201C;A&#x201D; marked in red refers to the possible 6mA site, and the letter &#x201C;N&#x201D; indicates the unidentified nucleotide; In the feature vectors, &#x201C;p&#x201D; and &#x201C;n&#x201D; are short for the positive sample and the negative sample).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-845835-g001.tif"/>
</fig>
</sec>
<sec id="S2.SS2">
<title>Datasets</title>
<p>The quality of the dataset affects the performance of the classification model. In this study, four high-quality datasets that have been applied in the 6mA prediction domain were selected.</p>
<p>The Rosaceae dataset was collected, collated, and constructed by Hasan&#x2019;s team (<xref ref-type="bibr" rid="B12">Hasan et al., 2021</xref>). The part containing 6mA were derived from the MDR database (<xref ref-type="bibr" rid="B23">Liu et al., 2019</xref>). After removing similar sequences and excluding 90% sequence identity, 36,537 positive samples were obtained. The other part, including the same number of negative ones, was taken from NCBI, and it was generated by chromosomes with no 6mA detected. Finally, 80% of this dataset was randomly selected as the training dataset (DS1), and the remaining 20% was regarded as the test dataset (DS2).</p>
<p>The Rice dataset (DS3) was created by Lin&#x2019;s group (<xref ref-type="bibr" rid="B24">Lv et al., 2019</xref>). The positive portion and the negative one were obtained from the GEO database (<xref ref-type="bibr" rid="B8">Edgar et al., 2002</xref>) and NCBI. And they both included 154000 samples.</p>
<p>The Arabidopsis dataset (DS4) was also constructed by Hasan&#x2019;s team (<xref ref-type="bibr" rid="B12">Hasan et al., 2021</xref>). It extracted 31,873 6mA sites from the MethSMRT database (<xref ref-type="bibr" rid="B42">Ye et al., 2017</xref>) and replenished the same number of negative samples from NCBI using the same way as for the Rosaceae dataset.</p>
<p>Among them, DS1 was used for training the model, DS2, DS3, and DS4 were used to evaluate the generalization performance and cross-species prediction ability of the model.</p>
<p>All the above four datasets were downloaded from the online server of model Meta-i6mA (<xref ref-type="bibr" rid="B12">Hasan et al., 2021</xref>)<sup><xref ref-type="fn" rid="footnote1">1</xref></sup>. In addition, these datasets were also processed as follows: (1) Sequences longer than 41bp were removed. (2) If a sequence was repeated multiple times, it would be deleted, leaving only one copy. (3) If a sequence was present in both positive and negative samples, it would be removed from both parts. Finally, the number of samples included in each dataset is shown in <xref ref-type="table" rid="T1">Table 1</xref>. Their sequences all consisted of 41 nucleotides with an &#x201C;A&#x201D; in the middle.</p>
<table-wrap position="float" id="T1">
<label>TABLE 1</label>
<caption><p>Number of samples in each dataset.</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<thead>
<tr>
<td valign="top" align="left">Datasets</td>
<td valign="top" align="center">Number of positive samples</td>
<td valign="top" align="center">Number of negative samples</td>
<td valign="top" align="center">Total number</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">DS1</td>
<td valign="top" align="center">29237</td>
<td valign="top" align="center">29433</td>
<td valign="top" align="center">58670</td>
</tr>
<tr>
<td valign="top" align="left">DS2</td>
<td valign="top" align="center">7298</td>
<td valign="top" align="center">7300</td>
<td valign="top" align="center">14598</td>
</tr>
<tr>
<td valign="top" align="left">DS3</td>
<td valign="top" align="center">153635</td>
<td valign="top" align="center">153629</td>
<td valign="top" align="center">307264</td>
</tr>
<tr>
<td valign="top" align="left">DS4</td>
<td valign="top" align="center">31414</td>
<td valign="top" align="center">31843</td>
<td valign="top" align="center">63257</td>
</tr>
</tbody>
</table></table-wrap>
</sec>
<sec id="S2.SS3">
<title>Feature Extraction</title>
<p>To convert DNA sequences into feature vectors, One-hot encoding method for dinucleotides was employed in our model. This strategy and other concerned strategies will be described in detail below.</p>
<sec id="S2.SS3.SSS1">
<title>Our Encoding Strategy</title>
<p>One-hot encoding method for dinucleotides (One-hot2) is based on the one-hot encoding method in natural language processing. The one-hot encoding method compiles a dictionary using the words in the sentences and then encodes each word into a 0-1 vector through this dictionary. The length of the vector is equal to that of the dictionary, and each bit in the vector corresponds to a word in the dictionary. When encoding a word, its corresponding bit is set to 1 in the vector, and the other bits are kept at 0. Similarly, One-hot2 treats DNA sequences as sentences and dinucleotides as words.</p>
<p>A DNA sequence is usually composed of four standard nucleotide symbols: A, C, G, and T. However, sometimes the DNA sequence also include non-standard nucleotide symbol N, which means that the nucleotide was not identified. Accordingly, a DNA sequence may consist of 5 symbols, and it contains 25 possible symbol combinations of dinucleotides like AA, AC, AN. In our method, the one-hot2 encoded each dinucleotide into a 25-dimensional 0-1 vector. The vector of each dinucleotide is shown in Formula (1).</p>
<disp-formula id="S2.E1">
<label>(1)</label>
<mml:math id="M1">
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mtable displaystyle="true" rowspacing="0pt">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mrow>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mpadded width="+3.3pt">
<mml:mi>A</mml:mi>
</mml:mpadded>
</mml:mrow>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mrow>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mpadded width="+3.3pt">
<mml:mi>C</mml:mi>
</mml:mpadded>
</mml:mrow>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mrow>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mpadded width="+3.3pt">
<mml:mi>G</mml:mi>
</mml:mpadded>
</mml:mrow>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mi mathvariant="normal">&#x22EE;</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mpadded width="+3.3pt">
<mml:mi>T</mml:mi>
</mml:mpadded>
</mml:mrow>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mpadded width="+3.3pt">
<mml:mi>N</mml:mi>
</mml:mpadded>
</mml:mrow>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mi/>
</mml:mrow>
</mml:math>
</disp-formula>
<p>To show how One-hot2 encodes DNA sequences, an example is given below. DNA sequence <italic>D</italic> = <italic>ACGTNA</italic> can be split into five dinucleotides (AC, CG, GT, TN, NA), and then they are replaced with their corresponding one-hot codes. In this way, a vector with the dimension of 125 is generated.</p>
<p>Because the length of the DNA sequences in our datasets are 41bp, the sequences can be spliced into 40 dinucleotides and thus the vectors of these dinucleotides were concatenated into a 1000-dimensional feature vector to describe their primary sequence.</p>
<p>There are three reasons why One-hot2 was chosen: (1) It can solve the problem that classifiers are not good at handling continuous data. In addition, it generates sparse vectors, allowing many machine learning problems to be linearly separated and models more efficient to be stored. (2) It considers the relationship between adjacent nucleotides as it is encoded in dinucleotide. (3) Some studies (<xref ref-type="bibr" rid="B5">Chen et al., 2019</xref>; <xref ref-type="bibr" rid="B9">Feng et al., 2019</xref>) found position-specific features can better represent sequences containing 6mA sites, and One-hot2 happens to be this kind of method.</p>
</sec>
<sec id="S2.SS3.SSS2">
<title>The Concerned Encoding Strategies</title>
<sec id="S2.SS3.SSS2.Px1">
<title>Density-Based Approach</title>
<p>Accumulated Mono-Nucleotide Frequency (AMNF) represent the frequency of single nucleotides in the subsequence which ranges from the first nucleotide to the current nucleotide of the original sequence. Similarly, Accumulated Di-Nucleotide Frequency (ADNF) (<xref ref-type="bibr" rid="B6">Chen et al., 2017</xref>) denotes the nucleotide pairs which appears before current nucleotide. For example, DNA sequence <italic>D</italic> = <italic>ACGTNA</italic> can be encoded as (1, 0.5, 0.33, 0.25, 0.2, 0.33) and (1, 0.5, 0.33, 0.25, 0.2) by AMNF and ADNF, respectively.</p>
</sec>
<sec id="S2.SS3.SSS2.Px2">
<title>Physicochemical-Properties-Based Approach</title>
<p>Dinucleotide Physical-Chemical Properties (DPCP) and Trinucleotide Physical-Chemical Properties (TPCP) (<xref ref-type="bibr" rid="B25">Manavalan et al., 2019</xref>; <xref ref-type="bibr" rid="B39">Wei et al., 2019</xref>) replace the DNA sequences with the vectors calculated by Equation (2) using the physicochemical-properties in <xref ref-type="supplementary-material" rid="TS1">Supplementary Tables 1</xref>,<xref ref-type="supplementary-material" rid="TS1">2</xref>. In <xref ref-type="supplementary-material" rid="TS1">Supplementary Table 1</xref>, the columns represent 15 physicochemical properties, and the rows represent 25 dinucleotides. In <xref ref-type="supplementary-material" rid="TS1">Supplementary Table 2</xref>, the columns represent 11 physicochemical properties, and the rows represent 125 trinucleotides.</p>
<disp-formula id="S2.E2">
<label>(2)</label>
<mml:math id="M2">
<mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mpadded width="+3.3pt">
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mpadded>
</mml:mrow>
<mml:mo rspace="5.8pt">=</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mpadded width="+3.3pt">
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mpadded>
<mml:mo rspace="5.8pt">&#x00D7;</mml:mo>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>x</italic> = <italic>D</italic> refers to Dinucleotide and <italic>x</italic> = <italic>T</italic> denotes Trinucleotide. When <italic>x</italic> = <italic>D</italic>, the values of <italic>i</italic> range from 1 to 25, the values of <italic>j</italic> range from 1 to 15, <italic>DPCP</italic><sub><italic>i</italic></sub>is the DPCP value of the <italic>i</italic>th dinucleotides, <italic>N_i</italic> is the count of the <italic>i</italic>th dinucleotides in the DNA sequence, and <italic>DPC</italic><sub><italic>ij</italic></sub> is the <italic>j</italic>th properties of the <italic>i</italic>th dinucleotides; When <italic>x</italic> = <italic>T</italic>, the values of <italic>i</italic> range from 1 to 125, the values of <italic>j</italic> range from 1 to 11, <italic>TPCP</italic><sub><italic>i</italic></sub> is the TPCP value of the <italic>i</italic>th trinucleotides, <italic>N_i</italic> is the count of the <italic>i</italic>th trinucleotides in the DNA sequence, and <italic>TPC</italic><sub><italic>ij</italic></sub> is the <italic>j</italic>th properties of the <italic>i</italic>th trinucleotides.</p>
</sec>
<sec id="S2.SS3.SSS2.Px3">
<title>Position-Based Approach</title>
<p>One-hot encoding method for mononucleotide (One-hot1) is similar to One-hot2, except that its encoding unit is the mononucleotide. It converts a mononucleotide into a one-hot code with a length of five, corresponding to five mononucleotides (A, C, G, T, and N). For instance, the encoded vector of DNA sequence<italic>D</italic> = <italic>ACGTNA</italic> is (1,0,0,0,0| 0,1,0,0,0| 0,0,1,0,0| 0,0,0,1,0| 0,0,0,0,1| 1,0,0,0,0).</p>
</sec>
</sec>
</sec>
<sec id="S2.SS4">
<title>Classifier</title>
<p>To train a classification model with stable and good performance, five machine learning algorithms was utilized to construct five base-classifiers. Subsequently, majority voting was adopted to integrate these five base-classifiers. Its detailed procedure is illustrated in the following steps.</p>
<p>(1) The processed training dataset was inputted into five machine learning algorithms, and five base-classifiers were generated. These five algorithms were random forest (RF), multi-layer perceptron (MLP), stochastic gradient descent (SGD), linear discriminant analysis (LDA), extreme gradient boosting (XGB). Among them, RF refers to one type of classifier that utilizes multiple decision trees to train and predict samples. MLP, as a simple neural network, contains three fully connected layers, the input layer, the hidden layer, and the output layer. SGD is a kind of support vector machine model. LDA is a classifier generated according to Bayes&#x2019; rule. XGB is also based on trees, but unlike random forests, its trees are regressive, and it also optimizes the algorithm itself, the efficiency and robustness of the algorithm.</p>
<p>(2) The five base classifiers were combined into one ensemble classifier by majority voting. That is, when three or more base classifiers judge a sequence to be a positive (or negative) sample, then their combination also treats this sequence as a positive (or negative) sample.</p>
<p>It should be noted that the hyperparameters of the base-learners were optimized by grid search strategy. After manually specifying variation ranges of hyperparameters, this strategy adopted an exhaustive method-like approach to find the best-performing combination from these hyperparameters. In addition, all classifier algorithms in this paper were implemented by sklearn (<xref ref-type="bibr" rid="B15">Hinton, 1989</xref>; <xref ref-type="bibr" rid="B1">Belhumeur et al., 1997</xref>; <xref ref-type="bibr" rid="B29">Platt, 2000</xref>; <xref ref-type="bibr" rid="B3">Breiman, 2001</xref>; <xref ref-type="bibr" rid="B2">Bengio and Glorot, 2010</xref>; <xref ref-type="bibr" rid="B27">Pedregosa et al., 2011</xref>; <xref ref-type="bibr" rid="B18">Kingma and Ba, 2014</xref>; <xref ref-type="bibr" rid="B14">He et al., 2015</xref>; <xref ref-type="bibr" rid="B4">Chen and Guestrin, 2016</xref>).</p>
</sec>
<sec id="S2.SS5">
<title>Performance Evaluation</title>
<p>Our model was validated according to accuracy (ACC), Matthew correlation coefficient (MCC), Sensitivity (SN), Specificity (SP) which had been widely adopted in the field of bioinformatics (<xref ref-type="bibr" rid="B16">Huang and Gong, 2020</xref>; <xref ref-type="bibr" rid="B22">Liu et al., 2020</xref>; <xref ref-type="bibr" rid="B32">Smolarczyk et al., 2020</xref>; <xref ref-type="bibr" rid="B37">Wang H. et al., 2020</xref>; <xref ref-type="bibr" rid="B38">Wang J. et al., 2020</xref>; <xref ref-type="bibr" rid="B31">Shao and Liu, 2021</xref>; <xref ref-type="bibr" rid="B44">Zhang et al., 2021</xref>). These metrics can be calculated by equations (3) &#x223C; (6).</p>
<disp-formula id="S2.E3">
<label>(3)</label>
<mml:math id="M3">
<mml:mrow>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mpadded width="+3.3pt">
<mml:mi>C</mml:mi>
</mml:mpadded>
</mml:mrow>
<mml:mo rspace="10.8pt">=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="S2.Ex1">
<label>(4)</label>
<mml:math id="M4">
<mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mpadded width="+3.3pt">
<mml:mi>C</mml:mi>
</mml:mpadded>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi/>
</mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mrow>
<mml:mpadded width="+3.3pt">
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mpadded>
<mml:mo rspace="5.8pt">&#x00D7;</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>-</mml:mo>
<mml:mrow>
<mml:mpadded width="+3.3pt">
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mpadded>
<mml:mo rspace="5.8pt">&#x00D7;</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo rspace="5.8pt">)</mml:mo>
</mml:mrow>
<mml:mo>&#x00D7;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo rspace="5.8pt">)</mml:mo>
</mml:mrow>
<mml:mo>&#x00D7;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x00D7;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mfrac>
</mml:math>
</disp-formula>
<disp-formula id="S2.E5">
<label>(5)</label>
<mml:math id="M5">
<mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mpadded width="+3.3pt">
<mml:mi>N</mml:mi>
</mml:mpadded>
</mml:mrow>
<mml:mo rspace="10.8pt">=</mml:mo>
<mml:mfrac>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="S2.E6">
<label>(6)</label>
<mml:math id="M6">
<mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mpadded width="+3.3pt">
<mml:mi>P</mml:mi>
</mml:mpadded>
</mml:mrow>
<mml:mo rspace="10.8pt">=</mml:mo>
<mml:mfrac>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <italic>TP</italic> and <italic>TN</italic> refer to correctly predicted 6mA and non-6mA; <italic>FP</italic> and <italic>FN</italic> denote incorrectly predicted non-6mA and 6mA; <italic>n_x</italic> means the number of <italic>x</italic>.</p>
</sec>
</sec>
<sec id="S3" sec-type="results|discussion">
<title>Results and Discussion</title>
<sec id="S3.SS1">
<title>DNA Sequence Logos</title>
<p>To find optimal features of samples, the DNA sequences of samples should be analyzed. Since these sequences were of equal length, they could be analyzed sequence logos (<xref ref-type="bibr" rid="B30">Schneider and Stephens, 1990</xref>). Two Sample Logo was employed (<xref ref-type="bibr" rid="B34">Vacic et al., 2006</xref>), which calculated the statistical difference between positive and negative samples at specific positions. The logo consists of three parts, the upper and lower parts represent the enriched and depleted nucleotides at specific positions, and the middle part denotes the consistent results of positive and negative samples. The x-axis indicates the position. The length of DNA sequences in our datasets is 41bp, so there are 41 scales on the <italic>x</italic>-axis. Additionally, as the middle nucleotide is consistent in both positive and negative samples, it is set to the 0th scale. The y-axis represents the amount of information at the position. The higher the symbol in a position, the more information the position contains. In addition, the relative size of a base letter shows its relative frequency at one position. If a letter is larger than the other letters in the column, it has a high frequency in that position. At each position, the base letters are arranged in the order of dominance from top to bottom. Generally, the consensus motif can be found by reading the top of each position.</p>
<p><xref ref-type="fig" rid="F2">Figures 2A-C</xref> are the sequence logos established for Rosaceae, Rice, and Arabidopsis. From the three figures, it can be seen that the sequences have a length of 41bp with &#x201C;A&#x201D;s at the center. In addition, &#x201C;A&#x201D; enriched at positions &#x2212;6, &#x2212;4, &#x2212;3, 4, 7, 8, 10, 11, 12, &#x201C;C&#x201D; enriched at positions &#x2212;7, &#x2212;2, 2, 6, 9, &#x201C;G&#x201D; enriched at positions &#x2212;8, &#x2212;1, 2, 3, 5, 8, and &#x201C;T&#x201D; enriched at positions 3. Since these sequences containing 6mA are enriched with some nucleotides at some positions, it is speculated that position-based approaches are more suitable for extracting information from the sequences in our datasets.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption><p>Sequence logos of Rosaceae <bold>(A)</bold>, Rice <bold>(B),</bold> and Arabidopsis <bold>(C)</bold>.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-845835-g002.tif"/>
</fig>
</sec>
<sec id="S3.SS2">
<title>Performance Evaluation of Models</title>
<p>To verify the conjecture in the previous section, six methods were chosen to extract the datasets as features and then they were applied to five commonly used well-performing algorithms in sklearn. Since the conjecture is too intuitive and may lead to some significant features being overlooked, not only nucleotide position-based methods are compared, but also density-based and physicochemical property-based methods were also compared.</p>
<p>The experimental results of 5-fold cross-validation are displayed in <xref ref-type="table" rid="T2">Table 2</xref>. The columns indicate the feature extraction methods which have been introduced in the &#x201C;Feature Extraction&#x201D; section. The rows denote classifier algorithms and their evaluation metrics, and they have been briefly described in the &#x201C;Classifiers&#x201D; section and the &#x201C;Performance Evaluation&#x201D; section.</p>
<table-wrap position="float" id="T2">
<label>TABLE 2</label>
<caption><p>Indicators of different features and classifier algorithms.</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<thead>
<tr>
<td valign="top" align="left"></td>
<td/>
<td valign="top" align="center">AMNF</td>
<td valign="top" align="center">ADNF</td>
<td valign="top" align="center">DPCP</td>
<td valign="top" align="center">TPCP</td>
<td valign="top" align="center">One-hot1</td>
<td valign="top" align="center">One-hot2</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Random forest</td>
<td valign="top" align="center">ACC</td>
<td valign="top" align="center">0.786</td>
<td valign="top" align="center">0.642</td>
<td valign="top" align="center">0.594</td>
<td valign="top" align="center">0.669</td>
<td valign="top" align="center">0.935</td>
<td valign="top" align="center"><bold>0.938</bold></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="center">SN</td>
<td valign="top" align="center">0.746</td>
<td valign="top" align="center">0.627</td>
<td valign="top" align="center">0.587</td>
<td valign="top" align="center">0.683</td>
<td valign="top" align="center">0.937</td>
<td valign="top" align="center"><bold>0.939</bold></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="center">SP</td>
<td valign="top" align="center">0.825</td>
<td valign="top" align="center">0.655</td>
<td valign="top" align="center">0.602</td>
<td valign="top" align="center">0.656</td>
<td valign="top" align="center">0.933</td>
<td valign="top" align="center"><bold>0.937</bold></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="center">MCC</td>
<td valign="top" align="center">0.573</td>
<td valign="top" align="center">0.283</td>
<td valign="top" align="center">0.189</td>
<td valign="top" align="center">0.339</td>
<td valign="top" align="center">0.870</td>
<td valign="top" align="center"><bold>0.877</bold></td>
</tr>
<tr>
<td valign="top" align="left">Linear discriminant analysis</td>
<td valign="top" align="center">ACC</td>
<td valign="top" align="center">0.643</td>
<td valign="top" align="center">0.609</td>
<td valign="top" align="center">0.614</td>
<td valign="top" align="center">0.660</td>
<td valign="top" align="center">0.908</td>
<td valign="top" align="center"><bold>0.931</bold></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="center">SN</td>
<td valign="top" align="center">0.650</td>
<td valign="top" align="center">0.624</td>
<td valign="top" align="center">0.597</td>
<td valign="top" align="center">0.629</td>
<td valign="top" align="center">0.937</td>
<td valign="top" align="center"><bold>0.945</bold></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="center">SP</td>
<td valign="top" align="center">0.637</td>
<td valign="top" align="center">0.594</td>
<td valign="top" align="center">0.632</td>
<td valign="top" align="center">0.692</td>
<td valign="top" align="center">0.879</td>
<td valign="top" align="center"><bold>0.917</bold></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="center">MCC</td>
<td valign="top" align="center">0.287</td>
<td valign="top" align="center">0.219</td>
<td valign="top" align="center">0.228</td>
<td valign="top" align="center">0.321</td>
<td valign="top" align="center">0.818</td>
<td valign="top" align="center"><bold>0.862</bold></td>
</tr>
<tr>
<td valign="top" align="left">Multi-layer perceptron</td>
<td valign="top" align="center">ACC</td>
<td valign="top" align="center">0.755</td>
<td valign="top" align="center">0.627</td>
<td valign="top" align="center">0.602</td>
<td valign="top" align="center">0.625</td>
<td valign="top" align="center">0.937</td>
<td valign="top" align="center"><bold>0.939</bold></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="center">SN</td>
<td valign="top" align="center">0.743</td>
<td valign="top" align="center">0.602</td>
<td valign="top" align="center">0.576</td>
<td valign="top" align="center">0.621</td>
<td valign="top" align="center">0.936</td>
<td valign="top" align="center"><bold>0.942</bold></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="center">SP</td>
<td valign="top" align="center">0.767</td>
<td valign="top" align="center">0.652</td>
<td valign="top" align="center">0.628</td>
<td valign="top" align="center">0.629</td>
<td valign="top" align="center"><bold>0.939</bold></td>
<td valign="top" align="center">0.937</td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="center">MCC</td>
<td valign="top" align="center">0.510</td>
<td valign="top" align="center">0.255</td>
<td valign="top" align="center">0.204</td>
<td valign="top" align="center">0.250</td>
<td valign="top" align="center">0.875</td>
<td valign="top" align="center"><bold>0.878</bold></td>
</tr>
<tr>
<td valign="top" align="left">Stochastic gradient descent</td>
<td valign="top" align="center">ACC</td>
<td valign="top" align="center">0.643</td>
<td valign="top" align="center">0.605</td>
<td valign="top" align="center">0.549</td>
<td valign="top" align="center">0.577</td>
<td valign="top" align="center">0.910</td>
<td valign="top" align="center"><bold>0.931</bold></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="center">SN</td>
<td valign="top" align="center">0.631</td>
<td valign="top" align="center">0.647</td>
<td valign="top" align="center">0.561</td>
<td valign="top" align="center">0.481</td>
<td valign="top" align="center">0.917</td>
<td valign="top" align="center"><bold>0.936</bold></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="center">SP</td>
<td valign="top" align="center">0.654</td>
<td valign="top" align="center">0.563</td>
<td valign="top" align="center">0.538</td>
<td valign="top" align="center">0.672</td>
<td valign="top" align="center">0.904</td>
<td valign="top" align="center"><bold>0.926</bold></td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="center">MCC</td>
<td valign="top" align="center">0.287</td>
<td valign="top" align="center">0.212</td>
<td valign="top" align="center">0.099</td>
<td valign="top" align="center">0.157</td>
<td valign="top" align="center">0.821</td>
<td valign="top" align="center"><bold>0.861</bold></td>
</tr>
<tr>
<td valign="top" align="left">Extreme gradient boosting</td>
<td valign="top" align="center">ACC</td>
<td valign="top" align="center">0.790</td>
<td valign="top" align="center">0.647</td>
<td valign="top" align="center">0.616</td>
<td valign="top" align="center">0.673</td>
<td valign="top" align="center"><bold>0.944</bold></td>
<td valign="top" align="center">0.940</td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="center">SN</td>
<td valign="top" align="center">0.788</td>
<td valign="top" align="center">0.650</td>
<td valign="top" align="center">0.617</td>
<td valign="top" align="center">0.672</td>
<td valign="top" align="center"><bold>0.948</bold></td>
<td valign="top" align="center">0.942</td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="center">SP</td>
<td valign="top" align="center">0.791</td>
<td valign="top" align="center">0.644</td>
<td valign="top" align="center">0.616</td>
<td valign="top" align="center">0.675</td>
<td valign="top" align="center"><bold>0.939</bold></td>
<td valign="top" align="center">0.937</td>
</tr>
<tr>
<td valign="top" align="left"/><td valign="top" align="center">MCC</td>
<td valign="top" align="center">0.579</td>
<td valign="top" align="center">0.294</td>
<td valign="top" align="center">0.233</td>
<td valign="top" align="center">0.346</td>
<td valign="top" align="center"><bold>0.888</bold></td>
<td valign="top" align="center">0.880</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn><p><italic>Bold values indicate the best performance.</italic></p></fn>
</table-wrap-foot>
</table-wrap>
<p>As can be seen in <xref ref-type="table" rid="T2">Table 2</xref>, whichever classifier algorithm is selected, the ACCs, SNs, SPs, and MCCs of AMNF, ADNF, DPCP and TPCP are all lower than 0.80, 0.79, 0.83, and 0.60, whereas them of One-hot1 and One-hot2 are all higher than 0.93, 0.93, 0.91, and 0.86. These illustrate that compared with density-based and physicochemical property-based approaches, position-based ways can better express the characteristics contained in DNA sequences in our datasets. XGB performed slightly better with one-hot1 than one-hot2. This may be because XGB may lose some valuable information when it was applied on high-dimensional one-hot2 features. Specifically, XGB divides the high-dimensional feature space into many small parts which may be treated as noise. In addition, if the feature descriptor is One-hot1 or One-hot2, all classifiers show good performance, which indicates that all these algorithms are appropriate for this classification task.</p>
<p>Moreover, to judge intuitively whether the above six feature extraction methods were good at distinguishing between positive and negative samples, the tSNE (<xref ref-type="bibr" rid="B35">van der Maaten and Hinton, 2008</xref>) technique in sklearn (<xref ref-type="bibr" rid="B27">Pedregosa et al., 2011</xref>) was used to project the sample points of these methods from the high-dimensional space to the two-dimensional space. If the positive and negative sample points can be well separated in the two-dimensional space, they are also separable in the high-dimensional space. The visualization plots of the projection are shown in <xref ref-type="fig" rid="F3">Figure 3</xref>. It can be seen from <xref ref-type="fig" rid="F3">Figure 3</xref> that the samples of the two labels are separated by certain dividing lines in <xref ref-type="fig" rid="F3">Figures 3E,F</xref>, while in other subgraphs, the negative sample points are almost covered by the positive ones. These illustrate that One-hot1 and One-hot2 can better discriminate the sample points of the two labels in a high dimensional space than the other four methods.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption><p>The tSNE scatterplots of AMNF <bold>(A)</bold>, ADNF <bold>(B)</bold>, DPCP <bold>(C)</bold>, TPCP <bold>(D)</bold>, One-hot1 <bold>(E)</bold>, and One-hot2 <bold>(F)</bold>. (Blue and pink dots indicate DNA sequence samples with and without 6mA sites, respectively).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-845835-g003.tif"/>
</fig>
<p>Through these arguments, the nucleotide position-based methods are indeed more suitable for extracting features from DNA sequences in our datasets, and the assumptions that was made in the previous section are proved to be correct. Therefore, in the subsequent analysis, only One-hot1 and One-hot2 would be considered.</p>
</sec>
<sec id="S3.SS3">
<title>Comparison of Features</title>
<p>In the previous section, it has been learned that the position-based approaches express the information contained in our DNA sequences well. However, it is not sure which is the best among One-hot1, One-hot2, and their fusion. Therefore, in this subsection, they are compared. The comparison results are shown in <xref ref-type="fig" rid="F4">Figure 4</xref>.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption><p>Comparison before and after feature fusion.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-845835-g004.tif"/>
</fig>
<p>As can be seen from <xref ref-type="fig" rid="F4">Figure 4</xref>, only when the classifier is XGB, the effect of the other two is slightly better than One-hot2; when the classifier is RF, LDA, MLP, or SGD, One-hot2 is significantly better than One-hot1 and slightly better than the fusion. The reason for this is that when encoding a dinucleotide, some information about the mononucleotide is involved. Therefore, in most cases, One-hot1 is not as good as One-hot2, and their fusion produces some redundant information. Consequently, One-hot2 is the best answer.</p>
</sec>
<sec id="S3.SS4">
<title>Efficiency of Ensemble Strategy</title>
<p>Using One-hot2 to extract features and take RF, LDA, MLP, SGD, and XGB as classifiers, five base models can be obtained. As shown in <xref ref-type="fig" rid="F5">Figure 5</xref>, except for some differences between SN and SP of LDA and SGD, SN and SP for the other three classifiers do not differ much, as well as these base models are all with excellent performance, so they were tried to be combined with the majority voting strategy. The integrated results are also shown in <xref ref-type="fig" rid="F5">Figure 5</xref>. It can be found that after voting, except for no enhancement in SP, all the other three metrics improved, which means that after this operation, the performance of the whole classification system has been risen to a higher level.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption><p>Effects of the ensemble strategy.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-845835-g005.tif"/>
</fig>
</sec>
<sec id="S3.SS5">
<title>Comparison With Other Machine Learning Models</title>
<p>To evaluate the generalization capability and cross-species identification ability of our model, it was applied to three test datasets, DS2, DS3, and DS4. Moreover, the test results were compared with several other machine learning models to demonstrate the advantages of our model. <xref ref-type="table" rid="T3">Table 3</xref> shows the comparative results on Rosaceae, Rice, Arabidopsis. The columns indicate four evaluation indicators that have been introduced in the &#x201C;Performance Evaluation&#x201D; section. The rows represent the species and the models applied on these species. The models include Meta-i6mA (<xref ref-type="bibr" rid="B12">Hasan et al., 2021</xref>), i6mA-Fuse (<xref ref-type="bibr" rid="B13">Hasan et al., 2020</xref>), i6mA-stack (<xref ref-type="bibr" rid="B17">Khanal et al., 2021</xref>), i6mA-Pred (<xref ref-type="bibr" rid="B5">Chen et al., 2019</xref>), iDNA6mA-Rice (<xref ref-type="bibr" rid="B24">Lv et al., 2019</xref>), MM-6mAPred (<xref ref-type="bibr" rid="B28">Pian et al., 2019</xref>), and 6mA-Finder (<xref ref-type="bibr" rid="B41">Xu et al., 2020</xref>). Among them, i6mA-Fuse consists of two modules, which were trained by the datasets of Fragaria Vesca and Rosa Chinensis, respectively. To better distinguish them, i6mA-Fuse_FV and i6mA-Fuse_RC are used instead. The same situation is true for i6mA-stack.</p>
<table-wrap position="float" id="T3">
<label>TABLE 3</label>
<caption><p>Comparison with other machine learning models on Rosaceae, Rice, and Arabidopsis.</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<thead>
<tr>
<td valign="top" align="left"></td>
<td/>
<td valign="top" align="center">ACC</td>
<td valign="top" align="center">MCC</td>
<td valign="top" align="center">SN</td>
<td valign="top" align="center">SP</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Rosaceae</td>
<td valign="top" align="center">Meta-i6mA</td>
<td valign="top" align="center">0.953</td>
<td valign="top" align="center">0.905</td>
<td valign="top" align="center">0.954</td>
<td valign="top" align="center">0.951</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">i6mA-Fuse_FV</td>
<td valign="top" align="center">0.943</td>
<td valign="top" align="center">0.887</td>
<td valign="top" align="center">0.924</td>
<td valign="top" align="center"><bold>0.962</bold></td>
</tr>
<tr>
<td/>
<td valign="top" align="center">i6mA-Fuse_RC</td>
<td valign="top" align="center">0.893</td>
<td valign="top" align="center">0.786</td>
<td valign="top" align="center">0.890</td>
<td valign="top" align="center">0.895</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">i6mA-stack_FV</td>
<td valign="top" align="center">0.928</td>
<td valign="top" align="center">0.856</td>
<td valign="top" align="center">0.928</td>
<td valign="top" align="center">0.927</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">i6mA-stack_RC</td>
<td valign="top" align="center">0.899</td>
<td valign="top" align="center">0.798</td>
<td valign="top" align="center">0.920</td>
<td valign="top" align="center">0.877</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">i6mA-Pred</td>
<td valign="top" align="center">0.840</td>
<td valign="top" align="center">0.684</td>
<td valign="top" align="center">0.897</td>
<td valign="top" align="center">0.782</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">iDNA6mA-Rice</td>
<td valign="top" align="center">0.878</td>
<td valign="top" align="center">0.764</td>
<td valign="top" align="center">0.951</td>
<td valign="top" align="center">0.805</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">MM-6mAPred</td>
<td valign="top" align="center">0.873</td>
<td valign="top" align="center">0.758</td>
<td valign="top" align="center"><bold>0.961</bold></td>
<td valign="top" align="center">0.785</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">6mA-Finder</td>
<td valign="top" align="center">0.846</td>
<td valign="top" align="center">0.701</td>
<td valign="top" align="center">0.928</td>
<td valign="top" align="center">0.764</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">i6mA-vote</td>
<td valign="top" align="center"><bold>0.955</bold></td>
<td valign="top" align="center"><bold>0.909</bold></td>
<td valign="top" align="center">0.955</td>
<td valign="top" align="center">0.954</td>
</tr>
<tr>
<td valign="top" align="left">Rice</td>
<td valign="top" align="center">Meta-i6mA</td>
<td valign="top" align="center">0.880</td>
<td valign="top" align="center">0.768</td>
<td valign="top" align="center">0.957</td>
<td valign="top" align="center">0.802</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">i6mA-Fuse_FV</td>
<td valign="top" align="center"><bold>0.890</bold></td>
<td valign="top" align="center"><bold>0.781</bold></td>
<td valign="top" align="center">0.921</td>
<td valign="top" align="center"><bold>0.859</bold></td>
</tr>
<tr>
<td/>
<td valign="top" align="center">i6mA-Fuse_RC</td>
<td valign="top" align="center">0.775</td>
<td valign="top" align="center">0.571</td>
<td valign="top" align="center">0.907</td>
<td valign="top" align="center">0.644</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">i6mA-stack_FV</td>
<td valign="top" align="center">0.876</td>
<td valign="top" align="center">0.756</td>
<td valign="top" align="center">0.938</td>
<td valign="top" align="center">0.815</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">i6mA-stack_RC</td>
<td valign="top" align="center">0.813</td>
<td valign="top" align="center">0.640</td>
<td valign="top" align="center">0.915</td>
<td valign="top" align="center">0.712</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">i6mA-Pred</td>
<td valign="top" align="center">0.791</td>
<td valign="top" align="center">0.592</td>
<td valign="top" align="center">0.878</td>
<td valign="top" align="center">0.705</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">iDNA6mA-Rice</td>
<td valign="top" align="center">0.755</td>
<td valign="top" align="center">0.561</td>
<td valign="top" align="center">0.960</td>
<td valign="top" align="center">0.547</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">MM-6mAPred</td>
<td valign="top" align="center">0.834</td>
<td valign="top" align="center">0.689</td>
<td valign="top" align="center">0.958</td>
<td valign="top" align="center">0.710</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">6mA-Finder</td>
<td valign="top" align="center">0.809</td>
<td valign="top" align="center">0.636</td>
<td valign="top" align="center">0.928</td>
<td valign="top" align="center">0.690</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">i6mA-vote</td>
<td valign="top" align="center">0.882</td>
<td valign="top" align="center">0.774</td>
<td valign="top" align="center"><bold>0.961</bold></td>
<td valign="top" align="center">0.803</td>
</tr>
<tr>
<td valign="top" align="left">Arabidopsis</td>
<td valign="top" align="center">Meta-i6mA</td>
<td valign="top" align="center">0.787</td>
<td valign="top" align="center">0.600</td>
<td valign="top" align="center">0.636</td>
<td valign="top" align="center">0.936</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">i6mA-Fuse_FV</td>
<td valign="top" align="center">0.749</td>
<td valign="top" align="center">0.542</td>
<td valign="top" align="center">0.545</td>
<td valign="top" align="center"><bold>0.949</bold></td>
</tr>
<tr>
<td/>
<td valign="top" align="center">i6mA-Fuse_RC</td>
<td valign="top" align="center">0.757</td>
<td valign="top" align="center">0.534</td>
<td valign="top" align="center">0.615</td>
<td valign="top" align="center">0.897</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">i6mA-stack_FV</td>
<td valign="top" align="center">0.770</td>
<td valign="top" align="center">0.570</td>
<td valign="top" align="center">0.604</td>
<td valign="top" align="center">0.933</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">i6mA-stack_RC</td>
<td valign="top" align="center">0.751</td>
<td valign="top" align="center">0.514</td>
<td valign="top" align="center">0.634</td>
<td valign="top" align="center">0.865</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">i6mA-Pred</td>
<td valign="top" align="center">0.730</td>
<td valign="top" align="center">0.462</td>
<td valign="top" align="center">0.679</td>
<td valign="top" align="center">0.780</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">iDNA6mA-Rice</td>
<td valign="top" align="center">0.734</td>
<td valign="top" align="center">0.473</td>
<td valign="top" align="center">0.655</td>
<td valign="top" align="center">0.812</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">MM-6mAPred</td>
<td valign="top" align="center">0.765</td>
<td valign="top" align="center">0.531</td>
<td valign="top" align="center"><bold>0.784</bold></td>
<td valign="top" align="center">0.747</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">6mA-Finder</td>
<td valign="top" align="center">0.724</td>
<td valign="top" align="center">0.448</td>
<td valign="top" align="center">0.741</td>
<td valign="top" align="center">0.706</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">i6mA-vote</td>
<td valign="top" align="center"><bold>0.798</bold></td>
<td valign="top" align="center"><bold>0.617</bold></td>
<td valign="top" align="center">0.666</td>
<td valign="top" align="center">0.929</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn><p><italic>Bold values indicate the best performance.</italic></p></fn>
</table-wrap-foot>
</table-wrap>
<p>As can be seen from <xref ref-type="table" rid="T3">Table 3</xref>, when the species is Rosaceae, although our SN and SP values only rank second, our ACC and MCC values are the maximum, suggesting that our model has the best overall performance in Rosaceae. It can be concluded that our model can make cross-species predictions for Rice as all four metrics of our model rank at the top. And it can better find 6mA sites from unknown Rice sequences because our model has the highest SN value. Like Rosaceae, our model predicts 6mA sites well in Arabidopsis, and with the highest SP, our model can better screen out those sequences that do not contain 6mA sites. Considering the comparative results on the three species, our model has better generalization performance and cross-species prediction ability than other methods. This may be because only the best-performing feature descriptor was selected to represent the DNA sequences rather than the fusion of several well-performing features. Thereby, the risk of generating irrelevant and redundant features is reduced so that our model has better predictive performance. Furthermore, for Rosaceae, SN is approximately equal to SP and greater than 0.9, indicating that our model has a good discrimination between 6mAs and non-6mAs in the same plant family. For Rice, the SN is greater than 0.9, while the SP is less than 0.9, which may be due to a strong similarity between Rice sequences and Rosaceae positive sequences, resulting in a high false-positive rate and a low true-negative rate when the model recognizes Rice. The situation for Arabidopsis is contrary to that for Rice. It may be because the similarity between Arabidopsis sequences and Rosaceae positive sequences is weak, leading to some 6mAs in Arabidopsis being identified as non-6mAs.</p>
</sec>
</sec>
<sec id="S4" sec-type="conclusion">
<title>Conclusion</title>
<p>In this study, a plant cross-species 6mA site recognition model was constructed by ensemble learning. It has been applied on Rosaceae, Rice, and Arabidopsis and achieved good results. In the construction process, a hypothesis was put forward by analyzing the sequence logos of these three plants. The conjecture was that position-based approaches were more suitable for extracting information from the sequences in our datasets. Next, the hypothesis was verified by comparing different models and observing the tSNE visualization. Then, one-hot encoding for dinucleotide was chosen to represent the datasets by contrasting two nucleotide position-based feature extraction methods and their fusion. Finally, several well-performed models were integrated to form the final classifier by majority voting. To simulate a realistic prediction task, the model was trained on Rosaceae and tested on Rosaceae, Rice, and Arabidopsis. The experimental results showed that our model was adept at predicting the 6mA sites in homologous and heterologous species. In addition, it was also found that there might be a strong similarity between Rice sequences and Rosaceae positive sequences, and the similarity between Arabidopsis sequences and Rosaceae positive sequences is weak. The comparison with other models also showed the superiority of our model. In summary, i6mA-vote outperformed other concerned methods in predicting 6mA sites in the plant genomes. Meanwhile, our research also has the limitation that only three plants were considered. Therefore, future studies will focus on the 6mA site formation characteristics of more plants.</p>
</sec>
<sec id="S5" sec-type="data-availability">
<title>Data Availability Statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found below: <ext-link ext-link-type="uri" xlink:href="https://github.com/zhaozhengnan/i6mA-vote/tree/master">https://github.com/zhaozhengnan/i6mA-vote/tree/master</ext-link>, github.</p>
</sec>
<sec id="S6">
<title>Author Contributions</title>
<p>ZXT improved the model, designed experiments and drafted the manuscript. ZZ proposed the initial idea and implemented the experiments. YL prepared all datasets for experiments. ZT analyzed experimental results. MG revised the manuscript. QL designed experiments and revised the manuscript. GW conceived the whole research process and revised the manuscript. All authors have read and approved the final manuscript.</p>
</sec>
<sec id="conf1" sec-type="COI-statement">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="pudiscl1" sec-type="disclaimer">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<sec id="S7" sec-type="funding-information">
<title>Funding</title>
<p>This manuscript was sponsored by National Natural Science Foundation of China (Grant Nos. 61901103, 61801432, and 61771165), Natural Science Foundation of Heilongjiang Province (Grant No. LH2019F002) and Postdoctoral Science Foundation of Heilongjiang Province of China (Grant No. LBH-Z19106).</p>
</sec>
<sec id="S8" sec-type="supplementary-material">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fpls.2022.845835/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fpls.2022.845835/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Table_1.DOCX" id="TS1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Belhumeur</surname> <given-names>P. N.</given-names></name> <name><surname>Hespanha</surname> <given-names>J. P.</given-names></name> <name><surname>Kriegman</surname> <given-names>D. J.</given-names></name></person-group> (<year>1997</year>). <article-title>Eigenfaces vs. Fisherfaces: recognition using class specific linear projection.</article-title> <source><italic>IEEE Trans. Pattern Anal. Mach. Intell.</italic></source> <volume>19</volume> <fpage>711</fpage>&#x2013;<lpage>720</lpage>. <pub-id pub-id-type="doi">10.1109/34.598228</pub-id></citation></ref>
<ref id="B2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bengio</surname> <given-names>Y.</given-names></name> <name><surname>Glorot</surname> <given-names>X.</given-names></name></person-group> (<year>2010</year>). &#x201C;<article-title>Understanding the difficulty of training deep feed forward neural networks</article-title>,&#x201D; in <source><italic>Proceedings of the 13th International Conference on Artificial Intelligence and Statistics</italic></source>, (Sardinia: Italy), <fpage>249</fpage>&#x2013;<lpage>256</lpage>.</citation></ref>
<ref id="B3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Breiman</surname> <given-names>L.</given-names></name></person-group> (<year>2001</year>). <article-title>Random forests.</article-title> <source><italic>Mach. Learn.</italic></source> <volume>45</volume> <fpage>5</fpage>&#x2013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1023/A:1010933404324</pub-id></citation></ref>
<ref id="B4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>T.</given-names></name> <name><surname>Guestrin</surname> <given-names>C.</given-names></name></person-group> (<year>2016</year>). &#x201C;<article-title>XGBoost: a scalable tree boosting system</article-title>,&#x201D; in <source><italic>Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</italic></source> (<publisher-loc>San Francisco, CA</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>).</citation></ref>
<ref id="B5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>W.</given-names></name> <name><surname>Lv</surname> <given-names>H.</given-names></name> <name><surname>Nie</surname> <given-names>F.</given-names></name> <name><surname>Lin</surname> <given-names>H.</given-names></name></person-group> (<year>2019</year>). <article-title>i6mA-Pred: identifying DNA N6-methyladenine sites in the rice genome.</article-title> <source><italic>Bioinformatics</italic></source> <volume>35</volume> <fpage>2796</fpage>&#x2013;<lpage>2800</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz015</pub-id> <pub-id pub-id-type="pmid">30624619</pub-id></citation></ref>
<ref id="B6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>W.</given-names></name> <name><surname>Yang</surname> <given-names>H.</given-names></name> <name><surname>Feng</surname> <given-names>P.</given-names></name> <name><surname>Ding</surname> <given-names>H.</given-names></name> <name><surname>Lin</surname> <given-names>H.</given-names></name></person-group> (<year>2017</year>). <article-title>iDNA4mC: identifying DNA N4-methylcytosine sites based on nucleotide chemical properties.</article-title> <source><italic>Bioinformatics</italic></source> <volume>33</volume> <fpage>3518</fpage>&#x2013;<lpage>3523</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btx479</pub-id> <pub-id pub-id-type="pmid">28961687</pub-id></citation></ref>
<ref id="B7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Davis</surname> <given-names>B. M.</given-names></name> <name><surname>Chao</surname> <given-names>M. C.</given-names></name> <name><surname>Waldor</surname> <given-names>M. K.</given-names></name></person-group> (<year>2013</year>). <article-title>Entering the era of bacterial epigenomics with single molecule real time DNA sequencing.</article-title> <source><italic>Curr. Opin. Microbiol.</italic></source> <volume>16</volume> <fpage>192</fpage>&#x2013;<lpage>198</lpage>. <pub-id pub-id-type="doi">10.1016/j.mib.2013.01.011</pub-id> <pub-id pub-id-type="pmid">23434113</pub-id></citation></ref>
<ref id="B8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Edgar</surname> <given-names>R.</given-names></name> <name><surname>Domrachev</surname> <given-names>M.</given-names></name> <name><surname>Lash</surname> <given-names>A. E.</given-names></name></person-group> (<year>2002</year>). <article-title>Gene expression omnibus: NCBI gene expression and hybridization array data repository.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>30</volume> <fpage>207</fpage>&#x2013;<lpage>210</lpage>. <pub-id pub-id-type="doi">10.1093/nar/30.1.207</pub-id> <pub-id pub-id-type="pmid">11752295</pub-id></citation></ref>
<ref id="B9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Feng</surname> <given-names>P.</given-names></name> <name><surname>Yang</surname> <given-names>H.</given-names></name> <name><surname>Ding</surname> <given-names>H.</given-names></name> <name><surname>Lin</surname> <given-names>H.</given-names></name> <name><surname>Chen</surname> <given-names>W.</given-names></name> <name><surname>Chou</surname> <given-names>K.-C.</given-names></name></person-group> (<year>2019</year>). <article-title>iDNA6mA-PseKNC: identifying DNA N6-methyladenosine sites by incorporating nucleotide physicochemical properties into PseKNC.</article-title> <source><italic>Genomics</italic></source> <volume>111</volume> <fpage>96</fpage>&#x2013;<lpage>102</lpage>. <pub-id pub-id-type="doi">10.1016/j.ygeno.2018.01.005</pub-id> <pub-id pub-id-type="pmid">29360500</pub-id></citation></ref>
<ref id="B10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fu</surname> <given-names>Y.</given-names></name> <name><surname>Luo</surname> <given-names>G.-Z.</given-names></name> <name><surname>Chen</surname> <given-names>K.</given-names></name> <name><surname>Deng</surname> <given-names>X.</given-names></name> <name><surname>Yu</surname> <given-names>M.</given-names></name> <name><surname>Han</surname> <given-names>D.</given-names></name><etal/></person-group> (<year>2015</year>). <article-title>N6-methyldeoxyadenosine marks active transcription start sites in <italic>Chlamydomonas</italic>.</article-title> <source><italic>Cell</italic></source> <volume>161</volume> <fpage>879</fpage>&#x2013;<lpage>892</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2015.04.010</pub-id> <pub-id pub-id-type="pmid">25936837</pub-id></citation></ref>
<ref id="B11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Greer</surname> <given-names>E. L.</given-names></name> <name><surname>Blanco</surname> <given-names>M. A.</given-names></name> <name><surname>Gu</surname> <given-names>L.</given-names></name> <name><surname>Sendinc</surname> <given-names>E.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name> <name><surname>Aristiz&#x00E1;bal-Corrales</surname> <given-names>D.</given-names></name><etal/></person-group> (<year>2015</year>). <article-title>DNA methylation on N6-adenine in <italic>C. elegans</italic>.</article-title> <source><italic>Cell</italic></source> <volume>161</volume> <fpage>868</fpage>&#x2013;<lpage>878</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2015.04.005</pub-id> <pub-id pub-id-type="pmid">25936839</pub-id></citation></ref>
<ref id="B12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hasan</surname> <given-names>M. M.</given-names></name> <name><surname>Basith</surname> <given-names>S.</given-names></name> <name><surname>Khatun</surname> <given-names>M. S.</given-names></name> <name><surname>Lee</surname> <given-names>G.</given-names></name> <name><surname>Manavalan</surname> <given-names>B.</given-names></name> <name><surname>Kurata</surname> <given-names>H.</given-names></name></person-group> (<year>2021</year>). <article-title>Meta-i6mA: an interspecies predictor for identifying DNA N6-methyladenine sites of plant genomes by exploiting informative features in an integrative machine-learning framework.</article-title> <source><italic>Brief. Bioinform.</italic></source> <volume>22</volume>:<issue>bbaa202</issue>. <pub-id pub-id-type="doi">10.1093/bib/bbaa202</pub-id> <pub-id pub-id-type="pmid">32910169</pub-id></citation></ref>
<ref id="B13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hasan</surname> <given-names>M. M.</given-names></name> <name><surname>Manavalan</surname> <given-names>B.</given-names></name> <name><surname>Shoombuatong</surname> <given-names>W.</given-names></name> <name><surname>Khatun</surname> <given-names>M. S.</given-names></name> <name><surname>Kurata</surname> <given-names>H.</given-names></name></person-group> (<year>2020</year>). <article-title>i6mA-Fuse: improved and robust prediction of DNA 6 mA sites in the Rosaceae genome by fusing multiple feature representation.</article-title> <source><italic>Plant Mol. Biol.</italic></source> <volume>103</volume> <fpage>225</fpage>&#x2013;<lpage>234</lpage>. <pub-id pub-id-type="doi">10.1007/s11103-020-00988-y</pub-id> <pub-id pub-id-type="pmid">32140819</pub-id></citation></ref>
<ref id="B14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Ren</surname> <given-names>S.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2015</year>). &#x201C;<article-title>Delving deep into rectifiers: surpassing human-level performance on ImageNet classification</article-title>,&#x201D; in <source><italic>Proceedings of the 2015 IEEE International Conference on Computer Vision (ICCV)</italic></source>, <publisher-loc>Santiago</publisher-loc>, <fpage>1026</fpage>&#x2013;<lpage>1034</lpage>.</citation></ref>
<ref id="B15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hinton</surname> <given-names>G. E.</given-names></name></person-group> (<year>1989</year>). <article-title>Connectionist learning procedures.</article-title> <source><italic>Artif. Intell.</italic></source> <volume>40</volume> <fpage>185</fpage>&#x2013;<lpage>234</lpage>. <pub-id pub-id-type="doi">10.1016/0004-3702(89)90049-0</pub-id></citation></ref>
<ref id="B16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>H.</given-names></name> <name><surname>Gong</surname> <given-names>X.</given-names></name></person-group> (<year>2020</year>). <article-title>A review of protein inter-residue distance prediction.</article-title> <source><italic>Curr. Bioinformatics</italic></source> <volume>15</volume> <fpage>821</fpage>&#x2013;<lpage>830</lpage>. <pub-id pub-id-type="doi">10.2174/1574893615999200425230056</pub-id></citation></ref>
<ref id="B17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Khanal</surname> <given-names>J.</given-names></name> <name><surname>Lim</surname> <given-names>D. Y.</given-names></name> <name><surname>Tayara</surname> <given-names>H.</given-names></name> <name><surname>Chong</surname> <given-names>K. T.</given-names></name></person-group> (<year>2021</year>). <article-title>i6mA-stack: a stacking ensemble-based computational prediction of DNA N6-methyladenine (6mA) sites in the Rosaceae genome.</article-title> <source><italic>Genomics</italic></source> <volume>113(Pt 2)</volume> <fpage>582</fpage>&#x2013;<lpage>592</lpage>. <pub-id pub-id-type="doi">10.1016/j.ygeno.2020.09.054</pub-id> <pub-id pub-id-type="pmid">33010390</pub-id></citation></ref>
<ref id="B18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kingma</surname> <given-names>D.</given-names></name> <name><surname>Ba</surname> <given-names>J.</given-names></name></person-group> (<year>2014</year>). &#x201C;<article-title>Adam: a method for stochastic optimization</article-title>,&#x201D; in <source><italic>Proceedings of the International Conference on Learning Representations</italic></source>, <publisher-loc>San Diego, United States</publisher-loc>.</citation></ref>
<ref id="B19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kong</surname> <given-names>L.</given-names></name> <name><surname>Zhang</surname> <given-names>L.</given-names></name></person-group> (<year>2019</year>). <article-title>i6mA-DNCP: computational identification of DNA N6-methyladenine sites in the rice genome using optimized dinucleotide-based features.</article-title> <source><italic>Genes</italic></source> <volume>10</volume>:<issue>828</issue>. <pub-id pub-id-type="doi">10.3390/genes10100828</pub-id> <pub-id pub-id-type="pmid">31635172</pub-id></citation></ref>
<ref id="B20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Le</surname> <given-names>N. Q. K.</given-names></name></person-group> (<year>2019</year>). <article-title>iN6-methylat (5-step): identifying DNA N6-methyladenine sites in rice genome using continuous bag of nucleobases via Chou&#x2019;s 5-step rule.</article-title> <source><italic>Mol. Genet. Genomics</italic></source> <volume>294</volume> <fpage>1173</fpage>&#x2013;<lpage>1182</lpage>. <pub-id pub-id-type="doi">10.1007/s00438-019-01570-y</pub-id> <pub-id pub-id-type="pmid">31055655</pub-id></citation></ref>
<ref id="B21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>W.</given-names></name> <name><surname>Godzik</surname> <given-names>A.</given-names></name></person-group> (<year>2006</year>). <article-title>Cd-hit: a fast program for clustering and comparing large sets of protein or nucleotide sequences.</article-title> <source><italic>Bioinformatics</italic></source> <volume>22</volume> <fpage>1658</fpage>&#x2013;<lpage>1659</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btl158</pub-id> <pub-id pub-id-type="pmid">16731699</pub-id></citation></ref>
<ref id="B22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Ouyang</surname> <given-names>X.-H.</given-names></name> <name><surname>Xiao</surname> <given-names>Z.-X.</given-names></name> <name><surname>Zhang</surname> <given-names>L.</given-names></name> <name><surname>Cao</surname> <given-names>Y.</given-names></name></person-group> (<year>2020</year>). <article-title>A review on the methods of peptide-MHC binding prediction.</article-title> <source><italic>Curr. Bioinformatics</italic></source> <volume>15</volume> <fpage>878</fpage>&#x2013;<lpage>888</lpage>. <pub-id pub-id-type="doi">10.2174/1574893615999200429122801</pub-id></citation></ref>
<ref id="B23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Z.-Y.</given-names></name> <name><surname>Xing</surname> <given-names>J.-F.</given-names></name> <name><surname>Chen</surname> <given-names>W.</given-names></name> <name><surname>Luan</surname> <given-names>M.-W.</given-names></name> <name><surname>Xie</surname> <given-names>R.</given-names></name> <name><surname>Huang</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>MDR: an integrative DNA N6-methyladenine and N4-methylcytosine modification database for Rosaceae.</article-title> <source><italic>Hortic. Res.</italic></source> <volume>6</volume>:<issue>78</issue>. <pub-id pub-id-type="doi">10.1038/s41438-019-0160-4</pub-id> <pub-id pub-id-type="pmid">31240103</pub-id></citation></ref>
<ref id="B24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lv</surname> <given-names>H.</given-names></name> <name><surname>Dao</surname> <given-names>F.-Y.</given-names></name> <name><surname>Guan</surname> <given-names>Z.-X.</given-names></name> <name><surname>Zhang</surname> <given-names>D.</given-names></name> <name><surname>Tan</surname> <given-names>J.-X.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>iDNA6mA-Rice: a computational tool for detecting N6-methyladenine sites in rice.</article-title> <source><italic>Front. Genet.</italic></source> <volume>10</volume>:<issue>793</issue>. <pub-id pub-id-type="doi">10.3389/fgene.2019.00793</pub-id> <pub-id pub-id-type="pmid">31552096</pub-id></citation></ref>
<ref id="B25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Manavalan</surname> <given-names>B.</given-names></name> <name><surname>Basith</surname> <given-names>S.</given-names></name> <name><surname>Shin</surname> <given-names>T. H.</given-names></name> <name><surname>Wei</surname> <given-names>L.</given-names></name> <name><surname>Lee</surname> <given-names>G.</given-names></name></person-group> (<year>2019</year>). <article-title>Meta-4mCpred: a sequence-based meta-predictor for accurate DNA 4mC site prediction using effective feature representation.</article-title> <source><italic>Mol. Ther. Nucleic Acids</italic></source> <volume>16</volume> <fpage>733</fpage>&#x2013;<lpage>744</lpage>. <pub-id pub-id-type="doi">10.1016/j.omtn.2019.04.019</pub-id> <pub-id pub-id-type="pmid">31146255</pub-id></citation></ref>
<ref id="B26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Park</surname> <given-names>S.</given-names></name> <name><surname>Wahab</surname> <given-names>A.</given-names></name> <name><surname>Nazari</surname> <given-names>I.</given-names></name> <name><surname>Ryu</surname> <given-names>J. H.</given-names></name> <name><surname>Chong</surname> <given-names>K. T.</given-names></name></person-group> (<year>2020</year>). <article-title>i6mA-DNC: prediction of DNA N6-methyladenosine sites in rice genome based on dinucleotide representation using deep learning.</article-title> <source><italic>Chemometr. Intell. Lab. Syst.</italic></source> <volume>204</volume>:<issue>104102</issue>. <pub-id pub-id-type="doi">10.1016/j.chemolab.2020.104102</pub-id></citation></ref>
<ref id="B27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pedregosa</surname> <given-names>F.</given-names></name> <name><surname>Varoquaux</surname> <given-names>G.</given-names></name> <name><surname>Gramfort</surname> <given-names>A.</given-names></name> <name><surname>Michel</surname> <given-names>V.</given-names></name> <name><surname>Thirion</surname> <given-names>B.</given-names></name> <name><surname>Grisel</surname> <given-names>O.</given-names></name><etal/></person-group> (<year>2011</year>). <article-title>Scikit-learn: machine learning in Python.</article-title> <source><italic>J. Mach. Learn. Res.</italic></source> <volume>12</volume> <fpage>2825</fpage>&#x2013;<lpage>2830</lpage>.</citation></ref>
<ref id="B28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pian</surname> <given-names>C.</given-names></name> <name><surname>Zhang</surname> <given-names>G.</given-names></name> <name><surname>Li</surname> <given-names>F.</given-names></name> <name><surname>Fan</surname> <given-names>X.</given-names></name></person-group> (<year>2019</year>). <article-title>MM-6mAPred: identifying DNA N6-methyladenine sites based on Markov model.</article-title> <source><italic>Bioinformatics</italic></source> <volume>36</volume> <fpage>388</fpage>&#x2013;<lpage>392</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz556</pub-id> <pub-id pub-id-type="pmid">31297537</pub-id></citation></ref>
<ref id="B29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Platt</surname> <given-names>J.</given-names></name></person-group> (<year>2000</year>). &#x201C;<article-title>Probabilistic outputs for support vector machines and comparisons to regularized likelihood methods</article-title>,&#x201D; in <source><italic>Advances in Large Margin Classifiers</italic></source>, <volume>Vol. 10</volume> <role>eds</role> <person-group person-group-type="editor"><name><surname>Smola</surname> <given-names>A.</given-names></name> <name><surname>Bartlett</surname> <given-names>P.</given-names></name> <name><surname>Sch&#x00F6;lkopf</surname> <given-names>B.</given-names></name> <name><surname>Schuurmans</surname> <given-names>D.</given-names></name></person-group> (<publisher-loc>Cambridge, MA</publisher-loc>: <publisher-name>MIT Press</publisher-name>).</citation></ref>
<ref id="B30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schneider</surname> <given-names>T. D.</given-names></name> <name><surname>Stephens</surname> <given-names>R. M.</given-names></name></person-group> (<year>1990</year>). <article-title>Sequence logos: a new way to display consensus sequences.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>18</volume> <fpage>6097</fpage>&#x2013;<lpage>6100</lpage>. <pub-id pub-id-type="doi">10.1093/nar/18.20.6097</pub-id> <pub-id pub-id-type="pmid">2172928</pub-id></citation></ref>
<ref id="B31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shao</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>B.</given-names></name></person-group> (<year>2021</year>). <article-title>ProtFold-DFG: protein fold recognition by combining Directed Fusion Graph and PageRank algorithm.</article-title> <source><italic>Brief. Bioinform.</italic></source> <volume>22</volume>:<issue>bbaa192</issue>. <pub-id pub-id-type="doi">10.1093/bib/bbaa192</pub-id> <pub-id pub-id-type="pmid">32892224</pub-id></citation></ref>
<ref id="B32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Smolarczyk</surname> <given-names>T.</given-names></name> <name><surname>Roterman-Konieczna</surname> <given-names>I.</given-names></name> <name><surname>Stapor</surname> <given-names>K.</given-names></name></person-group> (<year>2020</year>). <article-title>Protein secondary structure prediction: a review of progress and directions.</article-title> <source><italic>Curr. Bioinformatics</italic></source> <volume>15</volume> <fpage>90</fpage>&#x2013;<lpage>107</lpage>. <pub-id pub-id-type="doi">10.2174/1574893614666191017104639</pub-id></citation></ref>
<ref id="B33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tahir</surname> <given-names>M.</given-names></name> <name><surname>Tayara</surname> <given-names>H.</given-names></name> <name><surname>Chong</surname> <given-names>K. T.</given-names></name></person-group> (<year>2019</year>). <article-title>iDNA6mA (5-step rule): identification of DNA N6-methyladenine sites in the rice genome by intelligent computational model via Chou&#x2019;s 5-step rule.</article-title> <source><italic>Chemometr. Intell. Lab. Syst.</italic></source> <volume>189</volume> <fpage>96</fpage>&#x2013;<lpage>101</lpage>. <pub-id pub-id-type="doi">10.1016/j.chemolab.2019.04.007</pub-id></citation></ref>
<ref id="B34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vacic</surname> <given-names>V.</given-names></name> <name><surname>Iakoucheva</surname> <given-names>L. M.</given-names></name> <name><surname>Radivojac</surname> <given-names>P.</given-names></name></person-group> (<year>2006</year>). <article-title>Two Sample Logo: a graphical representation of the differences between two sets of sequence alignments.</article-title> <source><italic>Bioinformatics</italic></source> <volume>22</volume> <fpage>1536</fpage>&#x2013;<lpage>1537</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btl151</pub-id> <pub-id pub-id-type="pmid">16632492</pub-id></citation></ref>
<ref id="B35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>van der Maaten</surname> <given-names>L. J. P.</given-names></name> <name><surname>Hinton</surname> <given-names>G. E.</given-names></name></person-group> (<year>2008</year>). <article-title>Visualizing high-dimensional data using t-SNE.</article-title> <source><italic>J. Mach. Learn. Res.</italic></source> <volume>9</volume> <fpage>2579</fpage>&#x2013;<lpage>2605</lpage>.</citation></ref>
<ref id="B36"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vanyushin</surname> <given-names>B. F.</given-names></name> <name><surname>Belozersky</surname> <given-names>A. N.</given-names></name> <name><surname>Kokurina</surname> <given-names>N. A.</given-names></name> <name><surname>Kadirova</surname> <given-names>D. X.</given-names></name></person-group> (<year>1968</year>). <article-title>5-Methylcytosine and 6-methylaminopurine in bacterial DNA.</article-title> <source><italic>Nature</italic></source> <volume>218</volume> <fpage>1066</fpage>&#x2013;<lpage>1067</lpage>. <pub-id pub-id-type="doi">10.1038/2181066a0</pub-id> <pub-id pub-id-type="pmid">5656625</pub-id></citation></ref>
<ref id="B37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Ding</surname> <given-names>Y.</given-names></name> <name><surname>Tang</surname> <given-names>J.</given-names></name> <name><surname>Guo</surname> <given-names>F.</given-names></name></person-group> (<year>2020</year>). <article-title>Identification of membrane protein types via multivariate information fusion with Hilbert&#x2013;Schmidt independence criterion.</article-title> <source><italic>Neurocomputing</italic></source> <volume>383</volume> <fpage>257</fpage>&#x2013;<lpage>269</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2019.11.103</pub-id></citation></ref>
<ref id="B38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Shi</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Chang</surname> <given-names>H.</given-names></name></person-group> (<year>2020</year>). <article-title>A drug target interaction prediction based on LINE-RF learning.</article-title> <source><italic>Curr. Bioinformatics</italic></source> <volume>15</volume> <fpage>750</fpage>&#x2013;<lpage>757</lpage>. <pub-id pub-id-type="doi">10.2174/1574893615666191227092453</pub-id></citation></ref>
<ref id="B39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wei</surname> <given-names>L.</given-names></name> <name><surname>Su</surname> <given-names>R.</given-names></name> <name><surname>Luan</surname> <given-names>S.</given-names></name> <name><surname>Liao</surname> <given-names>Z.</given-names></name> <name><surname>Manavalan</surname> <given-names>B.</given-names></name> <name><surname>Zou</surname> <given-names>Q.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Iterative feature representations improve N4-methylcytosine site prediction.</article-title> <source><italic>Bioinformatics</italic></source> <volume>35</volume> <fpage>4930</fpage>&#x2013;<lpage>4937</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz408</pub-id> <pub-id pub-id-type="pmid">31099381</pub-id></citation></ref>
<ref id="B40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wion</surname> <given-names>D.</given-names></name> <name><surname>Casades&#x00FA;s</surname> <given-names>J.</given-names></name></person-group> (<year>2006</year>). <article-title>N6-methyl-adenine: an epigenetic signal for DNA&#x2013;protein interactions.</article-title> <source><italic>Nat. Rev. Microbiol.</italic></source> <volume>4</volume> <fpage>183</fpage>&#x2013;<lpage>192</lpage>. <pub-id pub-id-type="doi">10.1038/nrmicro1350</pub-id> <pub-id pub-id-type="pmid">16489347</pub-id></citation></ref>
<ref id="B41"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>H.</given-names></name> <name><surname>Hu</surname> <given-names>R.</given-names></name> <name><surname>Jia</surname> <given-names>P.</given-names></name> <name><surname>Zhao</surname> <given-names>Z.</given-names></name></person-group> (<year>2020</year>). <article-title>6mA-Finder: a novel online tool for predicting DNA N6-methyladenine sites in genomes.</article-title> <source><italic>Bioinformatics</italic></source> <volume>36</volume> <fpage>3257</fpage>&#x2013;<lpage>3259</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa113</pub-id> <pub-id pub-id-type="pmid">32091591</pub-id></citation></ref>
<ref id="B42"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ye</surname> <given-names>P.</given-names></name> <name><surname>Luan</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>K.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Xiao</surname> <given-names>C.</given-names></name> <name><surname>Xie</surname> <given-names>Z.</given-names></name></person-group> (<year>2017</year>). <article-title>MethSMRT: an integrative database for DNA N6-methyladenine and N4-methylcytosine generated by single-molecular real-time sequencing.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>45</volume> <fpage>D85</fpage>&#x2013;<lpage>D89</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkw950</pub-id> <pub-id pub-id-type="pmid">27924023</pub-id></citation></ref>
<ref id="B43"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>H.</given-names></name> <name><surname>Dai</surname> <given-names>Z.</given-names></name></person-group> (<year>2019</year>). <article-title>SNNRice6mA: a deep learning method for predicting DNA N6-methyladenine sites in rice genome.</article-title> <source><italic>Front. Genet.</italic></source> <volume>10</volume>:<issue>1071</issue>. <pub-id pub-id-type="doi">10.3389/fgene.2019.01071</pub-id> <pub-id pub-id-type="pmid">31681441</pub-id></citation></ref>
<ref id="B44"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>D.</given-names></name> <name><surname>Chen</surname> <given-names>H.-D.</given-names></name> <name><surname>Zulfiqar</surname> <given-names>H.</given-names></name> <name><surname>Yuan</surname> <given-names>S.-S.</given-names></name> <name><surname>Huang</surname> <given-names>Q.-L.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.-Y.</given-names></name><etal/></person-group> (<year>2021</year>). <article-title>iBLP: an XGBoost-based predictor for identifying bioluminescent proteins.</article-title> <source><italic>Comput. Math. Methods Med.</italic></source> <volume>2021</volume>:<issue>6664362</issue>. <pub-id pub-id-type="doi">10.1155/2021/6664362</pub-id> <pub-id pub-id-type="pmid">33505515</pub-id></citation></ref>
<ref id="B45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>G.</given-names></name> <name><surname>Huang</surname> <given-names>H.</given-names></name> <name><surname>Liu</surname> <given-names>D.</given-names></name> <name><surname>Cheng</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Zhang</surname> <given-names>W.</given-names></name><etal/></person-group> (<year>2015</year>). <article-title>N6-methyladenine DNA modification in <italic>Drosophila</italic>.</article-title> <source><italic>Cell</italic></source> <volume>161</volume> <fpage>893</fpage>&#x2013;<lpage>906</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2015.04.018</pub-id> <pub-id pub-id-type="pmid">25936838</pub-id></citation></ref>
</ref-list>
<fn-group>
<fn id="footnote1">
<label>1</label>
<p><ext-link ext-link-type="uri" xlink:href="http://kurata14.bio.kyutech.ac.jp/Meta-i6mA/download_file/Meta-6mA-datasets.zip">http://kurata14.bio.kyutech.ac.jp/Meta-i6mA/download_file/Meta-6mA-datasets.zip</ext-link></p></fn>
</fn-group>
</back>
</article>
