<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Immunol.</journal-id>
<journal-title>Frontiers in Immunology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Immunol.</abbrev-journal-title>
<issn pub-type="epub">1664-3224</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fimmu.2023.1267755</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Immunology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Stacking-ac4C: an ensemble model using mixed features for identifying n4-acetylcytidine in mRNA</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Lou</surname>
<given-names>Li-Liang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Qiu</surname>
<given-names>Wang-Ren</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/808177"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Zi</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2258439"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xu</surname>
<given-names>Zhao-Chun</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/838383"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xiao</surname>
<given-names>Xuan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Huang</surname>
<given-names>Shun-Fa</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Computer Department, Jing-De-Zhen Ceramic Institute</institution>, <addr-line>Jingdezhen</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>School of Information Engineering , Jingdezhen University</institution>, <addr-line>Jingdezhen</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Shoubao Ma, City of Hope National Medical Center, United States</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Pengmian Feng, North China University of Science and Technology, China; Lei Wang, Changsha University, China; Nguyen Quoc Khanh Le, Taipei Medical University, Taiwan</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Wang-Ren Qiu, <email xlink:href="mailto:qiuone@163.com">qiuone@163.com</email>; Shun-Fa Huang, <email xlink:href="mailto:hsf65689@126.com">hsf65689@126.com</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>29</day>
<month>11</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>14</volume>
<elocation-id>1267755</elocation-id>
<history>
<date date-type="received">
<day>27</day>
<month>07</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>14</day>
<month>11</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Lou, Qiu, Liu, Xu, Xiao and Huang</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Lou, Qiu, Liu, Xu, Xiao and Huang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>N4-acetylcytidine (ac4C) is a modification of cytidine at the nitrogen-4 position, playing a significant role in the translation process of mRNA. However, the precise mechanism and details of how ac4C modifies translated mRNA remain unclear. Since identifying ac4C sites using conventional experimental methods is both labor-intensive and time-consuming, there is an urgent need for a method that can promptly recognize ac4C sites. In this paper, we propose a comprehensive ensemble learning model, the Stacking-based heterogeneous integrated ac4C model, engineered explicitly to identify ac4C sites. This innovative model integrates three distinct feature extraction methodologies: Kmer, electron-ion interaction pseudo-potential values (PseEIIP), and pseudo-K-tuple nucleotide composition (PseKNC). The model also incorporates the robust Cluster Centroids algorithm to enhance its performance in dealing with imbalanced data and alleviate underfitting issues. Our independent testing experiments indicate that our proposed model improves the Mcc by 15.61% and the ROC by 5.97% compared to existing models. To test our model&#x2019;s adaptability, we also utilized a balanced dataset assembled by the authors of iRNA-ac4C. Our model showed an increase in Sn of 4.1%, an increase in Acc of nearly 1%, and ROC improvement of 0.35% on this balanced dataset. The code for our model is freely accessible at <ext-link ext-link-type="uri" xlink:href="https://github.com/louliliang/ST-ac4C.git">https://github.com/louliliang/ST-ac4C.git</ext-link>, allowing users to quickly build their model without dealing with complicated mathematical equations.</p>
</abstract>
<kwd-group>
<kwd>N4-acetylcytidine</kwd>
<kwd>feature extraction</kwd>
<kwd>stacking heterogeneous integration</kwd>
<kwd>Cluster Centroids algorithm</kwd>
<kwd>ensemble model</kwd>
</kwd-group>
<counts>
<fig-count count="6"/>
<table-count count="8"/>
<equation-count count="8"/>
<ref-count count="46"/>
<page-count count="11"/>
<word-count count="5036"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Molecular Innate Immunity</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>To date, over 170 modified nucleosides have been found in RNA. These post-transcriptional modifications play a significant role in molecular interactions and intermolecular relations. Introducing subtle structural changes contributes to RNA&#x2019;s functional diversity by regulating translation efficiency, mRNA stability, and RNA-protein interactions &#x2013; all factors that are vital for cellular growth and development (<xref ref-type="bibr" rid="B1">1</xref>, <xref ref-type="bibr" rid="B2">2</xref>). Ac4C has been linked with various human diseases, including inflammation, metabolic disorders, autoimmune diseases, and cancer (<xref ref-type="bibr" rid="B3">3</xref>). Identifying and examining ac4C sites are critical areas in biological and bioinformatics research. In early studies, the identification of ac4C sites was mainly done through experiments such as high-performance liquid chromatography (HPLC) and HPLC-mass Spectrometry. However, as these experimental methods require substantial time to detect ac4C in mRNA, there is an urgent need for computer-based methods that can identify ac4C sites accurately and reliably.</p>
<p>In recent years, four computational methods have been developed to identify ac4C sites in human mRNA. The first one, PACES, was developed by Zhao et&#xa0;al. (<xref ref-type="bibr" rid="B4">4</xref>). PACES utilizes position-specific dinucleotide sequence Spectra and K-nucleotide frequencies as coding methods, with random forest (RF) (<xref ref-type="bibr" rid="B5">5</xref>) deployed as a training model to yield the outcomes. PACES (<xref ref-type="bibr" rid="B4">4</xref>) achieved an area under the characteristic curve and the exact recall curve of 0.874 and 0.485, respectively. The second approach is an integrated model (XGBoost) (<xref ref-type="bibr" rid="B6">6</xref>, <xref ref-type="bibr" rid="B7">7</xref>) proposed by Alam for predicting ac4C locations. The ROC and PRC of the XGBoost model are 0.889 and 0.581, respectively. Subsequently, the third method is Wang&#x2019;s DeepAc4C (<xref ref-type="bibr" rid="B8">8</xref>) model, which is built based on a convolutional neural network (CNN) (<xref ref-type="bibr" rid="B9">9</xref>) and a hybrid feature that integrates physicochemical patterns and nucleic acid distribution. The last prediction method is a gradient boosting decision tree (GBDT) (<xref ref-type="bibr" rid="B10">10</xref>, <xref ref-type="bibr" rid="B11">11</xref>) based on Kmer (<xref ref-type="bibr" rid="B12">12</xref>) nucleotide composition, nucleotide chemistry NCP (<xref ref-type="bibr" rid="B13">13</xref>), cumulative nucleotide frequency ANF (<xref ref-type="bibr" rid="B14">14</xref>), and minimum redundancy maximum correlation mRMR (<xref ref-type="bibr" rid="B15">15</xref>). The model achieved ROC values of 0.875 and 0.880 on the training and independent test datasets, respectively. Despite these models showing commendable performance, considerable scope remains for enhancing the predictive efficacy of all models, as mentioned earlier. To improve the performance of ac4C site identification, we have proposed a novel method based on integrated learning Stacking (<xref ref-type="bibr" rid="B16">16</xref>) called Stacking-ac4C, as shown in <xref ref-type="fig" rid="f1">
<bold>Figure 1</bold>
</xref>. This approach consolidates K nucleotide composition Kmer, electronic energy PseEIIP (<xref ref-type="bibr" rid="B17">17</xref>) based on normalized trinucleotide frequencies and four nucleotides, along with trinucleotide occurrence frequencies and six physicochemical indicators PseKNC (<xref ref-type="bibr" rid="B18">18</xref>). The ROC of the proposed model on the cross-validation and independent test datasets were 0.9540 and 0.9487, respectively, which shows excellent performance compared to all the predictors, as mentioned earlier.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>The scheme diagram for establishing Stacking-ac4C.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-14-1267755-g001.tif"/>
</fig>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>Data collection and preprocessing</title>
<p>To develop a valuable and unbiased model, we extracted the data from PACES (<xref ref-type="bibr" rid="B4">4</xref>), available at <ext-link ext-link-type="uri" xlink:href="http://www.rnanut.net/paces/">http://www.rnanut.net/paces/</ext-link>. These data were also used for training and testing the models of XG-ac4C (<xref ref-type="bibr" rid="B7">7</xref>) and DeepAc4C (<xref ref-type="bibr" rid="B8">8</xref>), initially extracted by Danial Arango from 2134 genes with positive and negative ac4C sites. All of these genes were experimentally validated by high-throughput acRIP seq. This study&#x2019;s training dataset consists of 1160 positive and 10855 negative samples. The independent testing dataset comprises 469 positive samples and 4343 negative samples. To demonstrate the model&#x2019;s portability, the results were validated using a dataset constructed from the iRNA-ac4C article, which was collected by Arango et&#xa0;al. (<xref ref-type="bibr" rid="B19">19</xref>). Data can be obtained from the website <ext-link ext-link-type="uri" xlink:href="http://lin-group.cn/server/iRNA-ac4C/">http://lin-group.cn/server/iRNA-ac4C/</ext-link>. During the experiment, the CD-HIT (<xref ref-type="bibr" rid="B20">20</xref>) tool was used to remove sequence pair similarity larger than 0.8. The training dataset we obtained consists of 2206 positive samples and 2206 negative samples. The independent testing dataset consists of 552 positive samples and 552 negative samples, which are balanced data. Finally, a balanced dataset with a sequence length of 201bp was obtained. The specific information about the dataset is shown in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Data source distribution table.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Data source</th>
<th valign="middle" align="center">subdataset</th>
<th valign="middle" align="center">Number of Positive Samples</th>
<th valign="middle" align="center">Number of Negative Samples</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="3" align="center">PACES (<xref ref-type="bibr" rid="B4">4</xref>)</td>
<td valign="top" align="center">Training</td>
<td valign="middle" align="center">1160</td>
<td valign="middle" align="center">10855</td>
</tr>
<tr>
<td valign="top" align="center">Testing</td>
<td valign="middle" align="center">469</td>
<td valign="middle" align="center">4343</td>
</tr>
<tr>
<td valign="top" align="center">Total</td>
<td valign="middle" align="center">1629</td>
<td valign="middle" align="center">15198</td>
</tr>
<tr>
<td valign="top" rowspan="3" align="center">iRNA-ac4C (<xref ref-type="bibr" rid="B10">10</xref>, <xref ref-type="bibr" rid="B11">11</xref>)</td>
<td valign="top" align="center">Training</td>
<td valign="middle" align="center">2206</td>
<td valign="middle" align="center">2206</td>
</tr>
<tr>
<td valign="top" align="center">Testing</td>
<td valign="middle" align="center">552</td>
<td valign="middle" align="center">552</td>
</tr>
<tr>
<td valign="top" align="center">Total</td>
<td valign="middle" align="center">2758</td>
<td valign="middle" align="center">2758</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Sample formulation</title>
<p>Once the benchmark dataset has been prepared for the study, the next important step is formulating the samples and extracting the best feature set for constructing a robust and superior computational model. In recent years, various feature encoding strategies have been used to form biological sequence fragments, such as PseKNC (<xref ref-type="bibr" rid="B17">17</xref>), One-hot (<xref ref-type="bibr" rid="B21">21</xref>, <xref ref-type="bibr" rid="B22">22</xref>), physicochemical features, and word2vec (<xref ref-type="bibr" rid="B23">23</xref>&#x2013;<xref ref-type="bibr" rid="B25">25</xref>). This study selected some of the most common feature encoding approaches, including six physicochemical feature encoding strategies and the frequency of occurrence of k-nearest neighbor nucleic acids, to describe RNA fragments. Below, we elaborate on their respective principles in detail.</p>
<sec id="s2_2_1">
<label>2.2.1</label>
<title>Kmer nucleotide composition</title>
<p>The main idea of Kmer is the frequency of <italic>k</italic> nucleotides in an RNA sequence. The RNA sequence R can be transformed into a vector with 4k dimensions by using the Kmer frequency as follows:</p>
<disp-formula>
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:msub>
<mml:mtext>R</mml:mtext>
<mml:mrow>
<mml:mtext>k</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>mer</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mn>4</mml:mn>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the normalized frequency of occurrence of the ith Kmer nucleotide in the sample sequence, and T denotes the transposition of the matrix. <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> can be expressed as:</p>
<disp-formula>
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>K</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where N(t) is the number of Kmer type t in the RNA sequence R.</p>
</sec>
<sec id="s2_2_2">
<label>2.2.2</label>
<title>PseKNC</title>
<p>The pseudo-k-tuple composition PseKNC is similar to SCPseDNC (<xref ref-type="bibr" rid="B26">26</xref>) and SCPseTNC (<xref ref-type="bibr" rid="B27">27</xref>), while PseKNC contains the frequency of trinucleotide occurrences and fusion information of six physicochemical indicators. The PseKNC contains a k-tuple nucleotide composition, which can be defined as:</p>
<disp-formula>
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>D</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">[</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>&#x2026;</mml:mi>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:msup>
<mml:mn>4</mml:mn>
<mml:mi>k</mml:mi>
</mml:msup>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mn>4</mml:mn>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mn>4</mml:mn>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
<mml:msup>
<mml:mo stretchy="false">]</mml:mo>
<mml:mi>T</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula>
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mn>4</mml:mn>
<mml:mi>k</mml:mi>
</mml:msup>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mn>4</mml:mn>
<mml:mi>k</mml:mi>
</mml:msup>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mn>4</mml:mn>
<mml:mi>k</mml:mi>
</mml:msup>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>&#x3bc;</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:msup>
<mml:mn>4</mml:mn>
<mml:mi>k</mml:mi>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>&#x3bc;</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mn>4</mml:mn>
<mml:mi>k</mml:mi>
</mml:msup>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>&#x3bc;</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where &#x3bb; is the number of total count levels (or hierarchies) of correlations along the nucleotide sequence; <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>&#x3bc;</mml:mi>
</mml:msub>
<mml:mo>&#xa0;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mn>4</mml:mn>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the frequency of nucleotides <inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mn>4</mml:mn>
<mml:mi>k</mml:mi>
</mml:msup>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> , w is the factor that &#x3b8;<sub>j</sub> defined as:</p>
<disp-formula>
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>,</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>&lt;</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> can be defined as:</p>
<disp-formula>
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>&#x3bc;</mml:mi>
</mml:mfrac>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>&#x3bc; is the number of physicochemical indices, i.e., six indices (&#x201c;rise&#x201d;, &#x201c;roll&#x201d;, &#x201c;translate&#x201d;, &#x201c;slide&#x201d;, &#x201c;slide &#x201c;tilt&#x201d;, &#x201c;twist&#x201d;) are set as RNA sequences, and <inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the value of the corresponding physicochemical index (<inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:mtext>v</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mtext>&#x3bc;</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>). The physicochemical index of the nucleotid <inline-formula>
<mml:math display="inline" id="im8">
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is at position <italic>i</italic>. <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> indicates the corresponding value of the nucleotide <inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> at position <italic>i</italic> + <italic>j</italic>.</p>
</sec>
<sec id="s2_2_3">
<label>2.2.3</label>
<title>PseEIIP</title>
<p>The electron-ion interaction pseudo-potential (EIIP) values for nucleotides A, G, C, and T are as follows: A (0.1260), C (0.1340), G (0.806), and T (0.1335) (<xref ref-type="bibr" rid="B28">28</xref>). The EIIP values for the nucleotides A, T, G, and C are denoted as EIIPA, EIIPT, EIIPG, and EIIPC, respectively. The average EIIP values of the three nucleotides in each sample were used to construct the eigenvectors, and the formula can be expressed as:</p>
<disp-formula>
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:mtext>V</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>I</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>E</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>I</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>E</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>I</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>T</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>T</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where f<sub>xyz</sub> denotes the normalized frequency of the <italic>i</italic>-th trinucleotide, and <inline-formula>
<mml:math display="inline" id="im11">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>I</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mi>y</mml:mi>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>E</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>I</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>x</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>E</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>I</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>y</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>E</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>I</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>Z</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the EIIP value of a trinucleotide and <inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:mtext>X</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>Y</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>Z</mml:mtext>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>C</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>G</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>T</mml:mtext>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>.The dimensionality of the vector representation is 64.</p>
</sec>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Feature fusion</title>
<p>Three feature codes, Kmer (<xref ref-type="bibr" rid="B12">12</xref>), PseEIIP (<xref ref-type="bibr" rid="B17">17</xref>), and PseKNC (<xref ref-type="bibr" rid="B17">17</xref>), were combined to describe the ac4C locus samples, 48-D, 84-D, and 64-D feature vectors were obtained, respectively. These feature vectors describe the adjacent positional correlation information of the sequence and enhance the extraction of sequence information by utilizing the physical and chemical properties of nucleotides. The hybrid features are obtained by fusing these features to reach 196-D. To investigate what feature fusion would arrive at the optimal training results, we compare the four combinations of PseKNC, Kmer, PseKNC+PseEIIP, and Kmer+PseKNC+ PseEIIPPSEEIIP encoded in the Stacking model validated by 10-fold cross-validation and measured by Sn, Sp, Acc, Mcc, ROC, and PRC, as shown in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>. Where the evaluation metrics (Acc, Mcc, ROC, PRC) derived from training the Stacking algorithm model with the coding approach (Kmer+PseKNC+PSEEIIP) in cross-validation are higher than the mean values of the evaluation metrics derived from training with the previous four coding approaches by 5.135%, 3.865%, 3.62%, and 4.575%. This may be because multiple feature sets can utilize the advantage of one of the local features to compensate for the disadvantage of another local feature due to compensating for the disadvantage of the other local feature, so that the individual local features can be fused more effectively, thus, significantly enhancing the robustness of multiple feature sets.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Comparison of cross-validation performance between Stacking algorithms trained with different feature combinations.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Feature</th>
<th valign="top" align="center">Sn</th>
<th valign="top" align="center">Sp</th>
<th valign="top" align="center">Acc</th>
<th valign="top" align="center">Mcc</th>
<th valign="top" align="center">ROC</th>
<th valign="top" align="center">PRC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">PseKNC</td>
<td valign="middle" align="center">0.8455</td>
<td valign="middle" align="center">0.7944</td>
<td valign="middle" align="center">0.8188</td>
<td valign="middle" align="center">0.6395</td>
<td valign="middle" align="center">0.9071</td>
<td valign="middle" align="center">0.8985</td>
</tr>
<tr>
<td valign="middle" align="center">Kmer</td>
<td valign="middle" align="center">0.8783</td>
<td valign="middle" align="center">0.8280</td>
<td valign="middle" align="center">0.8532</td>
<td valign="middle" align="center">0.7062</td>
<td valign="middle" align="center">0.9258</td>
<td valign="middle" align="center">0.9112</td>
</tr>
<tr>
<td valign="middle" align="center">PseEIIP</td>
<td valign="middle" align="center">0.8616</td>
<td valign="middle" align="center">0.8220</td>
<td valign="middle" align="center">0.8410</td>
<td valign="middle" align="center">0.6826</td>
<td valign="middle" align="center">0.9220</td>
<td valign="middle" align="center">0.9143</td>
</tr>
<tr>
<td valign="middle" align="center">PseKNC+PseEIIP</td>
<td valign="middle" align="center">0.8642</td>
<td valign="middle" align="center">0.8049</td>
<td valign="middle" align="center">0.8340</td>
<td valign="middle" align="center">0.6692</td>
<td valign="middle" align="center">0.9167</td>
<td valign="middle" align="center">0.9066</td>
</tr>
<tr>
<td valign="middle" align="center">Kmer+PseKNC+PseEIIP</td>
<td valign="middle" align="center">
<bold>0.8922</bold>
</td>
<td valign="middle" align="center">
<bold>0.8840</bold>
</td>
<td valign="middle" align="center">
<bold>0.8881</bold>
</td>
<td valign="middle" align="center">
<bold>0.7749</bold>
</td>
<td valign="middle" align="center">
<bold>0.9541</bold>
</td>
<td valign="middle" align="center">
<bold>0.9534</bold>
</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Stacking classification algorithm</title>
<p>A stacking model is not, strictly speaking, an algorithm but a strategy for model integration. As shown in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>, the Stacking integration algorithm can be understood as a two-layer integration, where the first layer contains several base classifiers, also called base classifiers, which provide the predicted results (meta-features) to the second layer. In contrast, the second layer classifier is a logistic regression, which takes the results of the first layer classifiers as features to fit the predicted results.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Stacking algorithm logic structure diagram.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-14-1267755-g002.tif"/>
</fig>
<p>The base classifiers of Stacking are usually obtained by training different learning algorithms. Stacking can also be considered a particular and specific combination strategy, typical of learning methods (<xref ref-type="bibr" rid="B16">16</xref>). In the Stacking training phase, the data used to train one layer of classifiers are also used to generate data for the second layer of classifiers, which runs the risk of overfitting. Thus, we used cross-validation to train the data and reduce this risk.</p>
</sec>
<sec id="s2_5">
<label>2.5</label>
<title>Imbalance data processing</title>
<p>The dataset of the PACES article is an unbalanced dataset with a ratio of positive and negative samples of 1 to 10, which results in the amount of information from the positive samples not being able to counteract the amount of information from the negative samples during the training of the Stacking-ac4C model, leading to a large number of misclassifications of the positive samples when the model is doing independent testing. Therefore, an algorithm is needed to reduce the imbalance between the number of positive and negative samples. Furthermore, data resampling is the most representative method to classify unbalanced data. In this study, we experimented with the two resampling widely adopted techniques: oversampling and undersampling (<xref ref-type="bibr" rid="B29">29</xref>). Due to the large number of negative samples of ac4C data with 415 nucleotide sequences in length, the use of the oversampling method is likely to lead to overfitting of the model, so we chose the cluster center method among the undersampling methods. This method first clusters the majority class samples using the K-means clustering algorithm and then reduces the number of majority class samples using the center of mass of each cluster to represent the clusters. <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref> compares the number of positive and negative samples before and after processing by the clustering centroid algorithm, and it can be clearly seen that after processing using the clustering centroid algorithm, the number of negative samples and the number of positive samples in the training set and test set are reduced to the same level, which indicates that the clustering centroid algorithm significantly reduces the number of negative samples of ac4C nucleotides.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>PACES dataset before and after imbalance treatment of the number of positive and negative samples in the dataset.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Whether or not imbalance treatment is performed</th>
<th valign="middle" align="center">subdataset</th>
<th valign="middle" align="center">Number of Positive Samples</th>
<th valign="middle" align="center">Number of Negative Samples</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" rowspan="3" align="center">no</td>
<td valign="top" align="center">Training</td>
<td valign="middle" align="center">1160</td>
<td valign="middle" align="center">10855</td>
</tr>
<tr>
<td valign="top" align="center">Testing</td>
<td valign="middle" align="center">469</td>
<td valign="middle" align="center">4343</td>
</tr>
<tr>
<td valign="top" align="center">Total</td>
<td valign="middle" align="center">1629</td>
<td valign="middle" align="center">15198</td>
</tr>
<tr>
<td valign="top" rowspan="3" align="center">yes</td>
<td valign="top" align="center">Training</td>
<td valign="middle" align="center">1148</td>
<td valign="middle" align="center">1148</td>
</tr>
<tr>
<td valign="top" align="center">Testing</td>
<td valign="middle" align="center">467</td>
<td valign="middle" align="center">466</td>
</tr>
<tr>
<td valign="top" align="center">Total</td>
<td valign="middle" align="center">1615</td>
<td valign="middle" align="center">1614</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2_6">
<label>2.6</label>
<title>Measures to prevent overfitting</title>
<p>To prevent the overfitting problem of the model, we used two measures. First, we used Stacking integrated learning, which is a method that combines the prediction results of multiple heterogeneous models to greatly reduce the variance of the model and avoid overfitting. Second, we added L2 regularization to the second layer of the LR model of the Stacking-ac4C model. Through L2 regularization, a &#x201c;regular term&#x201d; is added after the loss function to prevent overfitting of the model. This effectively prevents the model from assigning too much weight to any feature, thus helping to avoid overfitting (<xref ref-type="bibr" rid="B30">30</xref>).</p>
</sec>
<sec id="s2_7">
<label>2.7</label>
<title>Metrics formulation</title>
<p>To fully evaluate the performance of the model, 10-fold cross-validation and independent tests were used to evaluate the performance of the proposed model. In addition, the six metrics for evaluating the performance are specificity (Sp), sensitivity (Sn) (<xref ref-type="bibr" rid="B31">31</xref>, <xref ref-type="bibr" rid="B32">32</xref>), Accuracy (ACC) (<xref ref-type="bibr" rid="B31">31</xref>, <xref ref-type="bibr" rid="B32">32</xref>), correlation coefficient (MCC), the area under the receiver operating characteristic curve (ROC) (<xref ref-type="bibr" rid="B33">33</xref>, <xref ref-type="bibr" rid="B34">34</xref>), and Precision-Recall Curve (PRC) (<xref ref-type="bibr" rid="B35">35</xref>&#x2013;<xref ref-type="bibr" rid="B37">37</xref>), defined as follows:</p>
<disp-formula>
<label>(8)</label>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>c</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>c</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>*</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>*</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>*</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>*</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>*</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>TP, FN, TN, and FP denote true positive, false positive, true negative, and false negative, respectively. Sn and Sp denote model correctness, Acc is used to measure the Accuracy between ac4C and non-ac4C sequences; Mcc is a metric commonly used to evaluate the classification performance of unbalanced data. In addition, since&#xa0;the data is imbalanced with a ratio of 1:10, another visual way to compare the current models is to compare the working characteristic ROC curves. The area under the ROC curve is also an important metric for assessing model performance. The higher the ROC, the better the performance.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results</title>
<sec id="s3_1">
<label>3.1</label>
<title>Classifiers combination</title>
<p>Models such as logistic regression (<xref ref-type="bibr" rid="B38">38</xref>), random forest (<xref ref-type="bibr" rid="B39">39</xref>), KNN (<xref ref-type="bibr" rid="B40">40</xref>), SVM (<xref ref-type="bibr" rid="B41">41</xref>), and neural networks have been experimented with and illustrated in paper XG-ac4C, paper iRNA-ac4C, and paper DeepAc4C; however, the results obtained using these models alone are not satisfactory; therefore, it is necessary to combine these models using a stacking approach. In Stacking Integration, selecting the optimal combination of base classifiers is an effective integrative learning strategy to improve the accuracy and robustness of the models. In this study, we used five standard machine learning algorithms as base classifiers, including logistic regression (Logistic), support vector machine (SVM), random forest (RF), k-nearest neighbor (KNN), and multilayer perceptron (MLP) (<xref ref-type="bibr" rid="B42">42</xref>) algorithms. To elucidate the learning advantage of the present model, we first evaluate the prediction performance of the base model on human AC4C locus data measured with 10-fold cross-validation and metrics mentioned in 3.5. Their optimal parameters are determined by a Bayesian net parameter learning method during the 10-fold cross-validation. This process can be implemented in Python using BaysSearchCV (<xref ref-type="bibr" rid="B43">43</xref>, <xref ref-type="bibr" rid="B44">44</xref>), which tries all combinations of parameter values provided by the user and selects the best values from them. We attempted various parameter settings using the BaysSearchCV method and ultimately determined the optimal parameters. The optimal parameters for these classifiers are explained in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>. Subsequently, we selected five different single classifiers as the base classifier and then generated six combinations of the base classifiers.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>The optimal parameter settings for the Stacking-ac4c base model.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Base model</th>
<th valign="middle" align="center">Best setting</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">LR</td>
<td valign="middle" align="center">random_state=30, max_iter=1100</td>
</tr>
<tr>
<td valign="middle" align="center">SVM</td>
<td valign="middle" align="center">C= 1.6134, kernel=&#x2018;rbf&#x2019;, degree=0.2651,tol=0.078</td>
</tr>
<tr>
<td valign="middle" align="center">KNN</td>
<td valign="middle" align="center">n_neighbors=20, leaf_size=17</td>
</tr>
<tr>
<td valign="middle" align="center">RF</td>
<td valign="middle" align="center">max_depth=10, min_samples_Split=10, min_samples_leaf=1, random_state=30</td>
</tr>
<tr>
<td valign="middle" align="center">MLP</td>
<td valign="middle" align="center">activation=&#x2018;relu&#x2019;, alpha=1e-05, batch_size=37, beta_1 = 0.9,<break/>beta_2 = 0.999, epsilon=1e-08, hidden_layer_sizes=(11), learning_rate_init=0.021, max_iter=8532, momentum=0.58</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>We built an integrated model stacking to integrate all base models to get better results. The base and combined models are evaluated using independent test data. As shown in <xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref>, the average Acc (0.8574) of the Stacking Integration Classifier model is improved by 7.22% compared to the average Acc (0.7852) of the Single Classifier model, which indicates that the Stacking Classifier, has better accuracy than the single classifier. The Stacking Integration Classifier model achieves an average Mcc and an average ROC of (0.71625) and (0.9289), increased by 13.18% and 5.87% more than the Single Classifier model, respectively, which indicates that the Stacking Integration Classifier model is more suitable for handling unbalanced data. The Stacking integrated model obtained better performance than the Single models, indicating that the integration model strategy improved the performance of the models, which may be because the Stacking Integration Classifier model uses different types of models for training, thus fusing the strengths of different models and improving the model&#x2019;s generalization ability.</p>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>Independent test performance comparison between different combinations of base classifiers on unbalanced datasets.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Classification model</th>
<th valign="middle" align="center">Base-Classifier<break/>Combination</th>
<th valign="middle" align="center">Sn</th>
<th valign="middle" align="center">Sp</th>
<th valign="middle" align="center">Acc</th>
<th valign="middle" align="center">Mcc</th>
<th valign="middle" align="center">ROC</th>
<th valign="middle" align="center">PRC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="5" align="center">Single<break/>classifier</td>
<td valign="middle" align="center">(1) LR</td>
<td valign="middle" align="center">0.8148</td>
<td valign="middle" align="center">0.8051</td>
<td valign="middle" align="center">0.8148</td>
<td valign="middle" align="center">0.6297</td>
<td valign="middle" align="center">0.8652</td>
<td valign="middle" align="center">0.8374</td>
</tr>
<tr>
<td valign="middle" align="center">(2) Knn</td>
<td valign="middle" align="center">0.4347</td>
<td valign="middle" align="center">
<bold>0.9315</bold>
</td>
<td valign="middle" align="center">0.6831</td>
<td valign="middle" align="center">0.4219</td>
<td valign="middle" align="center">0.8419</td>
<td valign="middle" align="center">0.8115</td>
</tr>
<tr>
<td valign="middle" align="center">(3) SVM</td>
<td valign="middle" align="center">0.8244</td>
<td valign="middle" align="center">0.7473</td>
<td valign="middle" align="center">0.7859</td>
<td valign="middle" align="center">0.5734</td>
<td valign="middle" align="center">0.8438</td>
<td valign="middle" align="center">0.8092</td>
</tr>
<tr>
<td valign="middle" align="center">(4) RF</td>
<td valign="middle" align="center">0.8672</td>
<td valign="middle" align="center">0.8544</td>
<td valign="middle" align="center">0.8608</td>
<td valign="middle" align="center">0.7217</td>
<td valign="middle" align="center">0.9295</td>
<td valign="middle" align="center">0.9307</td>
</tr>
<tr>
<td valign="middle" align="center">(5) MLP</td>
<td valign="middle" align="center">0.6788</td>
<td valign="middle" align="center">0.8844</td>
<td valign="middle" align="center">0.7816</td>
<td valign="middle" align="center">0.5755</td>
<td valign="middle" align="center">0.8706</td>
<td valign="middle" align="center">0.8532</td>
</tr>
<tr>
<td valign="middle" rowspan="6" align="center">Stacking Integration Classifier</td>
<td valign="middle" align="center">(1)+(2)+(3)+(4)</td>
<td valign="middle" align="center">
<bold>0.8505</bold>
</td>
<td valign="middle" align="center">0.8737</td>
<td valign="middle" align="center">0.8621</td>
<td valign="middle" align="center">0.7303</td>
<td valign="middle" align="center">0.9303</td>
<td valign="middle" align="center">0.9368</td>
</tr>
<tr>
<td valign="middle" align="center">(1)+(2)+(3)+(5)</td>
<td valign="middle" align="center">0.7388</td>
<td valign="middle" align="center">0.8694</td>
<td valign="middle" align="center">0.8041</td>
<td valign="middle" align="center">0.6134</td>
<td valign="middle" align="center">0.8766</td>
<td valign="middle" align="center">0.8514</td>
</tr>
<tr>
<td valign="middle" align="center">(1)+(2)+(4)+(5)</td>
<td valign="middle" align="center">0.8480</td>
<td valign="middle" align="center">0.8801</td>
<td valign="middle" align="center">0.8640</td>
<td valign="middle" align="center">0.7284</td>
<td valign="middle" align="center">0.9400</td>
<td valign="middle" align="center">0.9383</td>
</tr>
<tr>
<td valign="middle" align="center">(1)+(3)+(4)+(5)</td>
<td valign="middle" align="center">0.8501</td>
<td valign="middle" align="center">0.8822</td>
<td valign="middle" align="center">0.8662</td>
<td valign="middle" align="center">0.7327</td>
<td valign="middle" align="center">0.9410</td>
<td valign="middle" align="center">0.9373</td>
</tr>
<tr>
<td valign="middle" align="center">(2)+(3)+(4)+(5)</td>
<td valign="middle" align="center">0.8501</td>
<td valign="middle" align="center">0.8908</td>
<td valign="middle" align="center">0.8704</td>
<td valign="middle" align="center">0.7401</td>
<td valign="middle" align="center">0.9371</td>
<td valign="middle" align="center">0.9303</td>
</tr>
<tr>
<td valign="middle" align="center">All</td>
<td valign="middle" align="center">0.8501</td>
<td valign="middle" align="center">0.9015</td>
<td valign="middle" align="center">
<bold>0.8758</bold>
</td>
<td valign="middle" align="center">
<bold>0.7526</bold>
</td>
<td valign="middle" align="center">
<bold>0.9487</bold>
</td>
<td valign="middle" align="center">
<bold>0.9503</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values indicate the best results.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Sequence composition analysis</title>
<p>To investigate the distribution and preference of nucleotides of ac4C, we used the online tool Weblogo (<xref ref-type="bibr" rid="B45">45</xref>, <xref ref-type="bibr" rid="B46">46</xref>) to mine the conserved motifs of ac4C sequences. <xref ref-type="fig" rid="f3">
<bold>Figures&#xa0;3</bold>
</xref> and <xref ref-type="fig" rid="f4">
<bold>4</bold>
</xref> show the conserved motifs of the ac4C sequence and the distribution and preference of ac4C nucleotides. <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref> is a highly enriched motif (CXX) in the ac4C-containing sequence, similar to the experimental results of Arango et&#xa0;al. (<xref ref-type="bibr" rid="B19">19</xref>). They utilize transcriptome-wide approaches to investigate ac4C localization and function in mRNA. They find that cytidine-containing mRNA codons are enriched in acetylated transcripts compared to other non-acetylated transcripts, which can enhance mRNA translation.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Nucleotide positive sample sequence diagram.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-14-1267755-g003.tif"/>
</fig>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Nucleotide negative sample sequence diagram.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-14-1267755-g004.tif"/>
</fig>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Stacking-ac4c model</title>
<p>The use of multiple heterogeneous models under appropriate integration strategies can achieve complementarity of models under training data compared to integration within a single model, thus significantly improving the reliability and efficiency of the model. In addition, machine learning models such as Logistic, KNN, SVM, RF, and MLP have been used in many bioinformatics fields and have made significant progress. Therefore, in the current study, a heterogeneous inheritance model stacking was used, 10 models were trained, and the simple averaging method was considered as the final result (<xref ref-type="table" rid="T6">
<bold>Table&#xa0;6</bold>
</xref>).</p>
<table-wrap id="T6" position="float">
<label>Table&#xa0;6</label>
<caption>
<p>Performance of the ten models trained on the unbalanced dataset.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="center">Cycle index</th>
<th valign="middle" colspan="4" align="center">Validation results</th>
<th valign="middle" colspan="4" align="center">Independent test results</th>
</tr>    <tr>
<th valign="middle" align="center">Acc</th>
<th valign="middle" align="center">Mcc</th>
<th valign="middle" align="center">ROC</th>
<th valign="middle" align="center">PRC</th>
<th valign="middle" align="center">Acc</th>
<th valign="middle" align="center">Mcc</th>
<th valign="middle" align="center">ROC</th>
<th valign="middle" align="center">PRC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">1</td>
<td valign="top" align="center">0.9000</td>
<td valign="top" align="center">0.8008</td>
<td valign="top" align="center">0.9681</td>
<td valign="top" align="center">0.9702</td>
<td valign="top" align="center">0.8771</td>
<td valign="top" align="center">0.7548</td>
<td valign="top" align="center">0.9488</td>
<td valign="top" align="center">0.9624</td>
</tr>
<tr>
<td valign="top" align="center">2</td>
<td valign="top" align="center">0.8957</td>
<td valign="top" align="center">0.7910</td>
<td valign="top" align="center">0.9520</td>
<td valign="top" align="center">0.9558</td>
<td valign="top" align="center">0.8676</td>
<td valign="top" align="center">0.7153</td>
<td valign="top" align="center">0.9392</td>
<td valign="top" align="center">0.9511</td>
</tr>
<tr>
<td valign="top" align="center">3</td>
<td valign="top" align="center">0.9087</td>
<td valign="top" align="center">0.8157</td>
<td valign="top" align="center">0.9748</td>
<td valign="top" align="center">0.9708</td>
<td valign="top" align="center">0.8831</td>
<td valign="top" align="center">0.7767</td>
<td valign="top" align="center">0.9583</td>
<td valign="top" align="center">0.9688</td>
</tr>
<tr>
<td valign="top" align="center">4</td>
<td valign="top" align="center">0.9087</td>
<td valign="top" align="center">0.8175</td>
<td valign="top" align="center">0.9656</td>
<td valign="top" align="center">0.9622</td>
<td valign="top" align="center">0.8887</td>
<td valign="top" align="center">0.7797</td>
<td valign="top" align="center">0.9553</td>
<td valign="top" align="center">0.9584</td>
</tr>
<tr>
<td valign="top" align="center">5</td>
<td valign="top" align="center">0.9000</td>
<td valign="top" align="center">0.7994</td>
<td valign="top" align="center">0.9564</td>
<td valign="top" align="center">0.9605</td>
<td valign="top" align="center">0.8608</td>
<td valign="top" align="center">0.7486</td>
<td valign="top" align="center">0.9523</td>
<td valign="top" align="center">0.9542</td>
</tr>
<tr>
<td valign="top" align="center">6</td>
<td valign="top" align="center">0.8435</td>
<td valign="top" align="center">0.6869</td>
<td valign="top" align="center">0.9373</td>
<td valign="top" align="center">0.9374</td>
<td valign="top" align="center">0.8830</td>
<td valign="top" align="center">0.7464</td>
<td valign="top" align="center">0.9492</td>
<td valign="top" align="center">0.9396</td>
</tr>
<tr>
<td valign="top" align="center">7</td>
<td valign="top" align="center">0.8734</td>
<td valign="top" align="center">0.7474</td>
<td valign="top" align="center">0.9498</td>
<td valign="top" align="center">0.9353</td>
<td valign="top" align="center">0.8751</td>
<td valign="top" align="center">0.7402</td>
<td valign="top" align="center">0.9509</td>
<td valign="top" align="center">0.9323</td>
</tr>
<tr>
<td valign="top" align="center">8</td>
<td valign="top" align="center">0.8777</td>
<td valign="top" align="center">0.753</td>
<td valign="top" align="center">0.9483</td>
<td valign="top" align="center">0.9571</td>
<td valign="top" align="center">0.8862</td>
<td valign="top" align="center">0.7524</td>
<td valign="top" align="center">0.9472</td>
<td valign="top" align="center">0.9591</td>
</tr>
<tr>
<td valign="top" align="center">9</td>
<td valign="top" align="center">0.8777</td>
<td valign="top" align="center">0.751</td>
<td valign="top" align="center">0.9425</td>
<td valign="top" align="center">0.9288</td>
<td valign="top" align="center">0.8544</td>
<td valign="top" align="center">0.7517</td>
<td valign="top" align="center">0.9389</td>
<td valign="top" align="center">0.9243</td>
</tr>
<tr>
<td valign="top" align="center">10</td>
<td valign="top" align="center">0.8952</td>
<td valign="top" align="center">0.7866</td>
<td valign="top" align="center">0.9463</td>
<td valign="top" align="center">0.9556</td>
<td valign="top" align="center">0.8821</td>
<td valign="top" align="center">0.7603</td>
<td valign="top" align="center">0.9465</td>
<td valign="top" align="center">0.9501</td>
</tr>
<tr>
<td valign="top" align="center">Avg</td>
<td valign="top" align="center">0.8881</td>
<td valign="top" align="center">0.7749</td>
<td valign="top" align="center">0.9541</td>
<td valign="top" align="center">0.9534</td>
<td valign="top" align="center">0.8758</td>
<td valign="top" align="center">0.7526</td>
<td valign="top" align="center">0.9487</td>
<td valign="top" align="center">0.9503</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The Stacking-ac4C model combined Kmer, PseEIIP, and PseKNC as the input of the model, as shown in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>. The base model of the stacking model has SVM, Logistic, KNN, RF, and MLP composition, and the two-layer model meta-learner is LR after the training is completed; the model was validated. <xref ref-type="table" rid="T6">
<bold>Table&#xa0;6</bold>
</xref> shows the Acc values of the model training results set and the model test results set&#x2019;s Acc, Mcc, and ROC values. For the 10 unbalanced training datasets, the maximum Acc (0.9087) was obtained on Cycle 3 and Cycle 4, and the minimum Acc (0.8435) on Cycle 6. The maximum Acc (0.8862) was obtained on Cycle 8 and the minimum Acc (0.8544) on Cycle 9 for the independent test values. The relatively small variance of the training set and the independent test values indicate that the model is stable.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Framework diagram of Stacking ac4C stacked ensemble classifier.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-14-1267755-g005.tif"/>
</fig>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Comparison with published models</title>
<p>In this section, we will compare the proposed model with some existing ac4C site prediction models, namely PACES, XG-ac4C, and DeepAc4C. To validate the robustness and superiority of the proposed Stacking-ac4C, the three existing methods and our method were performed on the independent data set. As shown in <xref ref-type="table" rid="T7">
<bold>Table&#xa0;7</bold>
</xref>, DeepAc4C shows a 7.53% improvement in ROC and a 15.61% improvement in Mcc, indicating that the Stacking-ac4C model has better imbalance data handling capability. Compared to XG-ac4C, the Stacking-ac4C model has an increase in ROC of 5.97% (<xref ref-type="table" rid="T7">
<bold>Table&#xa0;7</bold>
</xref>), indicating that Stacking-ac4C has excellent stability and generalization ability. The low Sn and high Sp of PACES and XG-ac4C models may be due to the fact that the dataset used extracted specific motif sequences and ignored positive samples that did not match the feature. In addition, the reason for the lower Acc and Sp of DeepAc4C compared to XG-ac4C may be that the DeepAc4C model was trained with balanced data, but the test data was 1:10 unbalanced data.</p>
<table-wrap id="T7" position="float">
<label>Table&#xa0;7</label>
<caption>
<p>Results of independent tests of published models on unbalanced data sets.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Tools</th>
<th valign="top" align="center">Sn</th>
<th valign="top" align="center">Sp</th>
<th valign="top" align="center">Acc</th>
<th valign="top" align="center">Mcc</th>
<th valign="top" align="center">ROC</th>
<th valign="top" align="center">PRC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">PACES</td>
<td valign="top" align="center">0.1513</td>
<td valign="top" align="center">0.8920</td>
<td valign="top" align="center">0.8835</td>
<td valign="top" align="center">0.2763</td>
<td valign="top" align="center">0.8741</td>
<td valign="top" align="center">0.4852</td>
</tr>
<tr>
<td valign="top" align="center">XG-ac4C</td>
<td valign="top" align="center">0.5824</td>
<td valign="top" align="center">
<bold>0.9439</bold>
</td>
<td valign="top" align="center">
<bold>0.9045</bold>
</td>
<td valign="top" align="center">0.4918</td>
<td valign="top" align="center">0.8890</td>
<td valign="top" align="center">0.5815</td>
</tr>
<tr>
<td valign="top" align="center">DeepAc4C</td>
<td valign="top" align="center">0.8222</td>
<td valign="top" align="center">0.7734</td>
<td valign="top" align="center">0.7979</td>
<td valign="top" align="center">0.5965</td>
<td valign="top" align="center">0.8734</td>
<td valign="top" align="center">0.8535</td>
</tr>
<tr>
<td valign="top" align="center">Stacking-ac4C</td>
<td valign="top" align="center">
<bold>0.8501</bold>
</td>
<td valign="top" align="center">0.9015</td>
<td valign="top" align="center">0.8758</td>
<td valign="top" align="center">
<bold>0.7526</bold>
</td>
<td valign="top" align="center">
<bold>0.9487</bold>
</td>
<td valign="top" align="center">
<bold>0.9503</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values indicate the best results.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3_5">
<label>3.5</label>
<title>Comparison of results in the second data</title>
<p>To validate the superior generalization ability of our tissue-Specific model compared to existing tools, a dataset was constructed for testing iRNA-ac4C. The ratio of positive to negative samples in this dataset is 1:1. These datasets are available from <ext-link ext-link-type="uri" xlink:href="http://lin-group.cn/server/iRNA-ac4C/">http://lin-group.cn/server/iRNA-ac4C/</ext-link> (<xref ref-type="bibr" rid="B10">10</xref>). We performed independent tests on the four existing methods along with Stacking-ac4c, and the results of the final independent tests are listed in <xref ref-type="table" rid="T8">
<bold>Table&#xa0;8</bold>
</xref>. The corresponding ROC curves are shown in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6C</bold>
</xref>. All models showed relatively high results in Sp but relatively low results in Sn, especially the first three models, which may be due to training on a highly unbalanced dataset, where two models learned more information from negative samples than from positive samples. In addition, compared with iRNA-ac4C, the Sn of Stacking-ac4C increased by 4.1%, which indicates the high sensitivity of the model; the Acc increased by nearly 1%, which indicates a more Accurate model; and the ROC increased by 0.49%, which indicates the superior stability and generalization ability of the Stacking-ac4C model. Meanwhile, the Stacking ac4C model is used in the dataset <ext-link ext-link-type="uri" xlink:href="http://www.rnanut.net/paces/">http://www.rnanut.net/paces/</ext-link> and datasets <ext-link ext-link-type="uri" xlink:href="http://lin-group.cn/server/iRNA-ac4C/">http://lin-group.cn/server/iRNA-ac4C/</ext-link>. The above results are superior to other models, indicating that this model has good generalization ability.</p>
<table-wrap id="T8" position="float">
<label>Table&#xa0;8</label>
<caption>
<p>Results of independent tests of published models on balanced data sets.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Tools</th>
<th valign="top" align="center">Sn</th>
<th valign="top" align="center">Sp</th>
<th valign="top" align="center">Acc</th>
<th valign="top" align="center">Mcc</th>
<th valign="top" align="center">ROC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">PACES</td>
<td valign="top" align="center">0.0598</td>
<td valign="top" align="center">
<bold>1.0000</bold>
</td>
<td valign="top" align="center">0.5299</td>
<td valign="top" align="center">0.1760</td>
<td valign="top" align="center">\</td>
</tr>
<tr>
<td valign="top" align="center">XG-ac4C</td>
<td valign="top" align="center">0.3587</td>
<td valign="top" align="center">0.8243</td>
<td valign="top" align="center">0.5915</td>
<td valign="top" align="center">0.2070</td>
<td valign="top" align="center">\</td>
</tr>
<tr>
<td valign="top" align="center">DeepAc4C</td>
<td valign="top" align="center">0.1007</td>
<td valign="top" align="center">0.9710</td>
<td valign="top" align="center">0.5362</td>
<td valign="top" align="center">0.1470</td>
<td valign="top" align="center">0.8030</td>
</tr>
<tr>
<td valign="top" align="center">iRNA-ac4C</td>
<td valign="top" align="center">0.7670</td>
<td valign="top" align="center">0.8291</td>
<td valign="top" align="center">0.7981</td>
<td valign="top" align="center">0.5970</td>
<td valign="top" align="center">0.8800</td>
</tr>
<tr>
<td valign="top" align="center">Stacking-ac4C</td>
<td valign="top" align="center">
<bold>0.8080</bold>
</td>
<td valign="top" align="center">0.8080</td>
<td valign="top" align="center">
<bold>0.8080</bold>
</td>
<td valign="top" align="center">
<bold>0.6159</bold>
</td>
<td valign="top" align="center">
<bold>0.8835</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values indicate the best results.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Independent test results on the balanced dataset of the basic machine learning model.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-14-1267755-g006.tif"/>
</fig>
<p>To further evaluate the effectiveness of the Stacking architecture, we compared Stacking with five popular traditional machine learning algorithms, including logistic regression (LR), k-nearest neighbor (KNN), support vector machine (SVM), random forest (RF), and multilayer perceptron (MLP) algorithms. For a fair comparison, the models were trained using the dataset constructed from the iRNA-ac4C article and evaluated using an independent test dataset. <xref ref-type="fig" rid="f6">
<bold>Figures&#xa0;6A, B</bold>
</xref> show the Acc and Mcc of five popular traditional machine learning algorithms, and <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6C</bold>
</xref> shows the ROC of five popular traditional machine learning algorithms. This shows that Stacking-ac4C achieves the best scores on Acc, Mcc, and ROC, indicating that compared to traditional classifiers, the proposed model outperforms the ac4C recognition of traditional classifiers.</p>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Summary</title>
<p>Four models were built to identify the ac4C locus in human mRNA. However, there is still room to improve the performance of these predictors. In this study, a new predictor, Stacking-ac4C, is developed, which utilizes three coding methods (i.e., Kmer, PseKNC, and PseEIIP) and uses the Stacking-based algorithm in the classification method to identify ac4C sites. In addition, the results tested in an independent test set of PACES article data showed that Stacking-ac4C outperformed other existing tools. Similarly, the testing results in an independent test set of iRNA-ac4C article data showed that Stacking-ac4C also outperformed other existing tools. The benchmark test dataset and source code can be downloaded from <ext-link ext-link-type="uri" xlink:href="https://github.com/louliliang/ST-ac4C.git">https://github.com/louliliang/ST-ac4C.git</ext-link>. However, the proposed model still has some shortcomings. First, Stacking integrated learning, despite improving the sensitivity of the model in predicting real ac4C loci, lacks the application of a deep learning model on ac4C loci compared to DeepAc4C. Secondly, although the model achieved good results on both PACES and iRNA-ac4C data, the accuracy of the model in the PACES dataset appeared to be insufficient compared to the XG-ac4C model. In addition, a comparison of <xref ref-type="table" rid="T7">
<bold>Tables&#xa0;7</bold>
</xref> and 8 shows that the model is suitable for dealing with unbalanced datasets specific to motifs, and the enhancement effect on balanced baseline datasets is not great. In future work, we will further experiment with other approaches to enable our model to outperform existing predictors. In conclusion, Stacking-ac4C is an effective tool for identifying ac4C sites in mRNA and contributes to our functional understanding of ac4C in RNA.</p>
</sec>
<sec id="s7" sec-type="data-availability">
<title>Data availability statement</title>
<p>Information for existing publicly accessible datasets is contained within the article.</p>
</sec>
<sec id="s8" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>The manuscript presents research on animals that do not require ethical approval for their study.</p>
</sec>
<sec id="s9" sec-type="author-contributions">
<title>Author contributions</title>
<p>L-LL: Writing &#x2013; original draft, Data curation. W-RQ: Conceptualization, Project administration, Supervision, Writing &#x2013; review &amp; editing. ZL: Software, Validation, Writing &#x2013; original draft. Z-CX: Methodology, Validation, Writing &#x2013; review &amp; editing. XX: Funding acquisition, Writing &#x2013; review &amp; editing. S-FH: Supervision, Writing &#x2013; review &amp; editing.</p>
</sec>
</body>
<back>
<sec id="s10" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This work was supported by grants from the National Natural Science Foundation of China (No. 62162032, 62062043,32270789), the Scientific Research Plan of the Department of Education of Jiangxi Province, China (GJJ2201004, GJJ2201038).</p>
</sec>
<sec id="s11" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s12" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s13" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fimmu.2023.1267755/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fimmu.2023.1267755/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet_1.pdf" id="SM1" mimetype="application/pdf"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Boccaletto</surname> <given-names>P</given-names>
</name>
<name>
<surname>Bagi&#x144;ski</surname> <given-names>B</given-names>
</name>
</person-group>. <article-title>MODOMICS: an operational guide to the use of the RNA modification pathways database</article-title>. <source>RNA Bioinformatics</source> (<year>2021</year>) <volume>2284</volume>:<page-range>481&#x2013;505</page-range>. doi: <pub-id pub-id-type="doi">10.1007/978-1-0716-1307-8_26</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>J</given-names>
</name>
<name>
<surname>Pu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Zou</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>F</given-names>
</name>
</person-group>. <article-title>DeepATT: a hybrid category attention neural network for identifying functional effects of DNA sequences</article-title>. <source>Briefings in bioinformatics</source> (<year>2021</year>) <volume>22</volume>:<fpage>bbaa159</fpage>. doi: <pub-id pub-id-type="doi">10.1093/bib/bbaa159</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jin</surname> <given-names>G</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>M</given-names>
</name>
<name>
<surname>Zou</surname> <given-names>M</given-names>
</name>
<name>
<surname>Duan</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>The processing, gene regulation, biological functions, and clinical relevance of N4-acetylcytidine on RNA: a systematic review</article-title>. <source>Molecular Therapy-Nucleic Acids</source> (<year>2020</year>) <volume>20</volume>:<fpage>13</fpage>&#x2013;<lpage>24</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.omtn.2020.01.037</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>W</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Cui</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Y</given-names>
</name>
</person-group>. <article-title>PACES: prediction of N4-acetylcytidine (ac4C) modification sites in mRNA</article-title>. <source>Scientific reports</source> (<year>2019</year>) <volume>9</volume>:<fpage>11112</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41598-019-47594-7</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Azar</surname> <given-names>AT</given-names>
</name>
<name>
<surname>Elshazly</surname> <given-names>HI</given-names>
</name>
<name>
<surname>Hassanien</surname> <given-names>AE</given-names>
</name>
<name>
<surname>Elkorany</surname> <given-names>AM</given-names>
</name>
</person-group>. <article-title>A random forest classifier for lymph diseases</article-title>. <source>Computer methods and programs in biomedicine</source> (<year>2014</year>) <volume>113</volume>:<page-range>465&#x2013;73</page-range>. doi: <pub-id pub-id-type="doi">10.1016/j.cmpb.2013.11.004</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>T</given-names>
</name>
<name>
<surname>He</surname> <given-names>T</given-names>
</name>
<name>
<surname>Benesty</surname> <given-names>M</given-names>
</name>
<name>
<surname>Khotilovich</surname> <given-names>V</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Cho</surname> <given-names>H</given-names>
</name>
<etal/>
</person-group>. <article-title>Xgboost: extreme gradient boosting</article-title>. <source>R package version</source> (<year>2015</year>) <volume>1</volume>:<fpage>1</fpage>&#x2013;<lpage>4</lpage>.</citation>
</ref>
<ref id="B7">
<label>7</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alam</surname> <given-names>W</given-names>
</name>
<name>
<surname>Tayara</surname> <given-names>H</given-names>
</name>
<name>
<surname>Chong</surname> <given-names>KT</given-names>
</name>
</person-group>. <article-title>XG-ac4C: identification of N4-acetylcytidine (ac4C) in mRNA using eXtreme gradient boosting with electron-ion interaction pseudopotentials</article-title>. <source>Scientific reports</source> (<year>2020</year>) <volume>10</volume>:<fpage>1</fpage>&#x2013;<lpage>10</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41598-020-77824-2</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>C</given-names>
</name>
<name>
<surname>Ju</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Zou</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>C</given-names>
</name>
</person-group>. <article-title>DeepAc4C: a convolutional neural network model with hybrid features composed of physicochemical patterns and distributed representation information for identification of N4-acetylcytidine in mRNA</article-title>. <source>Bioinformatics</source> (<year>2022</year>) <volume>38</volume>:<page-range>52&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btab611</pub-id>
</citation>
</ref>
<ref id="B9">
<label>9</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chua</surname> <given-names>LO</given-names>
</name>
<name>
<surname>Roska</surname> <given-names>T</given-names>
</name>
</person-group>. <article-title>S.I.F. Theory, and applications, the CNN paradigm</article-title>. <source>IEEE Transactions on Circuits and Systems I: Fundamental Theory and Applications</source> (<year>1993</year>) <volume>40</volume>:<page-range>147&#x2013;56</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/81.222795</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Su</surname> <given-names>W</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>X-Q</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>X-W</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>D</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>C-Y</given-names>
</name>
<name>
<surname>Zulfiqar</surname> <given-names>H</given-names>
</name>
<etal/>
</person-group>. <article-title>iRNA-ac4C: a novel computational method for effectively detecting N4-acetylcytidine sites in human mRNA</article-title>. <source>International Journal of Biological Macromolecules</source> (<year>2023</year>) <volume>227</volume>:<page-range>1174&#x2013;81</page-range>. doi: <pub-id pub-id-type="doi">10.1016/j.ijbiomac.2022.11.299</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ogunleye</surname> <given-names>A</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Q-G</given-names>
</name>
</person-group>. <article-title>XGBoost model for chronic kidney disease diagnosis</article-title>. <source>IEEE/ACM Transactions on Computational Biology and Bioinformatics</source> (<year>2019</year>) <volume>17</volume>:<page-range>2131&#x2013;40</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TCBB.2019.2911071</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Melsted</surname> <given-names>P</given-names>
</name>
<name>
<surname>Pritchard</surname> <given-names>JK</given-names>
</name>
</person-group>. <article-title>Efficient counting of k-mers in DNA sequences using a bloom filter</article-title>. <source>BMC bioinformatics</source> (<year>2011</year>) <volume>12</volume>:<fpage>1</fpage>&#x2013;<lpage>7</lpage>. doi: <pub-id pub-id-type="doi">10.1186/1471-2105-12-333</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hong-Zhi</surname> <given-names>D</given-names>
</name>
<name>
<surname>Xiao-Ying</surname> <given-names>H</given-names>
</name>
<name>
<surname>Yu-Huan</surname> <given-names>M</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>B-S</given-names>
</name>
<name>
<surname>Da-Hui</surname> <given-names>L</given-names>
</name>
</person-group>. <article-title>Traditional Chinese Medicine: an effective treatment for 2019 novel coronavirus pneumonia (NCP)</article-title>. <source>Chinese Journal of Natural Medicines</source> (<year>2020</year>) <volume>18</volume>:<page-range>206&#x2013;10</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S1875-5364(20)30022-4</pub-id>
</citation>
</ref>
<ref id="B14">
<label>14</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname> <given-names>B</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>L</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>M</given-names>
</name>
<name>
<surname>Li</surname> <given-names>W</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Zhong</surname> <given-names>LJ</given-names>
</name>
</person-group>. <article-title>Advanced separators based on aramid nanofiber (ANF) membranes for lithium-ion batteries: a review of recent progress</article-title>. <source>Journal of Materials Chemistry A</source> (<year>2021</year>) <volume>9</volume>:<page-range>12923&#x2013;46</page-range>. doi: <pub-id pub-id-type="doi">10.1039/D1TA03125B</pub-id>
</citation>
</ref>
<ref id="B15">
<label>15</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yan</surname> <given-names>X</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Intelligent fault diagnosis of rotating machinery using improved multiscale dispersion entropy and mRMR feature selection</article-title>. <source>Knowledge-Based Systems</source> (<year>2019</year>) <volume>163</volume>:<page-range>450&#x2013;71</page-range>. doi: <pub-id pub-id-type="doi">10.1016/j.knosys.2018.09.004</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ting</surname> <given-names>KM</given-names>
</name>
<name>
<surname>Witten</surname> <given-names>IH</given-names>
</name>
</person-group>. <article-title>Stacking bagged and dagged models</article-title>. (<year>1997</year>). doi:&#xa0;<pub-id pub-id-type="doi">10.1109/BIBM.2017.8217729</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Luo</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Su</surname> <given-names>W</given-names>
</name>
<name>
<surname>Lou</surname> <given-names>L</given-names>
</name>
<name>
<surname>Qiu</surname> <given-names>W</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>X</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>Z</given-names>
</name>
</person-group>. <article-title>DLm6Am: A deep-learning-based tool for identifying N6, 2&#x2032;-O-dimethyladenosine sites in RNA sequences</article-title>. <source>International Journal of Molecular Sciences</source> (<year>2022</year>) <volume>23</volume>:<fpage>11026</fpage>. doi: <pub-id pub-id-type="doi">10.3390/ijms231911026</pub-id>
</citation>
</ref>
<ref id="B18">
<label>18</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Luo</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Lou</surname> <given-names>L</given-names>
</name>
<name>
<surname>Qiu</surname> <given-names>W</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>X</given-names>
</name>
</person-group>. <article-title>Predicting N6-methyladenosine sites in multiple tissues of mammals through ensemble deep learning</article-title>. <source>International Journal of Molecular Sciences</source> (<year>2022</year>) <volume>23</volume>:<fpage>15490</fpage>. doi: <pub-id pub-id-type="doi">10.3390/ijms232415490</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arango</surname> <given-names>D</given-names>
</name>
<name>
<surname>Sturgill</surname> <given-names>D</given-names>
</name>
<name>
<surname>Alhusaini</surname> <given-names>N</given-names>
</name>
<name>
<surname>Dillman</surname> <given-names>AA</given-names>
</name>
<name>
<surname>Sweet</surname> <given-names>TJ</given-names>
</name>
<name>
<surname>Hanson</surname> <given-names>G</given-names>
</name>
<etal/>
</person-group>. <article-title>Acetylation of cytidine in mRNA promotes translation efficiency</article-title>. <source>Cell</source> (<year>2018</year>) <volume>175</volume>:<fpage>1872</fpage>&#x2013;<lpage>1886.e24</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.cell.2018.10.030</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fu</surname> <given-names>L</given-names>
</name>
<name>
<surname>Niu</surname> <given-names>B</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>S</given-names>
</name>
<name>
<surname>Li</surname> <given-names>W</given-names>
</name>
</person-group>. <article-title>CD-HIT: accelerated for clustering the next-generation sequencing data</article-title>. <source>Bioinformatics</source> (<year>2012</year>) <volume>28</volume>:<page-range>3150&#x2013;2</page-range>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/bts565</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname> <given-names>J</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>X</given-names>
</name>
<name>
<surname>Qiu</surname> <given-names>W-R</given-names>
</name>
</person-group>. <article-title>iCDI-W2vCom: identifying the Ion channel&#x2013;Drug interaction in cellular networking based on word2vec and node2vec</article-title>. <source>Frontiers in Genetics</source> (<year>2021</year>) <volume>12</volume>:<elocation-id>738274</elocation-id>. doi: <pub-id pub-id-type="doi">10.3389/fgene.2021.738274</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qiu</surname> <given-names>W-R</given-names>
</name>
<name>
<surname>Guan</surname> <given-names>M-Y</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Q-K</given-names>
</name>
<name>
<surname>Lou</surname> <given-names>L-L</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>X</given-names>
</name>
</person-group>. <article-title>Identifying pupylation proteins and sites by incorporating multiple methods</article-title>. <source>Frontiers in Endocrinology</source> (<year>2022</year>) <volume>13</volume>:<elocation-id>849549</elocation-id>. doi: <pub-id pub-id-type="doi">10.3389/fendo.2022.849549</pub-id>
</citation>
</ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guan</surname> <given-names>M-Y</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Q-K</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>P</given-names>
</name>
<name>
<surname>Qiu</surname> <given-names>W-R</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>W-K</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>X</given-names>
</name>
</person-group>. <article-title>Prediction of plant ubiquitylation proteins and sites by fusing multiple features</article-title>. (<year>2022</year>). doi: <pub-id pub-id-type="doi">10.21203/rs.3.rs-2032518/v1</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname> <given-names>J</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>X</given-names>
</name>
<name>
<surname>Qiu</surname> <given-names>W-R</given-names>
</name>
</person-group>. <article-title>DTI-BERT: identifying drug-target interactions in cellular networking based on BERT and deep learning method</article-title>. <source>Frontiers in Genetics</source> (<year>2022</year>) <volume>13</volume>:<elocation-id>1189</elocation-id>. doi: <pub-id pub-id-type="doi">10.3389/fgene.2022.859188</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goldberg</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Levy</surname> <given-names>O</given-names>
</name>
</person-group>. <article-title>word2vec Explained: deriving Mikolov et&#xa0;al.'s negative-sampling word-embedding method</article-title>. <source>arXiv preprint</source> (<year>2014</year>) <volume>arXiv</volume>:<fpage>1402.3722</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1402.3722</pub-id>
</citation>
</ref>
<ref id="B26">
<label>26</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>W</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>J</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>G</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>W</given-names>
</name>
<name>
<surname>Yue</surname> <given-names>X</given-names>
</name>
<name>
<surname>Li</surname> <given-names>D</given-names>
</name>
</person-group>. &#x201c;<article-title>Predicting small RNAs in bacteria via sequence learning ensemble method</article-title>,&#x201d; <conf-name>2017 IEEE International Conference on Bioinformatics and Biomedicine (BIBM)</conf-name>, <conf-loc>Kansas City, MO, USA</conf-loc> (<year>2017</year>), <fpage>643</fpage>&#x2013;<lpage>647</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/BIBM.2017.8217729</pub-id>
</citation>
</ref>
<ref id="B27">
<label>27</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>D</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>L</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>W</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>F</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>F</given-names>
</name>
</person-group>. <article-title>A genetic algorithm-based weighted ensemble method for predicting transposon-derived piRNAs</article-title>. <source>BMC bioinformatics</source> (<year>2016</year>) <volume>17</volume>:<fpage>1</fpage>&#x2013;<lpage>11</lpage>. doi: <pub-id pub-id-type="doi">10.1186/s12859-016-1206-3</pub-id>
</citation>
</ref>
<ref id="B28">
<label>28</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nair</surname> <given-names>AS</given-names>
</name>
<name>
<surname>Sreenadhan</surname> <given-names>SP</given-names>
</name>
</person-group>. <article-title>A coding measure scheme employing electron-ion interaction pseudopotential (EIIP)</article-title>. <source>Bioinformation</source> (<year>2006</year>) <volume>1</volume>:<fpage>197</fpage>.</citation>
</ref>
<ref id="B29">
<label>29</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Mohammed</surname> <given-names>R</given-names>
</name>
<name>
<surname>Rawashdeh</surname> <given-names>J</given-names>
</name>
<name>
<surname>Abdullah</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Machine learning with oversampling and undersampling techniques: overview study and experimental results</article-title>, in: <conf-name>2020 11th International Conference on Information and Communication Systems (ICICS)</conf-name>, <conf-loc>Irbid, Jordan</conf-loc>. (<year>2020</year>), <page-range>243&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICICS49469.2020.239556</pub-id>. <publisher-name>IEEE</publisher-name>.</citation>
</ref>
<ref id="B30">
<label>30</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Van Laarhoven</surname> <given-names>T</given-names>
</name>
</person-group>. <article-title>L2 regularization versus batch and weight normalization</article-title>. <source>arXiv preprint</source> (<year>2017</year>) <fpage>arXiv:1706.05350</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1706.05350</pub-id>
</citation>
</ref>
<ref id="B31">
<label>31</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yuan</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>K</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Le</surname> <given-names>NQK</given-names>
</name>
<name>
<surname>Chua</surname> <given-names>MCH</given-names>
</name>
</person-group>. <article-title>Prediction of anticancer peptides based on an ensemble model of deep learning and machine learning using ordinal positional encoding</article-title>. <source>Briefings in Bioinformatics</source> (<year>2023</year>) <volume>24</volume>:<fpage>bbac630</fpage>. doi: <pub-id pub-id-type="doi">10.1093/bib/bbac630</pub-id>
</citation>
</ref>
<ref id="B32">
<label>32</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kha</surname> <given-names>Q-H</given-names>
</name>
<name>
<surname>Ho</surname> <given-names>Q-T</given-names>
</name>
<name>
<surname>Le</surname> <given-names>NQK</given-names>
</name>
</person-group>. <article-title>Identifying SNARE proteins using an alignment-free method based on multiscan convolutional neural network and PSSM profiles</article-title>. <source>Journal of Chemical Information and Modeling</source> (<year>2022</year>) <volume>62</volume>:<page-range>4820&#x2013;6</page-range>. doi: <pub-id pub-id-type="doi">10.1021/acs.jcim.2c01034</pub-id>
</citation>
</ref>
<ref id="B33">
<label>33</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jia</surname> <given-names>J</given-names>
</name>
<name>
<surname>Lei</surname> <given-names>R</given-names>
</name>
<name>
<surname>Qin</surname> <given-names>L</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>G</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>X</given-names>
</name>
</person-group>. <article-title>iEnhancer-DCSV: Predicting enhancers and their strength based on DenseNet and improved convolutional block attention module</article-title>. <source>Frontiers in Genetics</source> (<year>2023</year>) <volume>14</volume>:<elocation-id>1132018</elocation-id>. doi: <pub-id pub-id-type="doi">10.3389/fgene.2023.1132018</pub-id>
</citation>
</ref>
<ref id="B34">
<label>34</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qiu</surname> <given-names>W-R</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>B-Q</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>X</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>Z-C</given-names>
</name>
<name>
<surname>Chou</surname> <given-names>K-C</given-names>
</name>
</person-group>. <article-title>iPTM-mLys: identifying multiple lysine PTM sites and their different types</article-title>. <source>Bioinformatics</source> (<year>2016</year>) <volume>32</volume>:<page-range>3116&#x2013;23</page-range>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btw380</pub-id>
</citation>
</ref>
<ref id="B35">
<label>35</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qiu</surname> <given-names>W-R</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Q-K</given-names>
</name>
<name>
<surname>Guan</surname> <given-names>M-Y</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>J-H</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>X</given-names>
</name>
</person-group>. <article-title>Predicting S-nitrosylation proteins and sites by fusing multiple features</article-title>. <source>Mathematical Biosciences and Engineering</source> (<year>2021</year>) <volume>18</volume>:<page-range>9132&#x2013;47</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.3934/mbe.2021450</pub-id>
</citation>
</ref>
<ref id="B36">
<label>36</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ren</surname> <given-names>L</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Ning</surname> <given-names>L</given-names>
</name>
<name>
<surname>Pan</surname> <given-names>X</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>Q</given-names>
</name>
<etal/>
</person-group>. <article-title>TCM2COVID: A resource of anti-COVID-19 traditional Chinese medicine with effects and mechanisms</article-title>. <source>Imeta</source> (<year>2022</year>) <volume>1</volume>(<issue>4</issue>):<fpage>e42</fpage>. doi: <pub-id pub-id-type="doi">10.1002/imt2.42</pub-id>
</citation>
</ref>
<ref id="B37">
<label>37</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dong</surname> <given-names>L</given-names>
</name>
<name>
<surname>Jin</surname> <given-names>X</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>W</given-names>
</name>
<name>
<surname>Ye</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Li</surname> <given-names>W</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>S</given-names>
</name>
<etal/>
</person-group>. <article-title>Distinct clinical phenotype and genetic testing strategy for Lynch syndrome in China based on a large colorectal cancer cohort</article-title>. <source>Int J Cancer</source> (<year>2020</year>) <volume>146</volume>:<page-range>3077&#x2013;86</page-range>. doi: <pub-id pub-id-type="doi">10.1002/ijc.32914</pub-id>
</citation>
</ref>
<ref id="B38">
<label>38</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>LaValley</surname> <given-names>MPJC</given-names>
</name>
</person-group>. <article-title>Logistic regression</article-title>. <source>Circulation</source> (<year>2008</year>) <volume>117</volume>(<issue>18</issue>):<page-range>2395&#x2013;9</page-range>. doi: <pub-id pub-id-type="doi">10.1161/CIRCULATIONAHA.106.682658</pub-id>
</citation>
</ref>
<ref id="B39">
<label>39</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qiu</surname> <given-names>W-R</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>B-Q</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>X</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>Z-C</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>J-H</given-names>
</name>
<name>
<surname>Chou</surname> <given-names>K-C</given-names>
</name>
</person-group>. <article-title>iKcr-PseEns: Identify lysine crotonylation sites in histone proteins with pseudo components and ensemble classifier</article-title>. <source>Genomics</source> (<year>2018</year>) <volume>110</volume>:<page-range>239&#x2013;46</page-range>. doi: <pub-id pub-id-type="doi">10.1016/j.ygeno.2017.10.008</pub-id>
</citation>
</ref>
<ref id="B40">
<label>40</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Guo</surname> <given-names>G</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Bell</surname> <given-names>D</given-names>
</name>
<name>
<surname>Bi</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Greer</surname> <given-names>K</given-names>
</name>
</person-group>. <article-title>KNN model-based approach in classification</article-title>, in: <source>On The Move to Meaningful Internet Systems 2003: CoopIS, DOA, and ODBASE. OTM 2003. Lecture Notes in Computer Science</source>. <publisher-name>Springer</publisher-name>, <publisher-loc>Berlin, Heidelberg</publisher-loc> (<year>2003</year>), <page-range>986&#x2013;96</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-540-39964-3_62</pub-id>
</citation>
</ref>
<ref id="B41">
<label>41</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qiu</surname> <given-names>W-R</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>B-Q</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>H</given-names>
</name>
</person-group>. <article-title>Identify and analysis crotonylation sites in histone by using support vector machines</article-title>. <source>Artificial intelligence in medicine</source> (<year>2017</year>) <volume>83</volume>:<fpage>75</fpage>&#x2013;<lpage>81</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.artmed.2017.02.007</pub-id>
</citation>
</ref>
<ref id="B42">
<label>42</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pinkus</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Approximation theory of the MLP model in neural networks</article-title>. <source>Acta numerica</source> (<year>1999</year>) <volume>8</volume>:<page-range>143&#x2013;95</page-range>. doi: <pub-id pub-id-type="doi">10.1017/S0962492900002919</pub-id>
</citation>
</ref>
<ref id="B43">
<label>43</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Porter</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Studying the acquisition function of bayesian optimization with machine learning with DNA reads</article-title>.</citation>
</ref>
<ref id="B44">
<label>44</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Laverty</surname> <given-names>KU</given-names>
</name>
<name>
<surname>Jolma</surname> <given-names>A</given-names>
</name>
<name>
<surname>Pour</surname> <given-names>SE</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>H</given-names>
</name>
<name>
<surname>Ray</surname> <given-names>D</given-names>
</name>
<name>
<surname>Morris</surname> <given-names>Q</given-names>
</name>
<etal/>
</person-group>. <article-title>PRIESSTESS: interpretable, high-performing models of the sequence and structure preferences of RNA-binding proteins</article-title>. <source>Nucleic Acids Research</source> (<year>2022</year>) <volume>50</volume>:<page-range>e111&#x2013;1</page-range>. doi: <pub-id pub-id-type="doi">10.1093/nar/gkac694</pub-id>
</citation>
</ref>
<ref id="B45">
<label>45</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Crooks</surname> <given-names>GE</given-names>
</name>
<name>
<surname>Hon</surname> <given-names>G</given-names>
</name>
<name>
<surname>Chandonia</surname> <given-names>J-M</given-names>
</name>
<name>
<surname>Brenner</surname> <given-names>SE</given-names>
</name>
</person-group>. <article-title>WebLogo: a sequence logo generator</article-title>. <source>Genome research</source> (<year>2004</year>) <volume>14</volume>:<page-range>1188&#x2013;90</page-range>. <uri xlink:href="http://www.genome.org/cgi/doi/10.1101/gr.849004">http://www.genome.org/cgi/doi/10.1101/gr.849004</uri>
</citation>
</ref>
<ref id="B46">
<label>46</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Crooks</surname> <given-names>GE</given-names>
</name>
</person-group>. <source>WebLogo, Lawrence Berkeley National Lab</source>. <publisher-loc>Berkeley, CA (United States</publisher-loc>: <publisher-name>LBNL</publisher-name> (<year>2003</year>).</citation>
</ref>
</ref-list>
</back>
</article>